From 76cdcd19895d8823648894692301eb12b96a2094 Mon Sep 17 00:00:00 2001
From: AlexCuadron <alex.cl.2000@gmail.com>
Date: Tue, 15 Oct 2024 17:31:15 +0200
Subject: [PATCH 01/18] Updated tests

---
 tests/unit/test_runtime_build.py | 2 ++
 1 file changed, 2 insertions(+)

diff --git a/tests/unit/test_runtime_build.py b/tests/unit/test_runtime_build.py
index a8d215544303..e1e1c18eec94 100644
--- a/tests/unit/test_runtime_build.py
+++ b/tests/unit/test_runtime_build.py
@@ -259,6 +259,7 @@ def test_build_runtime_image_from_scratch(temp_dir):
                 f'{get_runtime_image_repo()}:{from_scratch_hash}',
                 f'{get_runtime_image_repo()}:{OH_VERSION}_image_debian_tag_11',
             ],
+            platform='linux/amd64',  # Added platform tag
         )
         assert image_name == f'{get_runtime_image_repo()}:{from_scratch_hash}'
 
@@ -340,6 +341,7 @@ def test_build_runtime_image_exact_hash_not_exist(mock_build_sandbox_image, temp
                 target_image_repo=repo,
                 target_image_hash_tag=from_scratch_hash,
                 target_image_tag=latest_image_tag,
+                platform='linux/amd64',  # Added platform argument
             )
             assert image_name == f'{repo}:{from_scratch_hash}'
 

From 3beaf5c02d97e5d01694899b3f9a1a1c2bbf0571 Mon Sep 17 00:00:00 2001
From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com>
Date: Tue, 15 Oct 2024 18:59:26 +0200
Subject: [PATCH 02/18] chore(deps): bump litellm from 1.49.3 to 1.49.4 (#4406)

Signed-off-by: dependabot[bot] <support@github.com>
Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com>
---
 poetry.lock | 6 +++---
 1 file changed, 3 insertions(+), 3 deletions(-)

diff --git a/poetry.lock b/poetry.lock
index 97457465f380..4fe8048b7fd1 100644
--- a/poetry.lock
+++ b/poetry.lock
@@ -3879,13 +3879,13 @@ types-tqdm = "*"
 
 [[package]]
 name = "litellm"
-version = "1.49.3"
+version = "1.49.4"
 description = "Library to easily interface with LLM API providers"
 optional = false
 python-versions = "!=2.7.*,!=3.0.*,!=3.1.*,!=3.2.*,!=3.3.*,!=3.4.*,!=3.5.*,!=3.6.*,!=3.7.*,>=3.8"
 files = [
-    {file = "litellm-1.49.3-py3-none-any.whl", hash = "sha256:300c3c9e1600441f8b6d3afe0fd79c6193f901b2091f3730883ffe3709eebfa2"},
-    {file = "litellm-1.49.3.tar.gz", hash = "sha256:e51ce30286894803dcf2949ddb4aab5c2e00809694a48ce6e997953566113c0b"},
+    {file = "litellm-1.49.4-py3-none-any.whl", hash = "sha256:3094a9f74979da993f4b3298372ec4416f7a3f82d11a0831c9c616098b3fb50a"},
+    {file = "litellm-1.49.4.tar.gz", hash = "sha256:5f16d40bfa7747fcc21f45f340454c57cbc705178244fe7326abac7c0759e05e"},
 ]
 
 [package.dependencies]

From c8db8aaf92d817c4ac7f867c5ee0edbbfe280276 Mon Sep 17 00:00:00 2001
From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com>
Date: Tue, 15 Oct 2024 19:05:33 +0200
Subject: [PATCH 03/18] chore(deps-dev): bump llama-index from 0.11.17 to
 0.11.18 (#4408)

Signed-off-by: dependabot[bot] <support@github.com>
Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com>
---
 poetry.lock | 14 +++++++-------
 1 file changed, 7 insertions(+), 7 deletions(-)

diff --git a/poetry.lock b/poetry.lock
index 4fe8048b7fd1..73b6b1c394b1 100644
--- a/poetry.lock
+++ b/poetry.lock
@@ -3922,19 +3922,19 @@ pydantic = ">=1.10"
 
 [[package]]
 name = "llama-index"
-version = "0.11.17"
+version = "0.11.18"
 description = "Interface between LLMs and your data"
 optional = false
 python-versions = "<4.0,>=3.8.1"
 files = [
-    {file = "llama_index-0.11.17-py3-none-any.whl", hash = "sha256:85a3d2cd1908181555ae926f880dfae5284b24fda6b866e60969a44302d87350"},
-    {file = "llama_index-0.11.17.tar.gz", hash = "sha256:b2176f400b33cd765e86775724d8ad5c6e4812ecdc36f2ca4500edcadc015af4"},
+    {file = "llama_index-0.11.18-py3-none-any.whl", hash = "sha256:dc54c7fdd4c8ee32aa0c5565038894295fc76bd95e21e70fa67ca6fb2413a1b3"},
+    {file = "llama_index-0.11.18.tar.gz", hash = "sha256:5c43b46ea9957d539ad823e008c9b6957fbaf4ec5c8bc6903accfb19863edfd9"},
 ]
 
 [package.dependencies]
 llama-index-agent-openai = ">=0.3.4,<0.4.0"
 llama-index-cli = ">=0.3.1,<0.4.0"
-llama-index-core = ">=0.11.17,<0.12.0"
+llama-index-core = ">=0.11.18,<0.12.0"
 llama-index-embeddings-openai = ">=0.2.4,<0.3.0"
 llama-index-indices-managed-llama-cloud = ">=0.3.0"
 llama-index-legacy = ">=0.9.48,<0.10.0"
@@ -3980,13 +3980,13 @@ llama-index-llms-openai = ">=0.2.0,<0.3.0"
 
 [[package]]
 name = "llama-index-core"
-version = "0.11.17"
+version = "0.11.18"
 description = "Interface between LLMs and your data"
 optional = false
 python-versions = "<4.0,>=3.8.1"
 files = [
-    {file = "llama_index_core-0.11.17-py3-none-any.whl", hash = "sha256:d65565b54ea55b2db12f9a1cd5c250b770d7e43d3363137cff431a6116ef069c"},
-    {file = "llama_index_core-0.11.17.tar.gz", hash = "sha256:1143baf8d819e27555bdb142abdf2833d3d37731f270f46fa1e07fc4b97116ae"},
+    {file = "llama_index_core-0.11.18-py3-none-any.whl", hash = "sha256:8e57522e69d3c8a219b29b5f1624c20269c9c3f87729eff9ecfb796eab51dd55"},
+    {file = "llama_index_core-0.11.18.tar.gz", hash = "sha256:f94ae8d740b65c3bf0bc0422b0210613664c1a9f8e98b7328e037a68255bed83"},
 ]
 
 [package.dependencies]

From 308dc62546ddad4a70bcd1c86841801b02071ed0 Mon Sep 17 00:00:00 2001
From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com>
Date: Tue, 15 Oct 2024 19:06:40 +0200
Subject: [PATCH 04/18] chore(deps): bump modal from 0.64.181 to 0.64.182
 (#4407)

Signed-off-by: dependabot[bot] <support@github.com>
Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com>
---
 poetry.lock | 4 ++--
 1 file changed, 2 insertions(+), 2 deletions(-)

diff --git a/poetry.lock b/poetry.lock
index 73b6b1c394b1..f396e504b4f4 100644
--- a/poetry.lock
+++ b/poetry.lock
@@ -4799,12 +4799,12 @@ type = ["mypy (==1.11.2)"]
 
 [[package]]
 name = "modal"
-version = "0.64.181"
+version = "0.64.182"
 description = "Python client library for Modal"
 optional = false
 python-versions = ">=3.8"
 files = [
-    {file = "modal-0.64.181-py3-none-any.whl", hash = "sha256:1d8dab39029abd2b11ca44b8401c71407b20e0e9a2e5c673ec6b1476d3b17fa2"},
+    {file = "modal-0.64.182-py3-none-any.whl", hash = "sha256:d3213550a0724b13b1dacf8b468d26c78f51d850fd2a76529f180921905bcad3"},
 ]
 
 [package.dependencies]

From 158a9230b0d89eaab448c5d371ab9f3516887550 Mon Sep 17 00:00:00 2001
From: Xingyao Wang <xingyao@all-hands.dev>
Date: Tue, 15 Oct 2024 14:31:49 -0500
Subject: [PATCH 05/18] refactor: move get_pairs from memory to shared utils
 (#4411)

---
 openhands/events/utils.py   | 56 +++++++++++++++++++++++++++++++++++++
 openhands/memory/history.py | 54 +++--------------------------------
 tests/unit/test_is_stuck.py | 23 +++++++++++++--
 3 files changed, 81 insertions(+), 52 deletions(-)
 create mode 100644 openhands/events/utils.py

diff --git a/openhands/events/utils.py b/openhands/events/utils.py
new file mode 100644
index 000000000000..6c8cc415f675
--- /dev/null
+++ b/openhands/events/utils.py
@@ -0,0 +1,56 @@
+from openhands.core.logger import openhands_logger as logger
+from openhands.events.action.action import Action
+from openhands.events.action.empty import NullAction
+from openhands.events.event import Event
+from openhands.events.observation.commands import CmdOutputObservation
+from openhands.events.observation.empty import NullObservation
+from openhands.events.observation.observation import Observation
+
+
+def get_pairs_from_events(events: list[Event]) -> list[tuple[Action, Observation]]:
+    """Return the history as a list of tuples (action, observation)."""
+    tuples: list[tuple[Action, Observation]] = []
+    action_map: dict[int, Action] = {}
+    observation_map: dict[int, Observation] = {}
+
+    # runnable actions are set as cause of observations
+    # (MessageAction, NullObservation) for source=USER
+    # (MessageAction, NullObservation) for source=AGENT
+    # (other_action?, NullObservation)
+    # (NullAction, CmdOutputObservation) background CmdOutputObservations
+
+    for event in events:
+        if event.id is None or event.id == -1:
+            logger.debug(f'Event {event} has no ID')
+
+        if isinstance(event, Action):
+            action_map[event.id] = event
+
+        if isinstance(event, Observation):
+            if event.cause is None or event.cause == -1:
+                logger.debug(f'Observation {event} has no cause')
+
+            if event.cause is None:
+                # runnable actions are set as cause of observations
+                # NullObservations have no cause
+                continue
+
+            observation_map[event.cause] = event
+
+    for action_id, action in action_map.items():
+        observation = observation_map.get(action_id)
+        if observation:
+            # observation with a cause
+            tuples.append((action, observation))
+        else:
+            tuples.append((action, NullObservation('')))
+
+    for cause_id, observation in observation_map.items():
+        if cause_id not in action_map:
+            if isinstance(observation, NullObservation):
+                continue
+            if not isinstance(observation, CmdOutputObservation):
+                logger.debug(f'Observation {observation} has no cause')
+            tuples.append((NullAction(), observation))
+
+    return tuples.copy()
diff --git a/openhands/memory/history.py b/openhands/memory/history.py
index 89e50d67e455..1e4cfb8b5f05 100644
--- a/openhands/memory/history.py
+++ b/openhands/memory/history.py
@@ -10,12 +10,12 @@
 from openhands.events.action.message import MessageAction
 from openhands.events.event import Event, EventSource
 from openhands.events.observation.agent import AgentStateChangedObservation
-from openhands.events.observation.commands import CmdOutputObservation
 from openhands.events.observation.delegate import AgentDelegateObservation
 from openhands.events.observation.empty import NullObservation
 from openhands.events.observation.observation import Observation
 from openhands.events.serialization.event import event_to_dict
 from openhands.events.stream import EventStream
+from openhands.events.utils import get_pairs_from_events
 
 
 class ShortTermHistory(list[Event]):
@@ -216,55 +216,9 @@ def on_event(self, event: Event):
     def compatibility_for_eval_history_pairs(self) -> list[tuple[dict, dict]]:
         history_pairs = []
 
-        for action, observation in self.get_pairs():
+        for action, observation in get_pairs_from_events(
+            self.get_events_as_list(include_delegates=True)
+        ):
             history_pairs.append((event_to_dict(action), event_to_dict(observation)))
 
         return history_pairs
-
-    def get_pairs(self) -> list[tuple[Action, Observation]]:
-        """Return the history as a list of tuples (action, observation)."""
-        tuples: list[tuple[Action, Observation]] = []
-        action_map: dict[int, Action] = {}
-        observation_map: dict[int, Observation] = {}
-
-        # runnable actions are set as cause of observations
-        # (MessageAction, NullObservation) for source=USER
-        # (MessageAction, NullObservation) for source=AGENT
-        # (other_action?, NullObservation)
-        # (NullAction, CmdOutputObservation) background CmdOutputObservations
-
-        for event in self.get_events_as_list(include_delegates=True):
-            if event.id is None or event.id == -1:
-                logger.debug(f'Event {event} has no ID')
-
-            if isinstance(event, Action):
-                action_map[event.id] = event
-
-            if isinstance(event, Observation):
-                if event.cause is None or event.cause == -1:
-                    logger.debug(f'Observation {event} has no cause')
-
-                if event.cause is None:
-                    # runnable actions are set as cause of observations
-                    # NullObservations have no cause
-                    continue
-
-                observation_map[event.cause] = event
-
-        for action_id, action in action_map.items():
-            observation = observation_map.get(action_id)
-            if observation:
-                # observation with a cause
-                tuples.append((action, observation))
-            else:
-                tuples.append((action, NullObservation('')))
-
-        for cause_id, observation in observation_map.items():
-            if cause_id not in action_map:
-                if isinstance(observation, NullObservation):
-                    continue
-                if not isinstance(observation, CmdOutputObservation):
-                    logger.debug(f'Observation {observation} has no cause')
-                tuples.append((NullAction(), observation))
-
-        return tuples.copy()
diff --git a/tests/unit/test_is_stuck.py b/tests/unit/test_is_stuck.py
index 5e23a849286b..4a1330752161 100644
--- a/tests/unit/test_is_stuck.py
+++ b/tests/unit/test_is_stuck.py
@@ -17,6 +17,7 @@
 from openhands.events.observation.empty import NullObservation
 from openhands.events.observation.error import ErrorObservation
 from openhands.events.stream import EventSource, EventStream
+from openhands.events.utils import get_pairs_from_events
 from openhands.memory.history import ShortTermHistory
 from openhands.storage import get_file_store
 
@@ -170,7 +171,16 @@ def test_is_stuck_repeating_action_observation(
 
         assert len(collect_events(event_stream)) == 10
         assert len(list(stuck_detector.state.history.get_events())) == 8
-        assert len(stuck_detector.state.history.get_pairs()) == 5
+        assert (
+            len(
+                get_pairs_from_events(
+                    stuck_detector.state.history.get_events_as_list(
+                        include_delegates=True
+                    )
+                )
+            )
+            == 5
+        )
 
         assert stuck_detector.is_stuck() is False
         assert stuck_detector.state.almost_stuck == 1
@@ -186,7 +196,16 @@ def test_is_stuck_repeating_action_observation(
 
         assert len(collect_events(event_stream)) == 12
         assert len(list(stuck_detector.state.history.get_events())) == 10
-        assert len(stuck_detector.state.history.get_pairs()) == 6
+        assert (
+            len(
+                get_pairs_from_events(
+                    stuck_detector.state.history.get_events_as_list(
+                        include_delegates=True
+                    )
+                )
+            )
+            == 6
+        )
 
         with patch('logging.Logger.warning') as mock_warning:
             assert stuck_detector.is_stuck() is True

From b6a916342736dae8907179290f04d5116f072814 Mon Sep 17 00:00:00 2001
From: mamoodi <mamoodiha@gmail.com>
Date: Tue, 15 Oct 2024 18:45:08 -0400
Subject: [PATCH 06/18] Fix eval output path in case of @ char (#4416)

---
 evaluation/utils/shared.py | 2 +-
 1 file changed, 1 insertion(+), 1 deletion(-)

diff --git a/evaluation/utils/shared.py b/evaluation/utils/shared.py
index bed679f342a2..d184b5b98037 100644
--- a/evaluation/utils/shared.py
+++ b/evaluation/utils/shared.py
@@ -152,7 +152,7 @@ def make_metadata(
     details: dict[str, Any] | None = None,
 ) -> EvalMetadata:
     model_name = llm_config.model.split('/')[-1]
-    model_path = model_name.replace(':', '_')
+    model_path = model_name.replace(':', '_').replace('@', '-')
     eval_note = f'_N_{eval_note}' if eval_note else ''
 
     eval_output_path = os.path.join(

From 8ba531a012eefe824af8ece6a505c4913dfe48cc Mon Sep 17 00:00:00 2001
From: tofarr <tofarr@gmail.com>
Date: Tue, 15 Oct 2024 17:52:21 -0600
Subject: [PATCH 07/18] Fix for lockup - create the runtime in a background
 thread (#4412)

Co-authored-by: Robert Brennan <contact@rbren.io>
---
 openhands/runtime/runtime.py              |  4 +-
 openhands/security/invariant/analyzer.py  |  4 +-
 openhands/server/listen.py                | 14 +++---
 openhands/server/session/agent_session.py |  9 +++-
 openhands/utils/async_utils.py            | 16 ++++++-
 tests/unit/test_async_utils.py            | 54 +++++++++++++++++------
 6 files changed, 74 insertions(+), 27 deletions(-)

diff --git a/openhands/runtime/runtime.py b/openhands/runtime/runtime.py
index 7e420643c347..44614ee0a3dc 100644
--- a/openhands/runtime/runtime.py
+++ b/openhands/runtime/runtime.py
@@ -28,7 +28,7 @@
 )
 from openhands.events.serialization.action import ACTION_TYPE_TO_CLASS
 from openhands.runtime.plugins import JupyterRequirement, PluginRequirement
-from openhands.utils.async_utils import sync_from_async
+from openhands.utils.async_utils import call_sync_from_async
 
 
 def _default_env_vars(sandbox_config: SandboxConfig) -> dict[str, str]:
@@ -123,7 +123,7 @@ async def on_event(self, event: Event) -> None:
             if event.timeout is None:
                 event.timeout = self.config.sandbox.timeout
             assert event.timeout is not None
-            observation = await sync_from_async(self.run_action, event)
+            observation = await call_sync_from_async(self.run_action, event)
             observation._cause = event.id  # type: ignore[attr-defined]
             source = event.source if event.source else EventSource.AGENT
             await self.event_stream.async_add_event(observation, source)  # type: ignore[arg-type]
diff --git a/openhands/security/invariant/analyzer.py b/openhands/security/invariant/analyzer.py
index 275888bb4197..9d8b280716a7 100644
--- a/openhands/security/invariant/analyzer.py
+++ b/openhands/security/invariant/analyzer.py
@@ -19,7 +19,7 @@
 from openhands.security.analyzer import SecurityAnalyzer
 from openhands.security.invariant.client import InvariantClient
 from openhands.security.invariant.parser import TraceElement, parse_element
-from openhands.utils.async_utils import sync_from_async
+from openhands.utils.async_utils import call_sync_from_async
 
 
 class InvariantAnalyzer(SecurityAnalyzer):
@@ -146,7 +146,7 @@ async def confirm(self, event: Event) -> None:
             {'action': 'change_agent_state', 'args': {'agent_state': 'user_confirmed'}}
         )
         event_source = event.source if event.source else EventSource.AGENT
-        await sync_from_async(self.event_stream.add_event, new_event, event_source)
+        await call_sync_from_async(self.event_stream.add_event, new_event, event_source)
 
     async def security_risk(self, event: Action) -> ActionSecurityRisk:
         logger.info('Calling security_risk on InvariantAnalyzer')
diff --git a/openhands/server/listen.py b/openhands/server/listen.py
index d7f177734971..32c93a117e23 100644
--- a/openhands/server/listen.py
+++ b/openhands/server/listen.py
@@ -14,7 +14,7 @@
 from openhands.security.options import SecurityAnalyzers
 from openhands.server.data_models.feedback import FeedbackDataModel, store_feedback
 from openhands.storage import get_file_store
-from openhands.utils.async_utils import sync_from_async
+from openhands.utils.async_utils import call_sync_from_async
 
 with warnings.catch_warnings():
     warnings.simplefilter('ignore')
@@ -211,8 +211,8 @@ async def attach_session(request: Request, call_next):
             content={'error': 'Invalid token'},
         )
 
-    request.state.conversation = session_manager.attach_to_conversation(
-        request.state.sid
+    request.state.conversation = await call_sync_from_async(
+        session_manager.attach_to_conversation, request.state.sid
     )
     if request.state.conversation is None:
         return JSONResponse(
@@ -441,7 +441,9 @@ async def list_files(request: Request, path: str | None = None):
         )
 
     runtime: Runtime = request.state.conversation.runtime
-    file_list = await sync_from_async(runtime.list_files, path)
+    file_list = await asyncio.create_task(
+        call_sync_from_async(runtime.list_files, path)
+    )
     if path:
         file_list = [os.path.join(path, f) for f in file_list]
 
@@ -490,7 +492,7 @@ async def select_file(file: str, request: Request):
 
     file = os.path.join(runtime.config.workspace_mount_path_in_sandbox, file)
     read_action = FileReadAction(file)
-    observation = await sync_from_async(runtime.run_action, read_action)
+    observation = await call_sync_from_async(runtime.run_action, read_action)
 
     if isinstance(observation, FileReadObservation):
         content = observation.content
@@ -687,7 +689,7 @@ async def save_file(request: Request):
             runtime.config.workspace_mount_path_in_sandbox, file_path
         )
         write_action = FileWriteAction(file_path, content)
-        observation = await sync_from_async(runtime.run_action, write_action)
+        observation = await call_sync_from_async(runtime.run_action, write_action)
 
         if isinstance(observation, FileWriteObservation):
             return JSONResponse(
diff --git a/openhands/server/session/agent_session.py b/openhands/server/session/agent_session.py
index f172021a37d2..6bc442ac731a 100644
--- a/openhands/server/session/agent_session.py
+++ b/openhands/server/session/agent_session.py
@@ -14,6 +14,7 @@
 from openhands.runtime.runtime import Runtime
 from openhands.security import SecurityAnalyzer, options
 from openhands.storage.files import FileStore
+from openhands.utils.async_utils import call_sync_from_async
 
 
 class AgentSession:
@@ -102,7 +103,13 @@ async def _start(
     ):
         self.loop = asyncio.get_running_loop()
         self._create_security_analyzer(config.security.security_analyzer)
-        self._create_runtime(runtime_name, config, agent, status_message_callback)
+        await call_sync_from_async(
+            self._create_runtime,
+            runtime_name=runtime_name,
+            config=config,
+            agent=agent,
+            status_message_callback=status_message_callback,
+        )
         self._create_controller(
             agent,
             config.security.confirmation_mode,
diff --git a/openhands/utils/async_utils.py b/openhands/utils/async_utils.py
index 7da8d05ff5c6..2a3b73f5da7d 100644
--- a/openhands/utils/async_utils.py
+++ b/openhands/utils/async_utils.py
@@ -7,7 +7,7 @@
 EXECUTOR = ThreadPoolExecutor()
 
 
-async def sync_from_async(fn: Callable, *args, **kwargs):
+async def call_sync_from_async(fn: Callable, *args, **kwargs):
     """
     Shorthand for running a function in the default background thread pool executor
     and awaiting the result. The nature of synchronous code is that the future
@@ -19,7 +19,7 @@ async def sync_from_async(fn: Callable, *args, **kwargs):
     return result
 
 
-def async_from_sync(
+def call_async_from_sync(
     corofn: Callable, timeout: float = GENERAL_TIMEOUT, *args, **kwargs
 ):
     """
@@ -27,6 +27,11 @@ def async_from_sync(
     and awaiting the result
     """
 
+    if corofn is None:
+        raise ValueError('corofn is None')
+    if not asyncio.iscoroutinefunction(corofn):
+        raise ValueError('corofn is not a coroutine function')
+
     async def arun():
         coro = corofn(*args, **kwargs)
         result = await coro
@@ -46,6 +51,13 @@ def run():
     return result
 
 
+async def call_coro_in_bg_thread(
+    corofn: Callable, timeout: float = GENERAL_TIMEOUT, *args, **kwargs
+):
+    """Function for running a coroutine in a background thread."""
+    await call_sync_from_async(call_async_from_sync, corofn, timeout, *args, **kwargs)
+
+
 async def wait_all(
     iterable: Iterable[Coroutine], timeout: int = GENERAL_TIMEOUT
 ) -> List:
diff --git a/tests/unit/test_async_utils.py b/tests/unit/test_async_utils.py
index 89dd1e0f6915..3dc99438968e 100644
--- a/tests/unit/test_async_utils.py
+++ b/tests/unit/test_async_utils.py
@@ -1,11 +1,13 @@
 import asyncio
+import time
 
 import pytest
 
 from openhands.utils.async_utils import (
     AsyncException,
-    async_from_sync,
-    sync_from_async,
+    call_async_from_sync,
+    call_coro_in_bg_thread,
+    call_sync_from_async,
     wait_all,
 )
 
@@ -80,44 +82,44 @@ async def dummy(value: int):
 
 
 @pytest.mark.asyncio
-async def test_sync_from_async():
+async def test_call_sync_from_async():
     def dummy(value: int = 2):
         return value * 2
 
-    result = await sync_from_async(dummy)
+    result = await call_sync_from_async(dummy)
     assert result == 4
-    result = await sync_from_async(dummy, 3)
+    result = await call_sync_from_async(dummy, 3)
     assert result == 6
-    result = await sync_from_async(dummy, value=5)
+    result = await call_sync_from_async(dummy, value=5)
     assert result == 10
 
 
 @pytest.mark.asyncio
-async def test_sync_from_async_error():
+async def test_call_sync_from_async_error():
     def dummy():
         raise ValueError()
 
     with pytest.raises(ValueError):
-        await sync_from_async(dummy)
+        await call_sync_from_async(dummy)
 
 
-def test_async_from_sync():
+def test_call_async_from_sync():
     async def dummy(value: int):
         return value * 2
 
-    result = async_from_sync(dummy, 0, 3)
+    result = call_async_from_sync(dummy, 0, 3)
     assert result == 6
 
 
-def test_async_from_sync_error():
+def test_call_async_from_sync_error():
     async def dummy(value: int):
         raise ValueError()
 
     with pytest.raises(ValueError):
-        async_from_sync(dummy, 0, 3)
+        call_async_from_sync(dummy, 0, 3)
 
 
-def test_async_from_sync_background_tasks():
+def test_call_async_from_sync_background_tasks():
     events = []
 
     async def bg_task():
@@ -132,9 +134,33 @@ async def dummy(value: int):
         asyncio.create_task(bg_task())
         events.append('dummy_started')
 
-    async_from_sync(dummy, 0, 3)
+    call_async_from_sync(dummy, 0, 3)
 
     # We check that the function did not return until all coroutines completed
     # (Even though some of these were started as background tasks)
     expected = ['dummy_started', 'dummy_started', 'bg_started', 'bg_finished']
     assert expected == events
+
+
+@pytest.mark.asyncio
+async def test_call_coro_in_bg_thread():
+    times = {}
+
+    async def bad_async(id_):
+        # Dummy demonstrating some bad async function that does not cede control
+        time.sleep(0.1)
+        times[id_] = time.time()
+
+    async def curve_ball():
+        # A curve ball - an async function that wants to run while the bad async functions are in progress
+        await asyncio.sleep(0.05)
+        times['curve_ball'] = time.time()
+
+    start = time.time()
+    asyncio.create_task(curve_ball())
+    await wait_all(
+        call_coro_in_bg_thread(bad_async, id_=f'bad_async_{id_}') for id_ in range(5)
+    )
+    assert (times['curve_ball'] - start) == pytest.approx(0.05, abs=0.1)
+    for id_ in range(5):
+        assert (times[f'bad_async_{id_}'] - start) == pytest.approx(0.1, abs=0.1)

From 79cb41a94cae1cb2313845450c941cafde4ee032 Mon Sep 17 00:00:00 2001
From: AlexCuadron <alex.cl.2000@gmail.com>
Date: Thu, 17 Oct 2024 14:33:30 +0200
Subject: [PATCH 08/18] Initial Commit for the Supervisor Agent

---
 evaluation/swe_bench/run_infer.py             |   2 +
 openhands/agenthub/__init__.py                |   2 +
 .../agenthub/supervisor_agent/__init__.py     |   4 +
 openhands/agenthub/supervisor_agent/agent.py  | 156 +++++++++++++++
 openhands/agenthub/supervisor_agent/prompt.py | 179 ++++++++++++++++++
 5 files changed, 343 insertions(+)
 create mode 100644 openhands/agenthub/supervisor_agent/__init__.py
 create mode 100644 openhands/agenthub/supervisor_agent/agent.py
 create mode 100644 openhands/agenthub/supervisor_agent/prompt.py

diff --git a/evaluation/swe_bench/run_infer.py b/evaluation/swe_bench/run_infer.py
index 91984c1a7b1a..80e8ccbee20a 100644
--- a/evaluation/swe_bench/run_infer.py
+++ b/evaluation/swe_bench/run_infer.py
@@ -41,11 +41,13 @@
 AGENT_CLS_TO_FAKE_USER_RESPONSE_FN = {
     'CodeActAgent': codeact_user_response,
     'CodeActSWEAgent': codeact_user_response,
+    'SupervisorAgent': codeact_user_response,
 }
 
 AGENT_CLS_TO_INST_SUFFIX = {
     'CodeActAgent': 'When you think you have fixed the issue through code changes, please run the following command: <execute_bash> exit </execute_bash>.\n',
     'CodeActSWEAgent': 'When you think you have fixed the issue through code changes, please run the following command: <execute_bash> exit </execute_bash>.\n',
+    'SupervisorAgent': 'When you think you have fixed the issue, please run the following command: <execute_bash> exit </execute_bash>.\n',
 }
 
 
diff --git a/openhands/agenthub/__init__.py b/openhands/agenthub/__init__.py
index 0076976c27ed..489ecc7aaead 100644
--- a/openhands/agenthub/__init__.py
+++ b/openhands/agenthub/__init__.py
@@ -14,6 +14,7 @@
     delegator_agent,
     dummy_agent,
     planner_agent,
+    supervisor_agent,
 )
 
 __all__ = [
@@ -23,6 +24,7 @@
     'delegator_agent',
     'dummy_agent',
     'browsing_agent',
+    'supervisor_agent',
 ]
 
 for agent in all_microagents.values():
diff --git a/openhands/agenthub/supervisor_agent/__init__.py b/openhands/agenthub/supervisor_agent/__init__.py
new file mode 100644
index 000000000000..6b07ea69fc67
--- /dev/null
+++ b/openhands/agenthub/supervisor_agent/__init__.py
@@ -0,0 +1,4 @@
+from openhands.agenthub.supervisor_agent.agent import SupervisorAgent
+from openhands.controller.agent import Agent
+
+Agent.register('SupervisorAgent', SupervisorAgent)
diff --git a/openhands/agenthub/supervisor_agent/agent.py b/openhands/agenthub/supervisor_agent/agent.py
new file mode 100644
index 000000000000..1c79e61e0c87
--- /dev/null
+++ b/openhands/agenthub/supervisor_agent/agent.py
@@ -0,0 +1,156 @@
+import copy
+import logging
+from typing import Dict, List
+
+from openhands.agenthub.supervisor_agent.prompt import (
+    adjust_milestones,
+    get_initial_prompt,
+)
+from openhands.controller.agent import Agent
+from openhands.controller.state.state import State
+from openhands.core.config import AgentConfig
+from openhands.core.message import Message, TextContent
+from openhands.core.utils import json
+from openhands.events.action import Action, AgentDelegateAction, AgentFinishAction
+from openhands.events.action.agent import AgentRejectAction
+from openhands.events.observation.delegate import AgentDelegateObservation
+from openhands.llm.llm import LLM
+
+
+class SupervisorAgent(Agent):
+    VERSION = '1.0'
+    """
+    The Supervisor Agent is an agent that collects information from other agents
+    and makes decisions based on the information.
+    """
+
+    current_delegate: str = ''
+    sub_goals: List[Dict[str, str]] = []
+    current_goal_index: int = 0
+    summary: str = ''
+    task: str = ''
+
+    def __init__(self, llm: LLM, config: AgentConfig):
+        """Initialize the Supervisor Agent with an LLM
+
+        Parameters:
+        - llm (LLM): The llm to be used by this agent
+        """
+        super().__init__(llm, config)
+        # Set up logger
+        self.logger = logging.getLogger(__name__)
+        logging.basicConfig(level=logging.DEBUG)  # Set the logging level
+
+    def step(self, state: State) -> Action:
+        """Checks to see if current step is completed, returns AgentFinishAction if True.
+        Otherwise, delegates the task to the next agent in the pipeline.
+
+        Parameters:
+        - state (State): The current state given the previous actions and observations
+
+        Returns:
+        - AgentFinishAction: If the last state was 'completed', 'verified', or 'abandoned'
+        - AgentDelegateAction: The next agent to delegate the task to
+        """
+        self.logger.debug('Starting step with state: %s', state)
+        # Example logic for breaking down tasks and delegating
+        if not self.sub_goals:
+            self.logger.debug('No sub-goals found, breaking down task.')
+            task, _ = state.get_current_user_intent()
+            self.sub_goals = self.break_down_task(task)
+            self.logger.debug('Sub-goals: %s', self.sub_goals)
+            # If the LLM returns an empty list, reject the action
+            if self.sub_goals is None or self.sub_goals == []:
+                return AgentRejectAction()
+
+        if self.current_delegate == '':
+            self.logger.debug("Current delegate is empty, assigning 'manager'.")
+            # First subgoal as the current delegate is empty
+            self.current_delegate = 'manager'
+            return AgentDelegateAction(
+                agent='ManagerAgent',
+                inputs={'task': json.dumps(self.sub_goals[self.current_goal_index])},
+            )
+        elif self.current_delegate == 'manager':
+            self.logger.debug("Current delegate is 'manager'.")
+            last_observation = state.history.get_last_observation()
+
+            if not isinstance(last_observation, AgentDelegateObservation):
+                raise Exception('Last observation is not an AgentDelegateObservation')
+
+            if last_observation.outputs.get('action', '') == 'reject':
+                self.logger.debug('No summary found, creating adjustment prompt.')
+                reason = getattr(last_observation, 'reason', '')
+                # Ensure reason is a string
+                prompt = self.create_adjustment_prompt(reason)
+                # Get the sub-goals from the language model using the generated prompt
+                self.sub_goals = self.get_sub_goals_from_llm(prompt)
+                # Add the summary to the current sub-goal
+                current_task = copy.deepcopy(self.sub_goals[self.current_goal_index])
+                current_task['summary'] = (
+                    f'Summary from previous milestones: {self.summary}'
+                )
+                return AgentDelegateAction(
+                    agent='ManagerAgent', inputs={'task': json.dumps(current_task)}
+                )
+            else:
+                # Append the current milestone and summary to the agent's summary
+                summary = last_observation.outputs.get('summary', '')
+                self.append_to_summary(
+                    self.sub_goals[self.current_goal_index]['task'], summary
+                )
+                self.current_goal_index += 1
+
+                if self.current_goal_index < len(self.sub_goals):
+                    # Add the summary to the current sub-goal
+                    current_task = copy.deepcopy(
+                        self.sub_goals[self.current_goal_index]
+                    )
+                    current_task['summary'] = (
+                        f'Summary from previous milestones: {self.summary}'
+                    )
+
+                    return AgentDelegateAction(
+                        agent='ManagerAgent', inputs={'task': json.dumps(current_task)}
+                    )
+
+        return AgentFinishAction()
+
+    def break_down_task(self, task: str) -> List[Dict[str, str]]:
+        # Generate the initial prompt for breaking down the task
+        prompt = get_initial_prompt(task)
+        # Get the sub-goals from the language model using the generated prompt
+        return self.get_sub_goals_from_llm(prompt)
+
+    def should_interrupt(self, observation) -> bool:
+        # Logic to determine if the task should be interrupted
+        return False  # Placeholder
+
+    def summarize_history(self, history) -> str:
+        # Logic to summarize the history
+        return 'summary'  # Placeholder
+
+    def provide_guidance(self, state: State) -> Action:
+        # Logic to provide high-level guidance
+        return AgentFinishAction()  # Placeholder
+
+    def create_adjustment_prompt(self, reason: str) -> str:
+        return adjust_milestones(
+            self.sub_goals,
+            self.sub_goals[self.current_goal_index],
+            reason,
+            self.summary,
+            self.task,
+        )
+
+    def get_sub_goals_from_llm(self, prompt: str) -> List[Dict[str, str]]:
+        content = [TextContent(text=prompt)]
+        message = Message(role='user', content=content)
+        response = self.llm.completion(
+            messages=self.llm.format_messages_for_llm(message)
+        )
+        return json.loads(response['choices'][0]['message']['content'])
+
+    def append_to_summary(self, milestone_name: str, summary: str):
+        """Appends the milestone name and summary to the agent's summary state."""
+        self.summary += f'Milestone: {milestone_name}\nSummary: {summary}\n\n'
diff --git a/openhands/agenthub/supervisor_agent/prompt.py b/openhands/agenthub/supervisor_agent/prompt.py
new file mode 100644
index 000000000000..5cdf5b76d6e2
--- /dev/null
+++ b/openhands/agenthub/supervisor_agent/prompt.py
@@ -0,0 +1,179 @@
+from typing import Dict, List
+
+from openhands.core.utils import json
+
+HISTORY_SIZE = 20
+
+# General Description
+general_description = """
+You are a strategic manager AI in a software development team. You MUST think CAREFULLY how to complete the task assigned to you.
+You MUST think on a HIGHER LEVEL view always.
+
+You've been given the following task:
+%(task)s
+
+As a strategic manager, you create a plan with different sub-tasks and delegate the tasks to your team.
+At your disposal, you have a team of agents who will complete tasks for you. However, those agents only focus on the details.
+They CANNOT see the big picture.
+They need you to define self-contained tasks, that are easy for them to understand and complete.
+
+"""
+
+# Initial Prompt
+initial_prompt = """
+## Plan
+Your goal is to create a high-level plan, a list of subtasks that will bring you closer to the completion of the task. Remember to think
+CAREFULLY about how to complete the task. With each subtask, you MUST provide a "suggested approach".
+Think, step by step, how you would complete the subtask. Then provide that as the suggested approach.
+Try to be as detailed as possible, your goal is to HELP the agent finish the subtask as soon as possible.
+
+You MAY provide a list of "important details" for each subtask. These are details that the agent MUST consider when completing the subtask.
+
+ONLY generate tasks that are necessary to complete the task.
+
+You MUST ONLY generate a list of JSONs:
+
+[
+    {
+      "task": "<Task 1 name>",
+      "suggested_approach": "<suggested approach>",
+      "important_details": "<important details>"
+    },
+    {
+      "task": "<Task 2 name>",
+      "suggested_approach": "<suggested approach>",
+      "important_details": "<important details>"
+    },
+    {
+      "task": "<Task 3 name>",
+      "suggested_approach": "<suggested approach>",
+      "important_details": "<important details>"
+    },
+]
+
+The tasks MUST be generated in order, they MUST NOT depend on future tasks or previous tasks. They MUST be independent.
+You MUST generate at least 1 task.
+
+For example:
+User prompt:
+
+"
+Enable quiet mode/no-verbose in CLI for use in pre-commit hook There seems to be only an option to increase the level of verbosity when using
+SQLFluff [CLI](https://docs.sqlfluff.com/en/stable/cli.html), not to limit it further. It would be great to have an option to further limit the amount of prints when running
+`sqlfluff fix`, especially in combination with deployment using a pre-commit hook. For example, only print the return status and the number of fixes applied, similar to how it
+is when using `black` in a pre-commit hook: ![image](https://user-images.githubusercontent.com/10177212/140480676-dc98d00b-4383-44f2-bb90-3301a6eedec2.png) This hides the potentially
+long list of fixes that are being applied to the SQL files, which can get quite verbose.
+"
+
+Your response:
+
+[
+    {
+        "task": "Research SQLFluff CLI verbosity options",
+        "suggested_approach": "Investigate the current SQLFluff CLI documentation and source code to understand how verbosity levels are currently implemented. Identify if there are any existing flags or settings that can be adjusted to reduce verbosity.",
+        "important_details": "Focus on the 'fix' command and any related verbosity settings. Document any findings that could be useful for implementing a quiet mode."
+    },
+    {
+        "task": "Design a quiet mode feature for SQLFluff CLI",
+        "suggested_approach": "Based on the research findings, design a new feature that allows users to enable a quiet mode. This mode should minimize output to only essential information such as return status and number of fixes applied.",
+        "important_details": "Ensure the design is compatible with existing CLI options and does not interfere with other functionalities."
+    },
+    {
+        "task": "Implement the quiet mode feature",
+        "suggested_approach": "Modify the SQLFluff CLI codebase to add the new quiet mode feature. Implement the necessary changes in the code to support this feature and ensure it can be activated via a command-line flag.",
+        "important_details": "Write unit tests to verify that the quiet mode works as expected and does not affect other CLI functionalities."
+    },
+    {
+        "task": "Test the quiet mode feature",
+        "suggested_approach": "Conduct thorough testing of the new quiet mode feature in various scenarios, including its use in a pre-commit hook. Ensure that it behaves as expected and provides the desired level of output reduction.",
+        "important_details": "Test with different verbosity levels to ensure compatibility and check for any edge cases that might cause unexpected behavior."
+    },
+    {
+        "task": "Document the new feature",
+        "suggested_approach": "Update the SQLFluff CLI documentation to include information about the new quiet mode feature. Provide examples of how to use it and explain its benefits.",
+        "important_details": "Ensure the documentation is clear and easy to understand for users who may not be familiar with the technical details."
+    }
+]
+"""
+
+adjustment_prompt = """
+
+    This is the current active plan that your subordinates are working on:
+    %(milestones)s
+
+    And this is the current subtask that your subordinates are working on:
+    ## Current subtask
+    subtask: %(milestone_task)s
+    Suggested Approach: %(milestone_suggested_approach)s
+    Important Details: %(milestone_important_details)s
+
+    However, it seems that the current subtask is not being completed successfully.
+    Because of the following reason: %(reason)s
+
+    You have the following contextual information that has been gathered up to this point.
+    This information MIGHT help you adjust the plan:
+    %(summary)s
+
+    ## Task
+    As a strategic manager, you must reflect on the failed subtask and decide on the necessary adjustments. Consider the following:
+
+    1. Analyze the reason for failure and determine if the suggested approach or important details need modification.
+    2. Decide if the failed subtask should be split into smaller, more manageable tasks.
+    3. Consider if new plan need to be added to address any gaps in the plan.
+    4. Update the remaining plan to ensure the overall plan remains feasible and effective.
+
+    You MUST NOT change the task you were given.
+
+    You MUST make changes to the current subtask or to the ones AFTER. In NO case you can change the ones BEFORE.
+    Generate ONLY a list of JSONs. Do NOT generate any markdown or comments.
+    """
+
+
+def get_initial_prompt(task: str) -> str:
+    """Gets the prompt for the planner agent.
+
+    Formatted with the most recent action-observation pairs, current task, and hint based on last action
+
+    Parameters:
+    - state (State): The state of the current agent
+
+    Returns: with historical values
+    """
+    return (general_description + initial_prompt) % {
+        'task': task,
+    }
+
+
+def adjust_milestones(
+    milestones: List[Dict],
+    subtask: Dict[str, str],
+    reason: str,
+    summary: str,
+    task: str,
+) -> str:
+    """Adjusts the milestones based on a failed subtask and its reason.
+
+    Parameters:
+    - milestones (List[Dict]): The current list of milestones.
+    - subtask (Dict): The subtask that was not completed successfully.
+    - reason (str): The reason provided for the failure.
+    - summary (str): A summary of everything up to this point.
+    - task (str): The user's task.
+
+    Returns: A prompt for the strategic manager agent to self-reflect and adjust the milestones.
+    """
+    # Extract values from the subtask dictionary
+    milestone_task = subtask['task']
+    milestone_suggested_approach = subtask['suggested_approach']
+    milestone_important_details = subtask['important_details']
+
+    # Use the extracted values in the string formatting
+    return (general_description + adjustment_prompt) % {
+        'milestones': json.dumps(milestones),
+        'reason': reason,
+        'summary': summary,
+        'task': task,
+        'milestone_task': milestone_task,
+        'milestone_suggested_approach': milestone_suggested_approach,
+        'milestone_important_details': milestone_important_details,
+    }

From 640f769e4dfc6349dd534705aa2cd652a3745a06 Mon Sep 17 00:00:00 2001
From: AlexCuadron <alex.cl.2000@gmail.com>
Date: Thu, 24 Oct 2024 12:21:21 +0200
Subject: [PATCH 09/18] enables codeactagent delegation

---
 openhands/agenthub/codeact_agent/codeact_agent.py | 5 +++++
 1 file changed, 5 insertions(+)

diff --git a/openhands/agenthub/codeact_agent/codeact_agent.py b/openhands/agenthub/codeact_agent/codeact_agent.py
index cacd68353732..634bbefbbadd 100644
--- a/openhands/agenthub/codeact_agent/codeact_agent.py
+++ b/openhands/agenthub/codeact_agent/codeact_agent.py
@@ -282,7 +282,12 @@ def _get_messages(self, state: State) -> list[Message]:
             ),
             None,
         )
+
         if latest_user_message:
+            # Enables AgentDelegation
+            task: str = state.inputs.get('task', '')
+            if task:
+                latest_user_message.content.append(TextContent(text=task))
             reminder_text = f'\n\nENVIRONMENT REMINDER: You have {state.max_iterations - state.iteration} turns left to complete the task. When finished reply with <finish></finish>.'
             latest_user_message.content.append(TextContent(text=reminder_text))
 

From d5d44e279eb6254f18449672b4a03cd4b025f29c Mon Sep 17 00:00:00 2001
From: AlexCuadron <alex.cl.2000@gmail.com>
Date: Thu, 24 Oct 2024 12:43:16 +0200
Subject: [PATCH 10/18] hacky way to enable different LLMs

---
 openhands/agenthub/supervisor_agent/agent.py  | 145 ++++++++++--------
 openhands/agenthub/supervisor_agent/prompt.py |  56 +++----
 openhands/controller/agent_controller.py      |   3 +
 openhands/events/action/agent.py              |   3 +-
 openhands/runtime/builder/docker.py           |   1 +
 openhands/runtime/utils/edit.py               |   2 +-
 6 files changed, 103 insertions(+), 107 deletions(-)

diff --git a/openhands/agenthub/supervisor_agent/agent.py b/openhands/agenthub/supervisor_agent/agent.py
index 1c79e61e0c87..61ef4bed8fc0 100644
--- a/openhands/agenthub/supervisor_agent/agent.py
+++ b/openhands/agenthub/supervisor_agent/agent.py
@@ -40,82 +40,91 @@ def __init__(self, llm: LLM, config: AgentConfig):
         # Set up logger
         self.logger = logging.getLogger(__name__)
         logging.basicConfig(level=logging.DEBUG)  # Set the logging level
+        self.llm_config = llm.config
 
     def step(self, state: State) -> Action:
-        """Checks to see if current step is completed, returns AgentFinishAction if True.
-        Otherwise, delegates the task to the next agent in the pipeline.
-
-        Parameters:
-        - state (State): The current state given the previous actions and observations
-
-        Returns:
-        - AgentFinishAction: If the last state was 'completed', 'verified', or 'abandoned'
-        - AgentDelegateAction: The next agent to delegate the task to
-        """
         self.logger.debug('Starting step with state: %s', state)
-        # Example logic for breaking down tasks and delegating
+        self.logger.debug('LLM config: %s', self.llm_config)
+
         if not self.sub_goals:
-            self.logger.debug('No sub-goals found, breaking down task.')
-            task, _ = state.get_current_user_intent()
-            self.sub_goals = self.break_down_task(task)
-            self.logger.debug('Sub-goals: %s', self.sub_goals)
-            # If the LLM returns an empty list, reject the action
-            if self.sub_goals is None or self.sub_goals == []:
-                return AgentRejectAction()
+            self.initialize_sub_goals(state)
 
         if self.current_delegate == '':
-            self.logger.debug("Current delegate is empty, assigning 'manager'.")
-            # First subgoal as the current delegate is empty
-            self.current_delegate = 'manager'
-            return AgentDelegateAction(
-                agent='ManagerAgent',
-                inputs={'task': json.dumps(self.sub_goals[self.current_goal_index])},
+            self.current_delegate = 'CodeActAgent'
+            return self.delegate_to_agent(
+                'CodeActAgent', self.construct_task_details(self.prepare_current_task())
             )
-        elif self.current_delegate == 'manager':
-            self.logger.debug("Current delegate is 'manager'.")
-            last_observation = state.history.get_last_observation()
-
-            if not isinstance(last_observation, AgentDelegateObservation):
-                raise Exception('Last observation is not an AgentDelegateObservation')
-
-            if last_observation.outputs.get('action', '') == 'reject':
-                self.logger.debug('No summary found, creating adjustment prompt.')
-                reason = getattr(last_observation, 'reason', '')
-                # Ensure reason is a string
-                prompt = self.create_adjustment_prompt(reason)
-                # Get the sub-goals from the language model using the generated prompt
-                self.sub_goals = self.get_sub_goals_from_llm(prompt)
-                # Add the summary to the current sub-goal
-                current_task = copy.deepcopy(self.sub_goals[self.current_goal_index])
-                current_task['summary'] = (
-                    f'Summary from previous milestones: {self.summary}'
-                )
-                return AgentDelegateAction(
-                    agent='ManagerAgent', inputs={'task': json.dumps(current_task)}
-                )
-            else:
-                # Append the current milestone and summary to the agent's summary
-                summary = last_observation.outputs.get('summary', '')
-                self.append_to_summary(
-                    self.sub_goals[self.current_goal_index]['task'], summary
-                )
-                self.current_goal_index += 1
-
-                if self.current_goal_index < len(self.sub_goals):
-                    # Add the summary to the current sub-goal
-                    current_task = copy.deepcopy(
-                        self.sub_goals[self.current_goal_index]
-                    )
-                    current_task['summary'] = (
-                        f'Summary from previous milestones: {self.summary}'
-                    )
-
-                    return AgentDelegateAction(
-                        agent='ManagerAgent', inputs={'task': json.dumps(current_task)}
-                    )
+
+        elif self.current_delegate == 'CodeActAgent':
+            return self.handle_code_act_agent(state)
+
+        return AgentFinishAction()
+
+    def initialize_sub_goals(self, state: State):
+        self.logger.debug('No sub-goals found, breaking down task.')
+        self.task, _ = state.get_current_user_intent()
+        self.sub_goals = self.break_down_task(self.task)
+        self.logger.debug('Sub-goals: %s', self.sub_goals)
+        if not self.sub_goals:
+            return AgentRejectAction()
+
+    def delegate_to_agent(self, agent_name: str, task: str) -> AgentDelegateAction:
+        self.logger.debug(f'Delegating to agent: {agent_name}')
+
+        return AgentDelegateAction(agent=agent_name, inputs={'task': task})
+
+    def handle_code_act_agent(self, state: State) -> Action:
+        self.logger.debug("Current delegate is 'CodeActAgent'.")
+        last_observation = state.history.get_last_observation()
+
+        if not isinstance(last_observation, AgentDelegateObservation):
+            raise Exception('Last observation is not an AgentDelegateObservation')
+
+        if last_observation.outputs.get('action', '') == 'reject':
+            return self.handle_rejection(last_observation)
+
+        return self.handle_success(last_observation)
+
+    def handle_rejection(
+        self, last_observation: AgentDelegateObservation
+    ) -> AgentDelegateAction:
+        self.logger.debug('No summary found, creating adjustment prompt.')
+        reason = getattr(last_observation, 'reason', '')
+        prompt = self.create_adjustment_prompt(reason)
+        self.sub_goals = self.get_sub_goals_from_llm(prompt)
+        current_task = self.prepare_current_task()
+        return self.delegate_to_agent(
+            'CodeActAgent', self.construct_task_details(current_task)
+        )
+
+    def handle_success(self, last_observation: AgentDelegateObservation) -> Action:
+        summary = last_observation.outputs.get('summary', '')
+        self.append_to_summary(summary)
+        self.current_goal_index += 1
+
+        if self.current_goal_index < len(self.sub_goals):
+            current_task = self.prepare_current_task()
+            task_details = self.construct_task_details(current_task)
+            return self.delegate_to_agent('CodeActAgent', task_details)
 
         return AgentFinishAction()
 
+    def prepare_current_task(self) -> Dict[str, str]:
+        current_task = copy.deepcopy(self.sub_goals[self.current_goal_index])
+        current_task['summary'] = self.summary if self.summary else ''
+        return current_task
+
+    def construct_task_details(self, current_task: Dict[str, str]) -> str:
+        task_details = (
+            f"Task: {self.task}\n\n"
+            f"Next Subtask: {current_task['task']}\n"
+            f"Suggested Approach: {current_task['suggested_approach']}\n"
+            f"Important Details: {current_task['important_details']}"
+        )
+        if self.summary:
+            task_details = f'Progress so far: {self.summary}\n\n' + task_details
+        return task_details
+
     def break_down_task(self, task: str) -> List[Dict[str, str]]:
         # Generate the initial prompt for breaking down the task
         prompt = get_initial_prompt(task)
@@ -151,6 +160,6 @@ def get_sub_goals_from_llm(self, prompt: str) -> List[Dict[str, str]]:
         )
         return json.loads(response['choices'][0]['message']['content'])
 
-    def append_to_summary(self, milestone_name: str, summary: str):
+    def append_to_summary(self, summary: str):
         """Appends the milestone name and summary to the agent's summary state."""
-        self.summary += f'Milestone: {milestone_name}\nSummary: {summary}\n\n'
+        self.summary += f'{summary}\n\n'
diff --git a/openhands/agenthub/supervisor_agent/prompt.py b/openhands/agenthub/supervisor_agent/prompt.py
index 5cdf5b76d6e2..e4e0eaedfa22 100644
--- a/openhands/agenthub/supervisor_agent/prompt.py
+++ b/openhands/agenthub/supervisor_agent/prompt.py
@@ -6,8 +6,9 @@
 
 # General Description
 general_description = """
-You are a strategic manager AI in a software development team. You MUST think CAREFULLY how to complete the task assigned to you.
-You MUST think on a HIGHER LEVEL view always.
+You are a strategic planner AI in a software development team. You have a team of agents
+who will complete the tasks you give them. Each agent is an expert in a specific area.
+You MUST think CAREFULLY how to complete the task assigned to you.
 
 You've been given the following task:
 %(task)s
@@ -44,15 +45,10 @@
       "suggested_approach": "<suggested approach>",
       "important_details": "<important details>"
     },
-    {
-      "task": "<Task 3 name>",
-      "suggested_approach": "<suggested approach>",
-      "important_details": "<important details>"
-    },
 ]
 
 The tasks MUST be generated in order, they MUST NOT depend on future tasks or previous tasks. They MUST be independent.
-You MUST generate at least 1 task.
+You MUST generate at least 1 task. The last task MUST be the implementation task. You WILL NOT need a test file.
 
 For example:
 User prompt:
@@ -73,35 +69,20 @@
         "suggested_approach": "Investigate the current SQLFluff CLI documentation and source code to understand how verbosity levels are currently implemented. Identify if there are any existing flags or settings that can be adjusted to reduce verbosity.",
         "important_details": "Focus on the 'fix' command and any related verbosity settings. Document any findings that could be useful for implementing a quiet mode."
     },
-    {
-        "task": "Design a quiet mode feature for SQLFluff CLI",
-        "suggested_approach": "Based on the research findings, design a new feature that allows users to enable a quiet mode. This mode should minimize output to only essential information such as return status and number of fixes applied.",
-        "important_details": "Ensure the design is compatible with existing CLI options and does not interfere with other functionalities."
-    },
     {
         "task": "Implement the quiet mode feature",
         "suggested_approach": "Modify the SQLFluff CLI codebase to add the new quiet mode feature. Implement the necessary changes in the code to support this feature and ensure it can be activated via a command-line flag.",
         "important_details": "Write unit tests to verify that the quiet mode works as expected and does not affect other CLI functionalities."
-    },
-    {
-        "task": "Test the quiet mode feature",
-        "suggested_approach": "Conduct thorough testing of the new quiet mode feature in various scenarios, including its use in a pre-commit hook. Ensure that it behaves as expected and provides the desired level of output reduction.",
-        "important_details": "Test with different verbosity levels to ensure compatibility and check for any edge cases that might cause unexpected behavior."
-    },
-    {
-        "task": "Document the new feature",
-        "suggested_approach": "Update the SQLFluff CLI documentation to include information about the new quiet mode feature. Provide examples of how to use it and explain its benefits.",
-        "important_details": "Ensure the documentation is clear and easy to understand for users who may not be familiar with the technical details."
     }
 ]
 """
 
 adjustment_prompt = """
 
-    This is the current active plan that your subordinates are working on:
+    This is the current active plan that your agents are working on:
     %(milestones)s
 
-    And this is the current subtask that your subordinates are working on:
+    And this is the current subtask that your agents are working on:
     ## Current subtask
     subtask: %(milestone_task)s
     Suggested Approach: %(milestone_suggested_approach)s
@@ -130,19 +111,15 @@
 
 
 def get_initial_prompt(task: str) -> str:
-    """Gets the prompt for the planner agent.
-
-    Formatted with the most recent action-observation pairs, current task, and hint based on last action
-
-    Parameters:
-    - state (State): The state of the current agent
-
-    Returns: with historical values
-    """
-    return (general_description + initial_prompt) % {
+    formatted_prompt = (general_description + initial_prompt) % {
         'task': task,
     }
 
+    # Add instruction to not include json formatting
+    formatted_prompt += '\n\nIMPORTANT: Do not include ```json at the start or ``` at the end of your response. Just return the raw JSON list.'
+
+    return formatted_prompt
+
 
 def adjust_milestones(
     milestones: List[Dict],
@@ -167,8 +144,8 @@ def adjust_milestones(
     milestone_suggested_approach = subtask['suggested_approach']
     milestone_important_details = subtask['important_details']
 
-    # Use the extracted values in the string formatting
-    return (general_description + adjustment_prompt) % {
+    # Get the formatted prompt
+    formatted_prompt = (general_description + adjustment_prompt) % {
         'milestones': json.dumps(milestones),
         'reason': reason,
         'summary': summary,
@@ -177,3 +154,8 @@ def adjust_milestones(
         'milestone_suggested_approach': milestone_suggested_approach,
         'milestone_important_details': milestone_important_details,
     }
+
+    # Add instruction to not include json formatting
+    formatted_prompt += '\n\nIMPORTANT: Do not include ```json at the start or ``` at the end of your response. Just return the raw JSON list.'
+
+    return formatted_prompt
diff --git a/openhands/controller/agent_controller.py b/openhands/controller/agent_controller.py
index 55ca61ddddee..139ba06ebc17 100644
--- a/openhands/controller/agent_controller.py
+++ b/openhands/controller/agent_controller.py
@@ -1,5 +1,6 @@
 import asyncio
 import copy
+import logging
 import traceback
 from typing import Type
 
@@ -63,6 +64,7 @@ class AgentController:
     parent: 'AgentController | None' = None
     delegate: 'AgentController | None' = None
     _pending_action: Action | None = None
+    logger: logging.Logger
 
     def __init__(
         self,
@@ -98,6 +100,7 @@ def __init__(
         self.id = sid
         self.agent = agent
         self.headless_mode = headless_mode
+        self.logger = logging.getLogger(f'AgentController-{sid}')
 
         # subscribe to the event stream
         self.event_stream = event_stream
diff --git a/openhands/events/action/agent.py b/openhands/events/action/agent.py
index f49f573ed698..eedac830422b 100644
--- a/openhands/events/action/agent.py
+++ b/openhands/events/action/agent.py
@@ -1,5 +1,5 @@
 from dataclasses import dataclass, field
-from typing import Any
+from typing import Any, Dict, Optional
 
 from openhands.core.schema import ActionType
 from openhands.events.action.action import Action
@@ -74,6 +74,7 @@ class AgentDelegateAction(Action):
     inputs: dict
     thought: str = ''
     action: str = ActionType.DELEGATE
+    llm_config: Optional[Dict[str, Any]] = None
 
     @property
     def message(self) -> str:
diff --git a/openhands/runtime/builder/docker.py b/openhands/runtime/builder/docker.py
index 09f94f103dff..9c5850982633 100644
--- a/openhands/runtime/builder/docker.py
+++ b/openhands/runtime/builder/docker.py
@@ -70,6 +70,7 @@ def build(
             f'--build-arg=OPENHANDS_RUNTIME_BUILD_TIME={datetime.datetime.now().isoformat()}',
             f'--tag={target_image_hash_name}',
             '--load',
+            '--platform=linux/amd64',
         ]
 
         # Include the platform argument only if platform is specified
diff --git a/openhands/runtime/utils/edit.py b/openhands/runtime/utils/edit.py
index 4ed5c0edafc6..350683175750 100644
--- a/openhands/runtime/utils/edit.py
+++ b/openhands/runtime/utils/edit.py
@@ -101,7 +101,7 @@ class FileEditRuntimeMixin(FileEditRuntimeInterface):
     def __init__(self, *args, **kwargs):
         super().__init__(*args, **kwargs)
 
-        llm_config = self.config.get_llm_config()
+        llm_config = self.config.get_llm_config_from_agent(self.config.default_agent)
 
         if llm_config.draft_editor is None:
             llm_config.draft_editor = copy.deepcopy(llm_config)

From f1d317c7004461f412728f20fd0f53dae70a9f68 Mon Sep 17 00:00:00 2001
From: AlexCuadron <alex.cl.2000@gmail.com>
Date: Fri, 25 Oct 2024 19:01:00 +0200
Subject: [PATCH 11/18] Some progress

---
 openhands/agenthub/__init__.py                |   2 +
 openhands/agenthub/searcher_agent/__init__.py |   4 +
 .../agenthub/searcher_agent/action_parser.py  | 158 ++++++++
 openhands/agenthub/searcher_agent/agent.py    | 175 ++++++++
 openhands/agenthub/searcher_agent/prompt.py   |  69 ++++
 openhands/agenthub/supervisor_agent/agent.py  | 182 ++++-----
 openhands/agenthub/supervisor_agent/prompt.py | 383 +++++++++++++-----
 7 files changed, 764 insertions(+), 209 deletions(-)
 create mode 100644 openhands/agenthub/searcher_agent/__init__.py
 create mode 100644 openhands/agenthub/searcher_agent/action_parser.py
 create mode 100644 openhands/agenthub/searcher_agent/agent.py
 create mode 100644 openhands/agenthub/searcher_agent/prompt.py

diff --git a/openhands/agenthub/__init__.py b/openhands/agenthub/__init__.py
index 489ecc7aaead..96c766124989 100644
--- a/openhands/agenthub/__init__.py
+++ b/openhands/agenthub/__init__.py
@@ -14,6 +14,7 @@
     delegator_agent,
     dummy_agent,
     planner_agent,
+    searcher_agent,
     supervisor_agent,
 )
 
@@ -24,6 +25,7 @@
     'delegator_agent',
     'dummy_agent',
     'browsing_agent',
+    'searcher_agent',
     'supervisor_agent',
 ]
 
diff --git a/openhands/agenthub/searcher_agent/__init__.py b/openhands/agenthub/searcher_agent/__init__.py
new file mode 100644
index 000000000000..1f4b7d50c642
--- /dev/null
+++ b/openhands/agenthub/searcher_agent/__init__.py
@@ -0,0 +1,4 @@
+from openhands.agenthub.searcher_agent.agent import SearcherAgent
+from openhands.controller.agent import Agent
+
+Agent.register('SearcherAgent', SearcherAgent)
diff --git a/openhands/agenthub/searcher_agent/action_parser.py b/openhands/agenthub/searcher_agent/action_parser.py
new file mode 100644
index 000000000000..54ad267cad8b
--- /dev/null
+++ b/openhands/agenthub/searcher_agent/action_parser.py
@@ -0,0 +1,158 @@
+import re
+
+from openhands.controller.action_parser import (
+    ActionParser,
+    ResponseParser,
+)
+from openhands.events.action import (
+    Action,
+    AgentFinishAction,
+    CmdRunAction,
+    IPythonRunCellAction,
+    MessageAction,
+)
+
+
+class SearcherAgentResponseParser(ResponseParser):
+    """Parser action:
+    - CmdRunAction(command) - bash command to run
+    - IPythonRunCellAction(code) - IPython code to run
+    - MessageAction(content) - Message action to run (e.g. ask for clarification)
+    - AgentFinishAction() - end the interaction
+    """
+
+    def __init__(self):
+        # Need pay attention to the item order in self.action_parsers
+        super().__init__()
+        self.action_parsers = [
+            SearcherAgentActionParserFinish(),
+            SearcherAgentActionParserCmdRun(),
+            SearcherAgentActionParserIPythonRunCell(),
+        ]
+        self.default_parser = SearcherAgentActionParserMessage()
+
+    def parse(self, response) -> Action:
+        action_str = self.parse_response(response)
+        return self.parse_action(action_str)
+
+    def parse_response(self, response) -> str:
+        action = response.choices[0].message.content
+        if action is None:
+            return ''
+        for lang in ['bash', 'ipython', 'browse']:
+            # special handling for DeepSeek: it has stop-word bug and returns </execute_ipython instead of </execute_ipython>
+            if f'</execute_{lang}' in action and f'</execute_{lang}>' not in action:
+                action = action.replace(f'</execute_{lang}', f'</execute_{lang}>')
+
+            if f'<execute_{lang}>' in action and f'</execute_{lang}>' not in action:
+                action += f'</execute_{lang}>'
+        if '<file_edit' in action and '</file_edit>' not in action:
+            action += '</file_edit>'
+        return action
+
+    def parse_action(self, action_str: str) -> Action:
+        for action_parser in self.action_parsers:
+            if action_parser.check_condition(action_str):
+                return action_parser.parse(action_str)
+        return self.default_parser.parse(action_str)
+
+
+class SearcherAgentActionParserFinish(ActionParser):
+    """Parser action:
+    - AgentFinishAction() - end the interaction
+    """
+
+    def __init__(
+        self,
+    ):
+        self.finish_command = None
+
+    def check_condition(self, action_str: str) -> bool:
+        self.finish_command = re.search(r'<finish>.*</finish>', action_str, re.DOTALL)
+        return self.finish_command is not None
+
+    def parse(self, action_str: str) -> Action:
+        assert (
+            self.finish_command is not None
+        ), 'self.finish_command should not be None when parse is called'
+        output = action_str.replace(self.finish_command.group(0), '').strip()
+        outputs = {'output': output}
+        return AgentFinishAction(outputs=outputs)
+
+
+class SearcherAgentActionParserCmdRun(ActionParser):
+    """Parser action:
+    - CmdRunAction(command) - bash command to run
+    - AgentFinishAction() - end the interaction
+    """
+
+    def __init__(
+        self,
+    ):
+        self.bash_command = None
+
+    def check_condition(self, action_str: str) -> bool:
+        self.bash_command = re.search(
+            r'<execute_bash>(.*?)</execute_bash>', action_str, re.DOTALL
+        )
+        return self.bash_command is not None
+
+    def parse(self, action_str: str) -> Action:
+        assert (
+            self.bash_command is not None
+        ), 'self.bash_command should not be None when parse is called'
+        thought = action_str.replace(self.bash_command.group(0), '').strip()
+        # a command was found
+        command_group = self.bash_command.group(1).strip()
+        if command_group.strip() == 'exit':
+            return AgentFinishAction(thought=thought)
+        return CmdRunAction(command=command_group, thought=thought)
+
+
+class SearcherAgentActionParserIPythonRunCell(ActionParser):
+    """Parser action:
+    - IPythonRunCellAction(code) - IPython code to run
+    """
+
+    def __init__(
+        self,
+    ):
+        self.python_code = None
+        self.jupyter_kernel_init_code: str = 'from agentskills import *'
+
+    def check_condition(self, action_str: str) -> bool:
+        self.python_code = re.search(
+            r'<execute_ipython>(.*?)</execute_ipython>', action_str, re.DOTALL
+        )
+        return self.python_code is not None
+
+    def parse(self, action_str: str) -> Action:
+        assert (
+            self.python_code is not None
+        ), 'self.python_code should not be None when parse is called'
+        code_group = self.python_code.group(1).strip()
+        thought = action_str.replace(self.python_code.group(0), '').strip()
+        return IPythonRunCellAction(
+            code=code_group,
+            thought=thought,
+            kernel_init_code=self.jupyter_kernel_init_code,
+        )
+
+
+class SearcherAgentActionParserMessage(ActionParser):
+    """Parser action:
+    - MessageAction(content) - Message action to run (e.g. ask for clarification)
+    """
+
+    def __init__(
+        self,
+    ):
+        pass
+
+    def check_condition(self, action_str: str) -> bool:
+        # We assume the LLM is GOOD enough that when it returns pure natural language
+        # it wants to talk to the user
+        return True
+
+    def parse(self, action_str: str) -> Action:
+        return MessageAction(content=action_str, wait_for_response=True)
diff --git a/openhands/agenthub/searcher_agent/agent.py b/openhands/agenthub/searcher_agent/agent.py
new file mode 100644
index 000000000000..f987c0708310
--- /dev/null
+++ b/openhands/agenthub/searcher_agent/agent.py
@@ -0,0 +1,175 @@
+import logging
+
+from openhands.agenthub.searcher_agent.action_parser import SearcherAgentResponseParser
+from openhands.agenthub.searcher_agent.prompt import get_prompt
+from openhands.controller.agent import Agent
+from openhands.controller.state.state import State
+from openhands.core.config import AgentConfig
+from openhands.core.config.llm_config import LLMConfig
+from openhands.core.message import Message, TextContent
+from openhands.events.action import Action, AgentFinishAction
+from openhands.events.action.commands import CmdRunAction, IPythonRunCellAction
+from openhands.events.action.message import MessageAction
+from openhands.events.observation.commands import (
+    CmdOutputObservation,
+    IPythonRunCellObservation,
+)
+from openhands.events.observation.error import ErrorObservation
+from openhands.events.observation.observation import Observation
+from openhands.events.observation.reject import UserRejectObservation
+from openhands.llm.llm import LLM
+
+
+class SearcherAgent(Agent):
+    VERSION = '1.0'
+    """
+    The Searcher Agent is an agent that searches the codebase for relevant information.
+    """
+
+    action_parser = SearcherAgentResponseParser()
+
+    def __init__(self, llm: LLM, config: AgentConfig):
+        """Initialize the Searcher Agent with an LLM
+
+        Parameters:
+        - llm (LLM): The llm to be used by this agent
+        - config (AgentConfig): The configuration for this agent
+        """
+        # TODO: Remove this once we have a real LLM config
+        llm_config = LLMConfig(
+            model='deepseek/deepseek-chat', api_key='REDACTED', temperature=0.0
+        )
+        llm = LLM(llm_config)
+        # TODO: Remove this once we have a real AgentConfig
+        config = AgentConfig(llm_config='deepseek')
+        super().__init__(llm, config)
+        # Set up logger
+        self.logger = logging.getLogger(__name__)
+        logging.basicConfig(level=logging.DEBUG)  # Set the logging level
+
+    def step(self, state: State) -> Action:
+        """Performs one step using the Searcher Agent.
+        This includes gathering info about the codebase and summarizing relevant information.
+
+        Parameters:
+        - state (State): used to get updated info
+
+        Returns:
+        - Action: The next action to take
+        """
+        # Check if we should exit
+        latest_user_message = state.history.get_last_user_message()
+        if latest_user_message and latest_user_message.strip() == '/exit':
+            return AgentFinishAction()
+
+        # Prepare messages for LLM
+        messages = []
+
+        # Add system and initial messages
+        task: str = state.inputs.get('task', '')
+        suggested_approach: str = state.inputs.get('suggested_approach', '')
+        messages.extend(
+            [
+                Message(
+                    role='system',
+                    content=[TextContent(text=get_prompt(task, suggested_approach))],
+                )
+            ]
+        )
+
+        # Add history messages
+        for event in state.history.get_events():
+            if isinstance(event, Action):
+                message = self.get_action_message(event)
+            elif isinstance(event, Observation):
+                message = self.get_observation_message(event)
+            else:
+                raise ValueError(f'Unknown event type: {type(event)}')
+
+            if message:
+                # Handle consecutive messages from same role
+                if messages and messages[-1].role == message.role:
+                    messages[-1].content.extend(message.content)
+                else:
+                    messages.append(message)
+
+        # Get response from LLM
+        params = {
+            'messages': self.llm.format_messages_for_llm(messages),
+            'stop': [
+                '</execute_ipython>',
+                '</execute_bash>',
+                '</finish>',
+            ],
+        }
+
+        response = self.llm.completion(**params)
+
+        # Parse and return the next action
+        return self.action_parser.parse(response)
+
+    def get_action_message(self, action: Action) -> Message | None:
+        """Convert an Action to a Message for the LLM conversation.
+
+        Parameters:
+        - action (Action): The action to convert
+
+        Returns:
+        - Message | None: The converted message, or None if action type is not supported
+        """
+        if isinstance(action, CmdRunAction):
+            return Message(
+                role='assistant',
+                content=[
+                    TextContent(
+                        text=f'{action.thought}\n<execute_bash>\n{action.command}\n</execute_bash>'
+                    )
+                ],
+            )
+        elif isinstance(action, IPythonRunCellAction):
+            return Message(
+                role='assistant',
+                content=[
+                    TextContent(
+                        text=f'{action.thought}\n<execute_ipython>\n{action.code}\n</execute_ipython>'
+                    )
+                ],
+            )
+        elif isinstance(action, MessageAction):
+            return Message(
+                role='user' if action.source == 'user' else 'assistant',
+                content=[TextContent(text=action.content)],
+            )
+        elif isinstance(action, AgentFinishAction) and action.source == 'agent':
+            return Message(role='assistant', content=[TextContent(text=action.thought)])
+        return None
+
+    def get_observation_message(self, obs: Observation) -> Message | None:
+        """Convert an Observation to a Message for the LLM conversation.
+
+        Parameters:
+        - obs (Observation): The observation to convert
+
+        Returns:
+        - Message | None: The converted message, or None if observation type is not supported
+        """
+        obs_prefix = 'OBSERVATION:\n'
+        if isinstance(obs, CmdOutputObservation):
+            text = obs_prefix + obs.content
+            text += (
+                f'\n[Command {obs.command_id} finished with exit code {obs.exit_code}]'
+            )
+            return Message(role='user', content=[TextContent(text=text)])
+        elif isinstance(obs, IPythonRunCellObservation):
+            text = obs_prefix + obs.content
+            return Message(role='user', content=[TextContent(text=text)])
+        elif isinstance(obs, ErrorObservation):
+            text = obs_prefix + obs.content
+            text += '\n[Error occurred in processing last action]'
+            return Message(role='user', content=[TextContent(text=text)])
+        elif isinstance(obs, UserRejectObservation):
+            text = obs_prefix + obs.content
+            text += '\n[Last action has been rejected by the user]'
+            return Message(role='user', content=[TextContent(text=text)])
+        else:
+            raise ValueError(f'Unknown observation type: {type(obs)}')
diff --git a/openhands/agenthub/searcher_agent/prompt.py b/openhands/agenthub/searcher_agent/prompt.py
new file mode 100644
index 000000000000..bfc9bc612647
--- /dev/null
+++ b/openhands/agenthub/searcher_agent/prompt.py
@@ -0,0 +1,69 @@
+# General Description, the goal is to devise a manager that is able to iterate if the solution has not been found yet.
+# In order to successfully fix an issue there are two phases:
+# 1. Exploring the codebase, finding the root cause of the issue.
+# 2. Implementing the solution.
+# Then the manager needs to check if the issue has been fixed, if not, it needs to iterate.
+general_description = """
+You are a detail-oriented AI, an expert in searching through files and code.
+You are also an expert in summarising code and its purpose.
+As a detail-oriented AI, you MUST always read more and more code until you are sure you have found
+all the information you need.
+
+Your goal is to gather information about the codebase to help the programmer fix the issue.
+Here is the task you are trying to complete:
+%(task)s
+
+IMPORTANT: YOU SHOULD NEVER TRY TO IMPLEMENT A SOLUTION. YOUR ONLY GOAL IS TO GATHER INFORMATION.
+As an expert in searching through files and code, you have been equipped with a set of tools
+that will help you gather information about the codebase:
+- You can execute bash commands wrapped with <execute_bash>, e.g. <execute_bash> ls </execute_bash>.
+- If a bash command returns exit code `-1`, this means the process is not yet finished.
+- You must then send a second <execute_bash>. The second <execute_bash> can be empty
+  (which will retrieve any additional logs), or it can contain text to be sent to STDIN of the running process,
+  or it can contain the text `ctrl+c` to interrupt the process.
+- For commands that may run indefinitely, the output should be redirected to a file and the command run
+  in the background, e.g. <execute_bash> python3 app.py > server.log 2>&1 & </execute_bash>
+- If a command execution result says "Command timed out. Sending SIGINT to the process",
+  you should retry running the command in the background.
+
+You should ONLY `run` commands that have no side-effects, like `ls` and `grep`.
+
+Your manager gave you a suggested approach that you should follow:
+%(suggested_approach)s
+
+Follow the suggested approach to gather information about the codebase.
+When you think you have gathered enough information, generate a JSON with the following format:
+<finish>
+[
+  {
+    "summary": "<a detailed summary of a relevant file>",
+    "location_of_the_file": "<path to the file>",
+    "functions_of_interest": [
+      {
+        "name": "<name of the function>",
+        "summary": "<a detailed summary of the function>",
+        "calls_to_this_function": ["<list of functions that call this function>"],
+        "is_called_by_these_functions": ["<list of functions that are called by this function>"]
+      },
+    ]
+  }
+]
+</finish>
+
+IMPORTANT: Every entry in the JSON MUST be relevant to the task.
+IMPORTANT: The JSON MUST be contained inside <finish> and </finish> tags.
+IMPORTANT: You MUST have at least one file in the response.
+
+"""
+
+
+def get_prompt(task: str, suggested_approach: str) -> str:
+    formatted_prompt = (general_description) % {
+        'task': task,
+        'suggested_approach': suggested_approach,
+    }
+
+    # Add instruction to not include json formatting
+    formatted_prompt += '\n\nIMPORTANT: Do not include ```json at the start or ``` at the end of your response. Just return the raw JSON list.'
+
+    return formatted_prompt
diff --git a/openhands/agenthub/supervisor_agent/agent.py b/openhands/agenthub/supervisor_agent/agent.py
index 61ef4bed8fc0..034a3ba1b7e1 100644
--- a/openhands/agenthub/supervisor_agent/agent.py
+++ b/openhands/agenthub/supervisor_agent/agent.py
@@ -1,15 +1,15 @@
-import copy
 import logging
-from typing import Dict, List
+from typing import Any, Dict, List, Literal, Union
 
 from openhands.agenthub.supervisor_agent.prompt import (
-    adjust_milestones,
-    get_initial_prompt,
+    TASK_TYPE_ISSUE,
+    get_prompt,
 )
 from openhands.controller.agent import Agent
 from openhands.controller.state.state import State
 from openhands.core.config import AgentConfig
 from openhands.core.message import Message, TextContent
+from openhands.core.schema.action import ActionType
 from openhands.core.utils import json
 from openhands.events.action import Action, AgentDelegateAction, AgentFinishAction
 from openhands.events.action.agent import AgentRejectAction
@@ -25,10 +25,13 @@ class SupervisorAgent(Agent):
     """
 
     current_delegate: str = ''
-    sub_goals: List[Dict[str, str]] = []
-    current_goal_index: int = 0
-    summary: str = ''
+    suggested_approaches: List[Dict[str, List[str]]] = []
+    suggested_approach_index: int = -1  # -1 Because we increment it before using it
+    results: Dict[str, List[Any]] = {'search': [], 'code': []}
+    condensed_information: str = ''
+    does_it_needs_a_test: str = ''
     task: str = ''
+    phase: Literal['search', 'summary', 'code'] = 'search'
 
     def __init__(self, llm: LLM, config: AgentConfig):
         """Initialize the Supervisor Agent with an LLM
@@ -46,120 +49,91 @@ def step(self, state: State) -> Action:
         self.logger.debug('Starting step with state: %s', state)
         self.logger.debug('LLM config: %s', self.llm_config)
 
-        if not self.sub_goals:
-            self.initialize_sub_goals(state)
+        if not self.suggested_approaches:
+            self.suggested_approaches = self.get_suggested_approaches(state)
+        self.suggested_approach_index += 1
 
-        if self.current_delegate == '':
-            self.current_delegate = 'CodeActAgent'
+        last_observation = state.history.get_last_observation()
+        if (
+            isinstance(last_observation, AgentDelegateObservation)
+            and last_observation.outputs.get('action', '') == ActionType.FINISH
+        ):
+            self.results[self.phase].append(last_observation.outputs.get('output', ''))
+
+        if len(self.results[self.phase]) < len(self.suggested_approaches):
+            # Delegate to the SearcherAgent as we need to gather more information
             return self.delegate_to_agent(
-                'CodeActAgent', self.construct_task_details(self.prepare_current_task())
+                'SearcherAgent',
+                self.task,
+                self.suggested_approaches[self.suggested_approach_index].get(
+                    'suggested_approach', []
+                ),
             )
 
-        elif self.current_delegate == 'CodeActAgent':
-            return self.handle_code_act_agent(state)
+        if self.phase == 'search':
+            # We don't change the phase until we have the condensed information
+            condensed_information = self.ask_llm(
+                self.task, '2', json.dumps(self.results['search'])
+            )[0]
+            if condensed_information.get('summary', '') != '':
+                self.phase = 'summary'
+                self.condensed_information = condensed_information.get('summary', '')
+            else:
+                suggested_approach: str | list[str] = condensed_information.get(
+                    'suggested_approach', []
+                )
+                self.results['search'].append(suggested_approach)
+                return self.delegate_to_agent(
+                    'SearcherAgent', self.task, suggested_approach
+                )
+
+        if self.phase == 'summary':
+            # Now we have to judge if this issue requires a test or not before fixing it
+            does_it_needs_a_test = self.ask_llm(
+                self.task, 'code', self.condensed_information
+            )[0]
+            if does_it_needs_a_test.get('suggested_approach', '') == TASK_TYPE_ISSUE:
+                self.phase = 'code'
+            else:
+                self.phase = 'code'
+
+        # WIP: Implement the code phase
 
         return AgentFinishAction()
 
-    def initialize_sub_goals(self, state: State):
-        self.logger.debug('No sub-goals found, breaking down task.')
+    def get_suggested_approaches(self, state: State):
+        self.logger.debug('No suggested approaches found, breaking down task.')
         self.task, _ = state.get_current_user_intent()
-        self.sub_goals = self.break_down_task(self.task)
-        self.logger.debug('Sub-goals: %s', self.sub_goals)
-        if not self.sub_goals:
+        suggested_approaches = self.ask_llm(self.task, 'search')
+        self.logger.debug('Suggested approaches: %s', self.suggested_approaches)
+        if not suggested_approaches:
             return AgentRejectAction()
+        return suggested_approaches
 
-    def delegate_to_agent(self, agent_name: str, task: str) -> AgentDelegateAction:
-        self.logger.debug(f'Delegating to agent: {agent_name}')
-
-        return AgentDelegateAction(agent=agent_name, inputs={'task': task})
-
-    def handle_code_act_agent(self, state: State) -> Action:
-        self.logger.debug("Current delegate is 'CodeActAgent'.")
-        last_observation = state.history.get_last_observation()
-
-        if not isinstance(last_observation, AgentDelegateObservation):
-            raise Exception('Last observation is not an AgentDelegateObservation')
-
-        if last_observation.outputs.get('action', '') == 'reject':
-            return self.handle_rejection(last_observation)
-
-        return self.handle_success(last_observation)
-
-    def handle_rejection(
-        self, last_observation: AgentDelegateObservation
+    def delegate_to_agent(
+        self, agent_name: str, task: str, suggested_approach: Union[str, List[str]]
     ) -> AgentDelegateAction:
-        self.logger.debug('No summary found, creating adjustment prompt.')
-        reason = getattr(last_observation, 'reason', '')
-        prompt = self.create_adjustment_prompt(reason)
-        self.sub_goals = self.get_sub_goals_from_llm(prompt)
-        current_task = self.prepare_current_task()
-        return self.delegate_to_agent(
-            'CodeActAgent', self.construct_task_details(current_task)
-        )
-
-    def handle_success(self, last_observation: AgentDelegateObservation) -> Action:
-        summary = last_observation.outputs.get('summary', '')
-        self.append_to_summary(summary)
-        self.current_goal_index += 1
-
-        if self.current_goal_index < len(self.sub_goals):
-            current_task = self.prepare_current_task()
-            task_details = self.construct_task_details(current_task)
-            return self.delegate_to_agent('CodeActAgent', task_details)
-
-        return AgentFinishAction()
-
-    def prepare_current_task(self) -> Dict[str, str]:
-        current_task = copy.deepcopy(self.sub_goals[self.current_goal_index])
-        current_task['summary'] = self.summary if self.summary else ''
-        return current_task
-
-    def construct_task_details(self, current_task: Dict[str, str]) -> str:
-        task_details = (
-            f"Task: {self.task}\n\n"
-            f"Next Subtask: {current_task['task']}\n"
-            f"Suggested Approach: {current_task['suggested_approach']}\n"
-            f"Important Details: {current_task['important_details']}"
+        self.logger.debug(f'Delegating to agent: {agent_name}')
+        # Join the list of strings with newlines if it's a list
+        approach = (
+            '\n'.join(suggested_approach)
+            if isinstance(suggested_approach, list)
+            else suggested_approach
         )
-        if self.summary:
-            task_details = f'Progress so far: {self.summary}\n\n' + task_details
-        return task_details
-
-    def break_down_task(self, task: str) -> List[Dict[str, str]]:
-        # Generate the initial prompt for breaking down the task
-        prompt = get_initial_prompt(task)
-        # Get the sub-goals from the language model using the generated prompt
-        return self.get_sub_goals_from_llm(prompt)
-
-    def should_interrupt(self, observation) -> bool:
-        # Logic to determine if the task should be interrupted
-        return False  # Placeholder
-
-    def summarize_history(self, history) -> str:
-        # Logic to summarize the history
-        return 'summary'  # Placeholder
-
-    def provide_guidance(self, state: State) -> Action:
-        # Logic to provide high-level guidance
-        return AgentFinishAction()  # Placeholder
-
-    def create_adjustment_prompt(self, reason: str) -> str:
-        return adjust_milestones(
-            self.sub_goals,
-            self.sub_goals[self.current_goal_index],
-            reason,
-            self.summary,
-            self.task,
+        return AgentDelegateAction(
+            agent=agent_name, inputs={'task': task, 'suggested_approach': approach}
         )
 
-    def get_sub_goals_from_llm(self, prompt: str) -> List[Dict[str, str]]:
+    def ask_llm(
+        self, task: str, phase: str, search_results: str = ''
+    ) -> List[Dict[str, str]]:
+        prompt = get_prompt(task, phase, search_results)
+        return self.get_response(prompt)
+
+    def get_response(self, prompt: str) -> List[Dict[str, str]]:
         content = [TextContent(text=prompt)]
         message = Message(role='user', content=content)
         response = self.llm.completion(
             messages=self.llm.format_messages_for_llm(message)
         )
         return json.loads(response['choices'][0]['message']['content'])
-
-    def append_to_summary(self, summary: str):
-        """Appends the milestone name and summary to the agent's summary state."""
-        self.summary += f'{summary}\n\n'
diff --git a/openhands/agenthub/supervisor_agent/prompt.py b/openhands/agenthub/supervisor_agent/prompt.py
index e4e0eaedfa22..4d8e68a92df9 100644
--- a/openhands/agenthub/supervisor_agent/prompt.py
+++ b/openhands/agenthub/supervisor_agent/prompt.py
@@ -1,57 +1,153 @@
-from typing import Dict, List
-
-from openhands.core.utils import json
-
 HISTORY_SIZE = 20
 
-# General Description
+# General Description, the goal is to devise a manager that is able to iterate if the solution has not been found yet.
+# In order to successfully fix an issue there are two phases:
+# 1. Exploring the codebase, finding the root cause of the issue.
+# 2. Implementing the solution.
+# Then the manager needs to check if the issue has been fixed, if not, it needs to iterate.
 general_description = """
 You are a strategic planner AI in a software development team. You have a team of agents
-who will complete the tasks you give them. Each agent is an expert in a specific area.
-You MUST think CAREFULLY how to complete the task assigned to you.
+who will complete the tasks you give them. Each agent is an expert in a specific area,
+but it can only focus on one very specific sub-task at a time.
 
-You've been given the following task:
+Your goal is to complete the following task:
 %(task)s
 
-As a strategic manager, you create a plan with different sub-tasks and delegate the tasks to your team.
-At your disposal, you have a team of agents who will complete tasks for you. However, those agents only focus on the details.
-They CANNOT see the big picture.
-They need you to define self-contained tasks, that are easy for them to understand and complete.
+This task is very complex, it requires careful planning and thinking.
+In order to properly complete the task, there are two phases:
+- Search: exploring the codebase, finding the relevant details. (e.g. what is the root cause of the issue?)
+- Summary: summarising the information you have gathered.
+- Code: implementing the solution. (e.g. how to fix the issue?)
 
+As a strategic manager, your goal is to create a suggested approach for phase %(phase)s.
+
+## Detailed Suggested Approaches
+Generate several detailed suggested approaches that will be used by your agents to complete the task.
+Each agent will be assigned one of the suggested approaches and will bring you back feedback.
+So, be creative and think of as many different approaches as possible.
+You are trying to HELP the agents complete the task, you MUST be AS DETAILED AS POSSIBLE.
 """
 
-# Initial Prompt
-initial_prompt = """
-## Plan
-Your goal is to create a high-level plan, a list of subtasks that will bring you closer to the completion of the task. Remember to think
-CAREFULLY about how to complete the task. With each subtask, you MUST provide a "suggested approach".
-Think, step by step, how you would complete the subtask. Then provide that as the suggested approach.
-Try to be as detailed as possible, your goal is to HELP the agent finish the subtask as soon as possible.
 
-You MAY provide a list of "important details" for each subtask. These are details that the agent MUST consider when completing the subtask.
+condense_information_prompt = """
+Previously, your agents were tasked to gather information about the codebase.
+They have now returned their findings.
+
+As a strategic manager, your job is to look CAREFULLY at the information they have gathered.
+You need to make sure you have a good understanding of the codebase, and the potential solutions
+to the task.
+
+## Information Gathered
+%(search_results)s
+
+## Summary
+Do you think you have enough information to complete the task?
+If not, you need to request more information from the agents.
+Return a list of 1 JSON describing what extra information you would need and the suggested approach to gather that information.
+[
+    {
+        "suggested_approach": ["<suggested approach to gather the missing information>"]
+    }
+]
+If you have enough information, you need to summarise the information you have gathered.
+How would you explain this to a new joiner to the team?
+Where would you point them to?
+Provide a detailed step by step guide.
+Remember, the agents DON'T have access to the internet. Every task must be conducted OFFLINE.
+The agents have cloned the repo, so they can open files, browse the code, interact with it...
+In the information gathered, there might be some repeated information, or some information
+that is actually not relevant.
+You need to be able to distinguish what is relevant, and what is not.
+In the information you have gathered, there might be file names, function names, class names. You MUST include
+them in the summary, so the agents know where to look.
+Generate a list of 1 JSON with the following format:
+[
+    {
+        "summary": ["<step by step guide>"]
+    }
+]
+
+IMPORTANT: Be VERY VERY VERY SPECIFIC.
+IMPORTANT: Include the file names, function names, class names, code blocks, in the step by step guide.
+IMPORTANT: Generate as many steps as possible.
+"""
+
+# Constants for task type choices
+TASK_TYPE_ISSUE = 'yes, the task is an issue that needs to be replicated'
+TASK_TYPE_FEATURE = 'no, the task is a new feature that needs to be implemented'
+
+does_it_needs_a_test_prompt = (
+    """
+As a strategic manager, you need to judge if the task is an issue that needs to be replicated first
+or if it is a new feature that just needs to be implemented.
+
+Your agents have already gathered information about the codebase.
 
-ONLY generate tasks that are necessary to complete the task.
+## Information Gathered
+%(search_results)s
+
+Think CAREFULLY before answering.
+What do you think is the best course of action?
+IMPORTANT: You MUST return a list of 1 JSON with the following format:
+[
+    {
+        "suggested_approach": ["<Choose ONE: either '"""
+    + TASK_TYPE_ISSUE
+    + """' OR '"""
+    + TASK_TYPE_FEATURE
+    + """'>"]
+    }
+]
+
+IMPORTANT: You MUST choose one of the two options.
+"""
+)
 
+initial_prompt = """
 You MUST ONLY generate a list of JSONs:
 
 [
     {
-      "task": "<Task 1 name>",
-      "suggested_approach": "<suggested approach>",
-      "important_details": "<important details>"
+      "suggested_approach": ["<suggested approach>"]
     },
     {
-      "task": "<Task 2 name>",
-      "suggested_approach": "<suggested approach>",
-      "important_details": "<important details>"
+      "suggested_approach": ["<suggested approach>"]
     },
 ]
 
-The tasks MUST be generated in order, they MUST NOT depend on future tasks or previous tasks. They MUST be independent.
-You MUST generate at least 1 task. The last task MUST be the implementation task. You WILL NOT need a test file.
+Suggested approaches MUST be independent.
+You MUST generate at least 1 suggested approach.
+IMPORTANT: the agents DON'T have access to the internet. Every task must be conducted OFFLINE.
+The agents have cloned the repo, so they can open files, browse the code, interact with it...
+The goal of phase 1, exploring the codebase, finding the relevant details is ONLY to collect information.
+Be as HELPFUL and DETAILED as possible.
+Use the suggested approach to guide the agents in their exploration of the codebase.
+They MUST interact with the environment:
+- Open as many files as needed to gather as much information as possible.
+- Read every piece of code that might be relevant to the task, summarise what does it do.
+- Decide which functions are important to the task, understand how they are used and how they are called.
+
+Remember that the agents can use a Python environment with <execute_ipython>, e.g.:
+<execute_ipython>
+print("Hello World!")
+</execute_ipython>
+
+They can execute bash commands wrapped with <execute_bash>, e.g. <execute_bash> ls </execute_bash>.
+If a bash command returns exit code `-1`, this means the process is not yet finished.
+They must then send a second <execute_bash>. The second <execute_bash> can be empty
+(which will retrieve any additional logs), or it can contain text to be sent to STDIN of the running process,
+or it can contain the text `ctrl+c` to interrupt the process.
+
+For commands that may run indefinitely, the output should be redirected to a file and the command run
+in the background, e.g. <execute_bash> python3 app.py > server.log 2>&1 & </execute_bash>
+If a command execution result says "Command timed out. Sending SIGINT to the process",
+the assistant should retry running the command in the background.
 
-For example:
-User prompt:
+Be VERY VERY SPECIFIC.
+
+---- START OF EXAMPLE ----
+
+## TASK
 
 "
 Enable quiet mode/no-verbose in CLI for use in pre-commit hook There seems to be only an option to increase the level of verbosity when using
@@ -61,98 +157,175 @@
 long list of fixes that are being applied to the SQL files, which can get quite verbose.
 "
 
-Your response:
+## YOUR RESPONSE:
 
 [
-    {
-        "task": "Research SQLFluff CLI verbosity options",
-        "suggested_approach": "Investigate the current SQLFluff CLI documentation and source code to understand how verbosity levels are currently implemented. Identify if there are any existing flags or settings that can be adjusted to reduce verbosity.",
-        "important_details": "Focus on the 'fix' command and any related verbosity settings. Document any findings that could be useful for implementing a quiet mode."
-    },
-    {
-        "task": "Implement the quiet mode feature",
-        "suggested_approach": "Modify the SQLFluff CLI codebase to add the new quiet mode feature. Implement the necessary changes in the code to support this feature and ensure it can be activated via a command-line flag.",
-        "important_details": "Write unit tests to verify that the quiet mode works as expected and does not affect other CLI functionalities."
-    }
+  {
+    "suggested_approach": [
+      "1. Open the SQLFluff codebase and navigate to the CLI module, likely located in 'src/sqlfluff/cli/'.",
+      "2. Locate the file responsible for parsing command-line arguments, such as 'commands.py' or 'cli.py'.",
+      "3. Examine how the '--verbose' flag is implemented in the code.",
+      "4. Identify if there is an existing '--quiet' or '--no-verbose' option.",
+      "5. Understand how verbosity levels are set and managed within the CLI code.",
+      "6. Look for any variables or settings that control the default verbosity level.",
+      "7. Determine how the '--verbose' flag increases verbosity and see if a similar mechanism can decrease verbosity.",
+      "8. Note down any functions or methods that output information to the console.",
+      "9. Identify how these functions can be controlled via verbosity levels.",
+      "10. Summarize findings and consider how to implement a '--quiet' flag."
+    ]
+  },
+  {
+    "suggested_approach": [
+      "1. Investigate the logging configuration in SQLFluff, possibly located in 'src/sqlfluff/core/logger.py' or similar.",
+      "2. Understand how logging levels are set (e.g., DEBUG, INFO, WARNING, ERROR).",
+      "3. Examine if the logging levels are affected by CLI arguments.",
+      "4. Identify where in the code the logging configuration is initialized based on user input.",
+      "5. Check if there is a way to adjust the logging level via a CLI option.",
+      "6. Determine if adding a '--quiet' flag can set the logging level to WARNING or ERROR to suppress INFO messages.",
+      "7. Note the changes needed in the logging setup to support a quiet mode.",
+      "8. Identify all logging statements that may need to respect the new logging level.",
+      "9. Consider the impact on existing functionality and ensure that critical messages are still displayed.",
+      "10. Summarize how logging can be adjusted to implement a quiet mode."
+    ]
+  },
+  {
+    "suggested_approach": [
+      "1. Analyze how output to the console is handled throughout the codebase.",
+      "2. Identify the functions used for outputting messages, such as 'click.echo', 'print', or custom wrapper functions.",
+      "3. Trace where these output functions are called in the code, especially during 'sqlfluff fix' execution.",
+      "4. Determine if there is a centralized output function or if output is scattered across multiple functions.",
+      "5. Assess whether output functions can be modified to check a verbosity level before printing.",
+      "6. Consider creating or modifying a wrapper function that respects a verbosity or quiet setting.",
+      "7. Identify any messages that should always be displayed, regardless of verbosity settings (e.g., errors).",
+      "8. Note the locations in the code where changes need to be made to control output.",
+      "9. Evaluate the feasibility of implementing a quiet mode by adjusting output functions.",
+      "10. Summarize the steps required to control output at the source."
+    ]
+  },
+  {
+    "suggested_approach": [
+      "1. Explore the configuration options available in SQLFluff by examining the configuration parser code, possibly in 'src/sqlfluff/core/config.py'.",
+      "2. Look for existing configuration parameters related to verbosity or output control.",
+      "3. Determine how configuration files (like '.sqlfluff') are parsed and applied.",
+      "4. Assess if a new configuration option can be introduced to control verbosity levels.",
+      "5. Identify how this configuration option can be read and applied during runtime.",
+      "6. Check if the CLI options can override configuration file settings for verbosity.",
+      "7. Map out the code changes required to implement and support a new configuration option.",
+      "8. Ensure that the new configuration integrates smoothly with existing settings.",
+      "9. Consider user documentation and how users would be informed about the new option.",
+      "10. Summarize the process of adding a verbosity control via configuration files."
+    ]
+  },
+  {
+    "suggested_approach": [
+      "1. Examine the implementation of the 'sqlfluff fix' command to understand its workflow.",
+      "2. Identify where the command generates output and how that output is formatted.",
+      "3. Determine if 'sqlfluff fix' has different output modes or formats based on context.",
+      "4. Check if the command detects when it's running in a pre-commit hook or similar environment.",
+      "5. Consider if output suppression can be contextually applied when running in certain environments.",
+      "6. Identify any existing mechanisms for output control based on execution context.",
+      "7. Explore how the 'black' formatter handles output suppression in pre-commit hooks.",
+      "8. Analyze if similar techniques can be applied within SQLFluff's codebase.",
+      "9. Note any dependencies or external factors that influence output generation.",
+      "10. Summarize how context-aware output control can be implemented."
+    ]
+  }
 ]
-"""
-
-adjustment_prompt = """
-
-    This is the current active plan that your agents are working on:
-    %(milestones)s
-
-    And this is the current subtask that your agents are working on:
-    ## Current subtask
-    subtask: %(milestone_task)s
-    Suggested Approach: %(milestone_suggested_approach)s
-    Important Details: %(milestone_important_details)s
 
-    However, it seems that the current subtask is not being completed successfully.
-    Because of the following reason: %(reason)s
 
-    You have the following contextual information that has been gathered up to this point.
-    This information MIGHT help you adjust the plan:
-    %(summary)s
+---- END OF EXAMPLE ----
 
-    ## Task
-    As a strategic manager, you must reflect on the failed subtask and decide on the necessary adjustments. Consider the following:
 
-    1. Analyze the reason for failure and determine if the suggested approach or important details need modification.
-    2. Decide if the failed subtask should be split into smaller, more manageable tasks.
-    3. Consider if new plan need to be added to address any gaps in the plan.
-    4. Update the remaining plan to ensure the overall plan remains feasible and effective.
+--- START OF EXAMPLE 2 ---
 
-    You MUST NOT change the task you were given.
+## TASK
+"
+ModelChain.prepare_inputs can succeed with missing dhi From the docstring for `ModelChain.prepare_inputs()`
+I believe the method should fail if `weather` does not have a `dhi` column. The validation checks for `'ghi'` twice,
+but not `'dhi`' https://github.com/pvlib/pvlib-python/blob/11c356f9a89fc88b4d3ff368ce1aae170a97ebd7/pvlib/modelchain.py#L1136
+"
 
-    You MUST make changes to the current subtask or to the ones AFTER. In NO case you can change the ones BEFORE.
-    Generate ONLY a list of JSONs. Do NOT generate any markdown or comments.
-    """
+## YOUR RESPONSE:
 
+[
+  {
+    "suggested_approach": [
+      "1. Open the file pvlib/modelchain.py and locate the ModelChain.prepare_inputs method. Carefully read through the method's code, focusing on the section where it validates the weather DataFrame columns, specifically around line 1136.",
+      "2. Identify the validation checks for the weather DataFrame. Note whether it checks for the presence of 'dhi' or mistakenly checks for 'ghi' twice.",
+      "3. Examine the docstring of ModelChain.prepare_inputs to understand the expected behavior when dhi is missing from the weather data.",
+      "4. Investigate any helper functions called within prepare_inputs that handle irradiance data, such as methods for inferring missing components.",
+      "5. Review the unit tests related to prepare_inputs in pvlib/tests/test_modelchain.py to see if cases with missing dhi are covered.",
+      "6. Use the Python environment to simulate calling prepare_inputs with weather data missing the dhi column and observe the outcome.",
+      "<execute_ipython>",
+      "import pvlib",
+      "from pvlib import modelchain, location, pvsystem",
+      "import pandas as pd",
+      "mc = modelchain.ModelChain(pvsystem.PVSystem(), location.Location(32.2, -110.9))",
+      "weather = pd.DataFrame({'ghi': [1000], 'dni': [800]})",
+      "mc.prepare_inputs(weather)",
+      "</execute_ipython>",
+      "7. Document any discrepancies between the code and the documentation, and note any unexpected behaviors."
+    ]
+  },
+  {
+    "suggested_approach": [
+      "1. Generate a flowchart of the prepare_inputs method to understand its logic and how it processes the weather DataFrame.",
+      "2. Open pvlib/modelchain.py and trace each step within prepare_inputs, paying attention to how it handles missing data.",
+      "3. Look for any conditional statements that manage cases where dhi is not provided and see if alternative calculations are performed or if an error is raised.",
+      "4. Explore related methods like complete_irradiance or irradiance.get_total_irradiance to see how missing components are handled.",
+      "5. Test different weather DataFrame scenarios in the Python environment to observe how prepare_inputs behaves with various missing columns.",
+      "<execute_ipython>",
+      "import pvlib",
+      "from pvlib import modelchain, location, pvsystem",
+      "import pandas as pd",
+      "mc = modelchain.ModelChain(pvsystem.PVSystem(), location.Location(32.2, -110.9))",
+      "# Weather data missing 'dhi'",
+      "weather_missing_dhi = pd.DataFrame({'ghi': [1000], 'dni': [800]})",
+      "mc.prepare_inputs(weather_missing_dhi)",
+      "# Weather data missing 'ghi'",
+      "weather_missing_ghi = pd.DataFrame({'dhi': [200], 'dni': [800]})",
+      "mc.prepare_inputs(weather_missing_ghi)",
+      "</execute_ipython>",
+      "6. Record the outcomes and any exceptions raised to determine if the method behaves as intended."
+    ]
+  },
+  {
+    "suggested_approach": [
+      "1. Analyze the git commit history for modelchain.py to identify when the validation issue was introduced.",
+      "<execute_bash>",
+      "cd pvlib-python",
+      "git log -L 1136,1140 /modelchain.py",
+      "</execute_bash>",
+      "2. Review the changes in each commit affecting the validation checks in prepare_inputs.",
+      "3. Open the relevant commits and examine the differences in the validation code.",
+      "4. Check for any related issues or pull requests in the repository's local clone that discuss missing dhi validation.",
+      "5. Look into the test coverage reports (if available locally) to see if the validation logic is adequately tested.",
+      "6. Summarize findings on whether the issue is a recent regression or an existing oversight."
+    ]
+  }
+]
 
-def get_initial_prompt(task: str) -> str:
-    formatted_prompt = (general_description + initial_prompt) % {
-        'task': task,
-    }
+--- END OF EXAMPLE 2 ---
 
-    # Add instruction to not include json formatting
-    formatted_prompt += '\n\nIMPORTANT: Do not include ```json at the start or ``` at the end of your response. Just return the raw JSON list.'
+--- YOUR TURN ---
 
-    return formatted_prompt
+## TASK
+%(task)s
 
+## YOUR RESPONSE:
+"""
 
-def adjust_milestones(
-    milestones: List[Dict],
-    subtask: Dict[str, str],
-    reason: str,
-    summary: str,
-    task: str,
-) -> str:
-    """Adjusts the milestones based on a failed subtask and its reason.
 
-    Parameters:
-    - milestones (List[Dict]): The current list of milestones.
-    - subtask (Dict): The subtask that was not completed successfully.
-    - reason (str): The reason provided for the failure.
-    - summary (str): A summary of everything up to this point.
-    - task (str): The user's task.
+def get_prompt(task: str, phase: str, search_results: str = '') -> str:
+    if phase == 'search':
+        base_prompt = general_description + initial_prompt
+    elif phase == 'summary':
+        base_prompt = general_description + condense_information_prompt
 
-    Returns: A prompt for the strategic manager agent to self-reflect and adjust the milestones.
-    """
-    # Extract values from the subtask dictionary
-    milestone_task = subtask['task']
-    milestone_suggested_approach = subtask['suggested_approach']
-    milestone_important_details = subtask['important_details']
-
-    # Get the formatted prompt
-    formatted_prompt = (general_description + adjustment_prompt) % {
-        'milestones': json.dumps(milestones),
-        'reason': reason,
-        'summary': summary,
+    formatted_prompt = base_prompt % {
         'task': task,
-        'milestone_task': milestone_task,
-        'milestone_suggested_approach': milestone_suggested_approach,
-        'milestone_important_details': milestone_important_details,
+        'phase': phase,
+        'search_results': search_results,
     }
 
     # Add instruction to not include json formatting

From 04c56c65449b1793cab879a2e5d54d056e7db37c Mon Sep 17 00:00:00 2001
From: AlexCuadron <alex.cl.2000@gmail.com>
Date: Mon, 28 Oct 2024 18:48:14 +0100
Subject: [PATCH 12/18] fix

---
 openhands/runtime/builder/docker.py | 1 -
 1 file changed, 1 deletion(-)

diff --git a/openhands/runtime/builder/docker.py b/openhands/runtime/builder/docker.py
index 2c5a965b1dd0..5a22302d61d5 100644
--- a/openhands/runtime/builder/docker.py
+++ b/openhands/runtime/builder/docker.py
@@ -70,7 +70,6 @@ def build(
             f'--build-arg=OPENHANDS_RUNTIME_BUILD_TIME={datetime.datetime.now().isoformat()}',
             f'--tag={target_image_hash_name}',
             '--load',
-            '--platform=linux/amd64',
         ]
 
         # Include the platform argument only if platform is specified

From 399f19ebf539e4cb44763e168b2ab25751ce7016 Mon Sep 17 00:00:00 2001
From: AlexCuadron <alex.cl.2000@gmail.com>
Date: Thu, 31 Oct 2024 11:14:16 -0700
Subject: [PATCH 13/18] MAS

---
 openhands/agenthub/__init__.py                |   2 +
 .../agenthub/codeact_agent/codeact_agent.py   |  11 +
 .../agenthub/searcher_agent/action_parser.py  |  19 +-
 openhands/agenthub/searcher_agent/agent.py    | 162 +++++----
 openhands/agenthub/searcher_agent/prompt.py   |  99 +++++-
 openhands/agenthub/supervisor_agent/agent.py  |  99 ++++--
 openhands/agenthub/tester_agent/__init__.py   |   4 +
 .../agenthub/tester_agent/action_parser.py    | 158 +++++++++
 openhands/agenthub/tester_agent/agent.py      | 201 +++++++++++
 openhands/agenthub/tester_agent/prompt.py     | 329 ++++++++++++++++++
 10 files changed, 962 insertions(+), 122 deletions(-)
 create mode 100644 openhands/agenthub/tester_agent/__init__.py
 create mode 100644 openhands/agenthub/tester_agent/action_parser.py
 create mode 100644 openhands/agenthub/tester_agent/agent.py
 create mode 100644 openhands/agenthub/tester_agent/prompt.py

diff --git a/openhands/agenthub/__init__.py b/openhands/agenthub/__init__.py
index 96c766124989..1ec266ce7501 100644
--- a/openhands/agenthub/__init__.py
+++ b/openhands/agenthub/__init__.py
@@ -16,6 +16,7 @@
     planner_agent,
     searcher_agent,
     supervisor_agent,
+    tester_agent,
 )
 
 __all__ = [
@@ -27,6 +28,7 @@
     'browsing_agent',
     'searcher_agent',
     'supervisor_agent',
+    'tester_agent',
 ]
 
 for agent in all_microagents.values():
diff --git a/openhands/agenthub/codeact_agent/codeact_agent.py b/openhands/agenthub/codeact_agent/codeact_agent.py
index d1f67eae9c2c..4dbb1503dcc0 100644
--- a/openhands/agenthub/codeact_agent/codeact_agent.py
+++ b/openhands/agenthub/codeact_agent/codeact_agent.py
@@ -10,6 +10,7 @@
 from openhands.controller.agent import Agent
 from openhands.controller.state.state import State
 from openhands.core.config import AgentConfig
+from openhands.core.config.llm_config import LLMConfig
 from openhands.core.logger import openhands_logger as logger
 from openhands.core.message import ImageContent, Message, TextContent
 from openhands.events.action import (
@@ -81,6 +82,16 @@ def __init__(
         Parameters:
         - llm (LLM): The llm to be used by this agent
         """
+
+        llm_config = LLMConfig(
+            model='litellm_proxy/claude-3-5-sonnet-20241022',
+            api_key='REDACTED',
+            temperature=0.0,
+            base_url='https://llm-proxy.app.all-hands.dev',
+        )
+        llm = LLM(llm_config)
+        # TODO: Remove this once we have a real AgentConfig
+        config = AgentConfig(llm_config='o1-mini')
         super().__init__(llm, config)
         self.reset()
 
diff --git a/openhands/agenthub/searcher_agent/action_parser.py b/openhands/agenthub/searcher_agent/action_parser.py
index 54ad267cad8b..46846641b8df 100644
--- a/openhands/agenthub/searcher_agent/action_parser.py
+++ b/openhands/agenthub/searcher_agent/action_parser.py
@@ -39,15 +39,12 @@ def parse_response(self, response) -> str:
         action = response.choices[0].message.content
         if action is None:
             return ''
-        for lang in ['bash', 'ipython', 'browse']:
-            # special handling for DeepSeek: it has stop-word bug and returns </execute_ipython instead of </execute_ipython>
+        for lang in ['bash', 'ipython']:
             if f'</execute_{lang}' in action and f'</execute_{lang}>' not in action:
                 action = action.replace(f'</execute_{lang}', f'</execute_{lang}>')
 
             if f'<execute_{lang}>' in action and f'</execute_{lang}>' not in action:
                 action += f'</execute_{lang}>'
-        if '<file_edit' in action and '</file_edit>' not in action:
-            action += '</file_edit>'
         return action
 
     def parse_action(self, action_str: str) -> Action:
@@ -68,14 +65,16 @@ def __init__(
         self.finish_command = None
 
     def check_condition(self, action_str: str) -> bool:
-        self.finish_command = re.search(r'<finish>.*</finish>', action_str, re.DOTALL)
+        self.finish_command = re.search(
+            r'<finish>(.*?)</finish>', action_str, re.DOTALL
+        )
         return self.finish_command is not None
 
     def parse(self, action_str: str) -> Action:
         assert (
             self.finish_command is not None
         ), 'self.finish_command should not be None when parse is called'
-        output = action_str.replace(self.finish_command.group(0), '').strip()
+        output = self.finish_command.group(1).strip()
         outputs = {'output': output}
         return AgentFinishAction(outputs=outputs)
 
@@ -114,9 +113,7 @@ class SearcherAgentActionParserIPythonRunCell(ActionParser):
     - IPythonRunCellAction(code) - IPython code to run
     """
 
-    def __init__(
-        self,
-    ):
+    def __init__(self):
         self.python_code = None
         self.jupyter_kernel_init_code: str = 'from agentskills import *'
 
@@ -127,9 +124,7 @@ def check_condition(self, action_str: str) -> bool:
         return self.python_code is not None
 
     def parse(self, action_str: str) -> Action:
-        assert (
-            self.python_code is not None
-        ), 'self.python_code should not be None when parse is called'
+        assert self.python_code is not None
         code_group = self.python_code.group(1).strip()
         thought = action_str.replace(self.python_code.group(0), '').strip()
         return IPythonRunCellAction(
diff --git a/openhands/agenthub/searcher_agent/agent.py b/openhands/agenthub/searcher_agent/agent.py
index f987c0708310..195f3823eed7 100644
--- a/openhands/agenthub/searcher_agent/agent.py
+++ b/openhands/agenthub/searcher_agent/agent.py
@@ -7,25 +7,35 @@
 from openhands.core.config import AgentConfig
 from openhands.core.config.llm_config import LLMConfig
 from openhands.core.message import Message, TextContent
-from openhands.events.action import Action, AgentFinishAction
-from openhands.events.action.commands import CmdRunAction, IPythonRunCellAction
+from openhands.events.action import Action, AgentFinishAction, IPythonRunCellAction
+from openhands.events.action.commands import CmdRunAction
 from openhands.events.action.message import MessageAction
-from openhands.events.observation.commands import (
-    CmdOutputObservation,
-    IPythonRunCellObservation,
-)
+from openhands.events.observation import IPythonRunCellObservation
+from openhands.events.observation.commands import CmdOutputObservation
 from openhands.events.observation.error import ErrorObservation
 from openhands.events.observation.observation import Observation
 from openhands.events.observation.reject import UserRejectObservation
 from openhands.llm.llm import LLM
+from openhands.runtime.plugins.agent_skills import AgentSkillsRequirement
+from openhands.runtime.plugins.jupyter import JupyterRequirement
+from openhands.runtime.plugins.requirement import PluginRequirement
 
 
+# WIP: Make this agent be able to detect when to stop and automatically stop (or make the supervisor able to stop the agent).
 class SearcherAgent(Agent):
     VERSION = '1.0'
     """
     The Searcher Agent is an agent that searches the codebase for relevant information.
     """
 
+    sandbox_plugins: list[PluginRequirement] = [
+        # NOTE: AgentSkillsRequirement need to go before JupyterRequirement, since
+        # AgentSkillsRequirement provides a lot of Python functions,
+        # and it needs to be initialized before Jupyter for Jupyter to use those functions.
+        AgentSkillsRequirement(),
+        JupyterRequirement(),
+    ]
+
     action_parser = SearcherAgentResponseParser()
 
     def __init__(self, llm: LLM, config: AgentConfig):
@@ -47,67 +57,6 @@ def __init__(self, llm: LLM, config: AgentConfig):
         self.logger = logging.getLogger(__name__)
         logging.basicConfig(level=logging.DEBUG)  # Set the logging level
 
-    def step(self, state: State) -> Action:
-        """Performs one step using the Searcher Agent.
-        This includes gathering info about the codebase and summarizing relevant information.
-
-        Parameters:
-        - state (State): used to get updated info
-
-        Returns:
-        - Action: The next action to take
-        """
-        # Check if we should exit
-        latest_user_message = state.history.get_last_user_message()
-        if latest_user_message and latest_user_message.strip() == '/exit':
-            return AgentFinishAction()
-
-        # Prepare messages for LLM
-        messages = []
-
-        # Add system and initial messages
-        task: str = state.inputs.get('task', '')
-        suggested_approach: str = state.inputs.get('suggested_approach', '')
-        messages.extend(
-            [
-                Message(
-                    role='system',
-                    content=[TextContent(text=get_prompt(task, suggested_approach))],
-                )
-            ]
-        )
-
-        # Add history messages
-        for event in state.history.get_events():
-            if isinstance(event, Action):
-                message = self.get_action_message(event)
-            elif isinstance(event, Observation):
-                message = self.get_observation_message(event)
-            else:
-                raise ValueError(f'Unknown event type: {type(event)}')
-
-            if message:
-                # Handle consecutive messages from same role
-                if messages and messages[-1].role == message.role:
-                    messages[-1].content.extend(message.content)
-                else:
-                    messages.append(message)
-
-        # Get response from LLM
-        params = {
-            'messages': self.llm.format_messages_for_llm(messages),
-            'stop': [
-                '</execute_ipython>',
-                '</execute_bash>',
-                '</finish>',
-            ],
-        }
-
-        response = self.llm.completion(**params)
-
-        # Parse and return the next action
-        return self.action_parser.parse(response)
-
     def get_action_message(self, action: Action) -> Message | None:
         """Convert an Action to a Message for the LLM conversation.
 
@@ -162,6 +111,13 @@ def get_observation_message(self, obs: Observation) -> Message | None:
             return Message(role='user', content=[TextContent(text=text)])
         elif isinstance(obs, IPythonRunCellObservation):
             text = obs_prefix + obs.content
+            splitted = text.split('\n')
+            for i, line in enumerate(splitted):
+                if '![image](data:image/png;base64,' in line:
+                    splitted[i] = (
+                        '![image](data:image/png;base64, ...) already displayed to user'
+                    )
+            text = '\n'.join(splitted)
             return Message(role='user', content=[TextContent(text=text)])
         elif isinstance(obs, ErrorObservation):
             text = obs_prefix + obs.content
@@ -173,3 +129,75 @@ def get_observation_message(self, obs: Observation) -> Message | None:
             return Message(role='user', content=[TextContent(text=text)])
         else:
             raise ValueError(f'Unknown observation type: {type(obs)}')
+
+    def step(self, state: State) -> Action:
+        """Performs one step using the SearcherAgent.
+        This includes gathering info on previous steps and prompting the model to make a command to execute.
+
+        Parameters:
+        - state (State): used to get updated info
+
+        Returns:
+        - CmdRunAction(command) - bash command to run
+        - IPythonRunCellAction(code) - IPython code to run
+        - MessageAction(content) - Message action to run (e.g. ask for clarification)
+        - AgentFinishAction() - end the interaction
+        """
+
+        # prepare what we want to send to the LLM
+        messages = self._get_messages(state)
+        params = {
+            'messages': self.llm.format_messages_for_llm(messages),
+            'stop': [
+                '</execute_bash>',
+                '</execute_ipython>',
+            ],
+        }
+
+        response = self.llm.completion(**params)
+
+        return self.action_parser.parse(response)
+
+    def _get_messages(self, state: State) -> list[Message]:
+        # Get task and suggested approach from state inputs
+        task = state.inputs.get('task', '')
+        suggested_approach = state.inputs.get('suggested_approach', '')
+
+        messages: list[Message] = [
+            Message(
+                role='system',
+                content=[
+                    TextContent(
+                        text=get_prompt(task, suggested_approach),
+                        cache_prompt=self.llm.is_caching_prompt_active(),
+                    )
+                ],
+            ),
+        ]
+
+        for event in state.history.get_events():
+            # create message from event
+            if isinstance(event, Action):
+                message = self.get_action_message(event)
+            elif isinstance(event, Observation):
+                message = self.get_observation_message(event)
+            else:
+                raise ValueError(f'Unknown event type: {type(event)}')
+
+            # add regular message
+            if message:
+                # handle error if the message is the SAME role as the previous message
+                if messages and messages[-1].role == message.role:
+                    messages[-1].content.extend(message.content)
+                else:
+                    messages.append(message)
+
+        # Add caching to the last 2 user messages
+        if self.llm.is_caching_prompt_active():
+            user_turns_processed = 0
+            for message in reversed(messages):
+                if message.role == 'user' and user_turns_processed < 2:
+                    message.content[-1].cache_prompt = True
+                    user_turns_processed += 1
+
+        return messages
diff --git a/openhands/agenthub/searcher_agent/prompt.py b/openhands/agenthub/searcher_agent/prompt.py
index bfc9bc612647..6479cda7eab6 100644
--- a/openhands/agenthub/searcher_agent/prompt.py
+++ b/openhands/agenthub/searcher_agent/prompt.py
@@ -4,21 +4,21 @@
 # 2. Implementing the solution.
 # Then the manager needs to check if the issue has been fixed, if not, it needs to iterate.
 general_description = """
-You are a detail-oriented AI, an expert in searching through files and code.
-You are also an expert in summarising code and its purpose.
+The assistant is a detail-oriented AI, an expert in searching through files and code.
+The assistant is also an expert in summarising code and its purpose.
 As a detail-oriented AI, you MUST always read more and more code until you are sure you have found
 all the information you need.
 
-Your goal is to gather information about the codebase to help the programmer fix the issue.
+The assistant's goal is to gather information about the codebase to help the programmer fix the issue.
 Here is the task you are trying to complete:
 %(task)s
 
-IMPORTANT: YOU SHOULD NEVER TRY TO IMPLEMENT A SOLUTION. YOUR ONLY GOAL IS TO GATHER INFORMATION.
+IMPORTANT: THE ASSISTANT SHOULD NEVER TRY TO IMPLEMENT A SOLUTION. THE ASSISTANTR ONLY GOAL IS TO GATHER INFORMATION.
 As an expert in searching through files and code, you have been equipped with a set of tools
 that will help you gather information about the codebase:
-- You can execute bash commands wrapped with <execute_bash>, e.g. <execute_bash> ls </execute_bash>.
+- The assistant can execute bash commands wrapped with <execute_bash>, e.g. <execute_bash> ls </execute_bash>.
 - If a bash command returns exit code `-1`, this means the process is not yet finished.
-- You must then send a second <execute_bash>. The second <execute_bash> can be empty
+- The assistant must then send a second <execute_bash>. The second <execute_bash> can be empty
   (which will retrieve any additional logs), or it can contain text to be sent to STDIN of the running process,
   or it can contain the text `ctrl+c` to interrupt the process.
 - For commands that may run indefinitely, the output should be redirected to a file and the command run
@@ -26,9 +26,85 @@
 - If a command execution result says "Command timed out. Sending SIGINT to the process",
   you should retry running the command in the background.
 
-You should ONLY `run` commands that have no side-effects, like `ls` and `grep`.
+The assistant should ONLY `run` commands that have no side-effects, like `ls` and `grep`.
 
-Your manager gave you a suggested approach that you should follow:
+The assistant can use a Python environment with <execute_ipython>, e.g.:
+<execute_ipython>
+print("Hello World!")
+</execute_ipython>
+
+The assistant can install Python packages using the %%pip magic command in an IPython environment by using the following syntax: <execute_ipython> %%pip install [package needed] </execute_ipython> and should always import packages and define variables before starting to use them.
+
+Apart from the standard Python library, the assistant can also use the following functions (already imported) in <execute_ipython> environment:
+open_file(path: str, line_number: int | None = 1, context_lines: int | None = 100) -> None:
+    Opens the file at the given path in the editor. IF the file is to be edited, first use `scroll_down` repeatedly to read the full file!
+    If line_number is provided, the window will be moved to include that line.
+    It only shows the first 100 lines by default! `context_lines` is the max number of lines to be displayed, up to 100. Use `scroll_up` and `scroll_down` to view more content up or down.
+    Args:
+    path: str: The path to the file to open, preferred absolute path.
+    line_number: int | None = 1: The line number to move to. Defaults to 1.
+    context_lines: int | None = 100: Only shows this number of lines in the context window (usually from line 1), with line_number as the center (if possible). Defaults to 100.
+
+goto_line(line_number: int) -> None:
+    Moves the window to show the specified line number.
+    Args:
+    line_number: int: The line number to move to.
+
+scroll_down() -> None:
+    Moves the window down by 100 lines.
+    Args:
+    None
+
+scroll_up() -> None:
+    Moves the window up by 100 lines.
+    Args:
+    None
+
+search_dir(search_term: str, dir_path: str = './') -> None:
+    Searches for search_term in all files in dir. If dir is not provided, searches in the current directory.
+    Args:
+    search_term: str: The term to search for.
+    dir_path: str: The path to the directory to search.
+
+search_file(search_term: str, file_path: str | None = None) -> None:
+    Searches for search_term in file. If file is not provided, searches in the current open file.
+    Args:
+    search_term: str: The term to search for.
+    file_path: str | None: The path to the file to search.
+
+find_file(file_name: str, dir_path: str = './') -> None:
+    Finds all files with the given name in the specified directory.
+    Args:
+    file_name: str: The name of the file to find.
+    dir_path: str: The path to the directory to search.
+
+parse_pdf(file_path: str) -> None:
+    Parses the content of a PDF file and prints it.
+    Args:
+    file_path: str: The path to the file to open.
+
+parse_docx(file_path: str) -> None:
+    Parses the content of a DOCX file and prints it.
+    Args:
+    file_path: str: The path to the file to open.
+
+parse_latex(file_path: str) -> None:
+    Parses the content of a LaTex file and prints it.
+    Args:
+    file_path: str: The path to the file to open.
+
+parse_pptx(file_path: str) -> None:
+    Parses the content of a pptx file and prints it.
+    Args:
+    file_path: str: The path to the file to open.
+
+
+IMPORTANT:
+- `open_file` only returns the first 100 lines of the file by default! The assistant MUST use `scroll_down` repeatedly to read the full file BEFORE making edits!
+- Indentation is important and code that is not indented correctly will fail and require fixing before it can be run.
+- Any code issued should be less than 50 lines to avoid context being cut off!
+
+The assistant's manager gave you a suggested approach that you should follow:
 %(suggested_approach)s
 
 Follow the suggested approach to gather information about the codebase.
@@ -52,13 +128,14 @@
 
 IMPORTANT: Every entry in the JSON MUST be relevant to the task.
 IMPORTANT: The JSON MUST be contained inside <finish> and </finish> tags.
-IMPORTANT: You MUST have at least one file in the response.
-
+IMPORTANT: The assistant MUST have at least one file in the response.
+IMPORTANT: THE ASSISTANT MUST NOT modify the codebase or NOT ADD any new files.
 """
 
 
 def get_prompt(task: str, suggested_approach: str) -> str:
-    formatted_prompt = (general_description) % {
+    # Escape any % characters in the input strings
+    formatted_prompt = general_description % {
         'task': task,
         'suggested_approach': suggested_approach,
     }
diff --git a/openhands/agenthub/supervisor_agent/agent.py b/openhands/agenthub/supervisor_agent/agent.py
index 034a3ba1b7e1..0c0ba4e83b3d 100644
--- a/openhands/agenthub/supervisor_agent/agent.py
+++ b/openhands/agenthub/supervisor_agent/agent.py
@@ -8,8 +8,8 @@
 from openhands.controller.agent import Agent
 from openhands.controller.state.state import State
 from openhands.core.config import AgentConfig
+from openhands.core.config.llm_config import LLMConfig
 from openhands.core.message import Message, TextContent
-from openhands.core.schema.action import ActionType
 from openhands.core.utils import json
 from openhands.events.action import Action, AgentDelegateAction, AgentFinishAction
 from openhands.events.action.agent import AgentRejectAction
@@ -29,8 +29,9 @@ class SupervisorAgent(Agent):
     suggested_approach_index: int = -1  # -1 Because we increment it before using it
     results: Dict[str, List[Any]] = {'search': [], 'code': []}
     condensed_information: str = ''
-    does_it_needs_a_test: str = ''
+    does_it_needs_a_test: bool = False
     task: str = ''
+    test_command: str = ''
     phase: Literal['search', 'summary', 'code'] = 'search'
 
     def __init__(self, llm: LLM, config: AgentConfig):
@@ -39,6 +40,12 @@ def __init__(self, llm: LLM, config: AgentConfig):
         Parameters:
         - llm (LLM): The llm to be used by this agent
         """
+        llm_config = LLMConfig(
+            model='openai/o1-mini', api_key='REDACTED', temperature=1.0
+        )
+        llm = LLM(llm_config)
+        # TODO: Remove this once we have a real AgentConfig
+        config = AgentConfig(llm_config='o1-mini')
         super().__init__(llm, config)
         # Set up logger
         self.logger = logging.getLogger(__name__)
@@ -49,18 +56,16 @@ def step(self, state: State) -> Action:
         self.logger.debug('Starting step with state: %s', state)
         self.logger.debug('LLM config: %s', self.llm_config)
 
-        if not self.suggested_approaches:
+        if len(self.suggested_approaches) == 0:
             self.suggested_approaches = self.get_suggested_approaches(state)
         self.suggested_approach_index += 1
 
         last_observation = state.history.get_last_observation()
-        if (
-            isinstance(last_observation, AgentDelegateObservation)
-            and last_observation.outputs.get('action', '') == ActionType.FINISH
-        ):
+        # At first the history is empty, so we proceed to the SearchAgent
+        if isinstance(last_observation, AgentDelegateObservation):
             self.results[self.phase].append(last_observation.outputs.get('output', ''))
 
-        if len(self.results[self.phase]) < len(self.suggested_approaches):
+        if self.suggested_approach_index < len(self.suggested_approaches):
             # Delegate to the SearcherAgent as we need to gather more information
             return self.delegate_to_agent(
                 'SearcherAgent',
@@ -71,33 +76,57 @@ def step(self, state: State) -> Action:
             )
 
         if self.phase == 'search':
-            # We don't change the phase until we have the condensed information
             condensed_information = self.ask_llm(
-                self.task, '2', json.dumps(self.results['search'])
-            )[0]
-            if condensed_information.get('summary', '') != '':
-                self.phase = 'summary'
-                self.condensed_information = condensed_information.get('summary', '')
-            else:
-                suggested_approach: str | list[str] = condensed_information.get(
-                    'suggested_approach', []
-                )
-                self.results['search'].append(suggested_approach)
-                return self.delegate_to_agent(
-                    'SearcherAgent', self.task, suggested_approach
-                )
+                self.task, 'summary', self.results[self.phase]
+            )
+            if condensed_information and len(condensed_information) > 0:
+                first_result = condensed_information[0]
+                if first_result.get('summary', '') != '':
+                    self.phase = 'summary'
+                    self.condensed_information = first_result.get('summary', '')
+                else:
+                    suggested_approach: str | list[str] = first_result.get(
+                        'suggested_approach', []
+                    )
+                    self.results['search'].append(suggested_approach)
+                    return self.delegate_to_agent(
+                        'SearcherAgent', self.task, suggested_approach
+                    )
 
         if self.phase == 'summary':
-            # Now we have to judge if this issue requires a test or not before fixing it
-            does_it_needs_a_test = self.ask_llm(
-                self.task, 'code', self.condensed_information
-            )[0]
-            if does_it_needs_a_test.get('suggested_approach', '') == TASK_TYPE_ISSUE:
-                self.phase = 'code'
-            else:
+            if not self.does_it_needs_a_test:
+                test_check = self.ask_llm(self.task, 'code', self.condensed_information)
+                first_check = (
+                    test_check[0] if test_check and len(test_check) > 0 else {}
+                )
+                self.does_it_needs_a_test = (
+                    first_check.get('suggested_approach', '') == TASK_TYPE_ISSUE
+                )
                 self.phase = 'code'
-
-        # WIP: Implement the code phase
+                if self.does_it_needs_a_test:
+                    self.current_delegate = 'TesterAgent'
+                    return AgentDelegateAction(
+                        agent='TesterAgent',
+                        inputs={
+                            'task': self.task,
+                            'summary': self.condensed_information,
+                        },
+                    )
+        if self.phase == 'code':
+            if (
+                self.does_it_needs_a_test
+                and last_observation is not None
+                and isinstance(last_observation, AgentDelegateObservation)
+            ):
+                self.test_command = last_observation.outputs.get('output', '')
+                return AgentDelegateAction(
+                    agent='CoderAgent',
+                    inputs={
+                        'task': self.task,
+                        'summary': self.condensed_information,
+                        'test_command': self.test_command,
+                    },
+                )
 
         return AgentFinishAction()
 
@@ -114,6 +143,7 @@ def delegate_to_agent(
         self, agent_name: str, task: str, suggested_approach: Union[str, List[str]]
     ) -> AgentDelegateAction:
         self.logger.debug(f'Delegating to agent: {agent_name}')
+        self.current_delegate = agent_name
         # Join the list of strings with newlines if it's a list
         approach = (
             '\n'.join(suggested_approach)
@@ -125,8 +155,11 @@ def delegate_to_agent(
         )
 
     def ask_llm(
-        self, task: str, phase: str, search_results: str = ''
+        self, task: str, phase: str, search_results: Union[str, List[str]] = ''
     ) -> List[Dict[str, str]]:
+        # Format search_results as one item per line if it's a list
+        if isinstance(search_results, list):
+            search_results = '\n'.join(search_results)
         prompt = get_prompt(task, phase, search_results)
         return self.get_response(prompt)
 
@@ -136,4 +169,6 @@ def get_response(self, prompt: str) -> List[Dict[str, str]]:
         response = self.llm.completion(
             messages=self.llm.format_messages_for_llm(message)
         )
+        if isinstance(response, list):
+            return json.loads(response[0]['message']['content'])
         return json.loads(response['choices'][0]['message']['content'])
diff --git a/openhands/agenthub/tester_agent/__init__.py b/openhands/agenthub/tester_agent/__init__.py
new file mode 100644
index 000000000000..54be665abd42
--- /dev/null
+++ b/openhands/agenthub/tester_agent/__init__.py
@@ -0,0 +1,4 @@
+from openhands.agenthub.tester_agent.agent import TesterAgent
+from openhands.controller.agent import Agent
+
+Agent.register('TesterAgent', TesterAgent)
diff --git a/openhands/agenthub/tester_agent/action_parser.py b/openhands/agenthub/tester_agent/action_parser.py
new file mode 100644
index 000000000000..8abc7c353916
--- /dev/null
+++ b/openhands/agenthub/tester_agent/action_parser.py
@@ -0,0 +1,158 @@
+import re
+
+from openhands.controller.action_parser import (
+    ActionParser,
+    ResponseParser,
+)
+from openhands.events.action import (
+    Action,
+    AgentFinishAction,
+    CmdRunAction,
+    IPythonRunCellAction,
+    MessageAction,
+)
+
+
+class TesterAgentResponseParser(ResponseParser):
+    """Parser action:
+    - CmdRunAction(command) - bash command to run
+    - IPythonRunCellAction(code) - IPython code to run
+    - MessageAction(content) - Message action to run (e.g. ask for clarification)
+    - AgentFinishAction() - end the interaction
+    """
+
+    def __init__(self):
+        # Need pay attention to the item order in self.action_parsers
+        super().__init__()
+        self.action_parsers = [
+            TesterAgentActionParserFinish(),
+            TesterAgentActionParserCmdRun(),
+            TesterAgentActionParserIPythonRunCell(),
+        ]
+        self.default_parser = TesterAgentActionParserMessage()
+
+    def parse(self, response) -> Action:
+        action_str = self.parse_response(response)
+        return self.parse_action(action_str)
+
+    def parse_response(self, response) -> str:
+        action = response.choices[0].message.content
+        if action is None:
+            return ''
+        for lang in ['bash', 'ipython', 'browse']:
+            # special handling for DeepSeek: it has stop-word bug and returns </execute_ipython instead of </execute_ipython>
+            if f'</execute_{lang}' in action and f'</execute_{lang}>' not in action:
+                action = action.replace(f'</execute_{lang}', f'</execute_{lang}>')
+
+            if f'<execute_{lang}>' in action and f'</execute_{lang}>' not in action:
+                action += f'</execute_{lang}>'
+        if '<file_edit' in action and '</file_edit>' not in action:
+            action += '</file_edit>'
+        return action
+
+    def parse_action(self, action_str: str) -> Action:
+        for action_parser in self.action_parsers:
+            if action_parser.check_condition(action_str):
+                return action_parser.parse(action_str)
+        return self.default_parser.parse(action_str)
+
+
+class TesterAgentActionParserFinish(ActionParser):
+    """Parser action:
+    - AgentFinishAction() - end the interaction
+    """
+
+    def __init__(
+        self,
+    ):
+        self.finish_command = None
+
+    def check_condition(self, action_str: str) -> bool:
+        self.finish_command = re.search(r'<finish>.*</finish>', action_str, re.DOTALL)
+        return self.finish_command is not None
+
+    def parse(self, action_str: str) -> Action:
+        assert (
+            self.finish_command is not None
+        ), 'self.finish_command should not be None when parse is called'
+        output = self.finish_command.group(1).strip()
+        outputs = {'output': output}
+        return AgentFinishAction(outputs=outputs)
+
+
+class TesterAgentActionParserCmdRun(ActionParser):
+    """Parser action:
+    - CmdRunAction(command) - bash command to run
+    - AgentFinishAction() - end the interaction
+    """
+
+    def __init__(
+        self,
+    ):
+        self.bash_command = None
+
+    def check_condition(self, action_str: str) -> bool:
+        self.bash_command = re.search(
+            r'<execute_bash>(.*?)</execute_bash>', action_str, re.DOTALL
+        )
+        return self.bash_command is not None
+
+    def parse(self, action_str: str) -> Action:
+        assert (
+            self.bash_command is not None
+        ), 'self.bash_command should not be None when parse is called'
+        thought = action_str.replace(self.bash_command.group(0), '').strip()
+        # a command was found
+        command_group = self.bash_command.group(1).strip()
+        if command_group.strip() == 'exit':
+            return AgentFinishAction(thought=thought)
+        return CmdRunAction(command=command_group, thought=thought)
+
+
+class TesterAgentActionParserIPythonRunCell(ActionParser):
+    """Parser action:
+    - IPythonRunCellAction(code) - IPython code to run
+    """
+
+    def __init__(
+        self,
+    ):
+        self.python_code = None
+        self.jupyter_kernel_init_code: str = 'from agentskills import *'
+
+    def check_condition(self, action_str: str) -> bool:
+        self.python_code = re.search(
+            r'<execute_ipython>(.*?)</execute_ipython>', action_str, re.DOTALL
+        )
+        return self.python_code is not None
+
+    def parse(self, action_str: str) -> Action:
+        assert (
+            self.python_code is not None
+        ), 'self.python_code should not be None when parse is called'
+        code_group = self.python_code.group(1).strip()
+        thought = action_str.replace(self.python_code.group(0), '').strip()
+        return IPythonRunCellAction(
+            code=code_group,
+            thought=thought,
+            kernel_init_code=self.jupyter_kernel_init_code,
+        )
+
+
+class TesterAgentActionParserMessage(ActionParser):
+    """Parser action:
+    - MessageAction(content) - Message action to run (e.g. ask for clarification)
+    """
+
+    def __init__(
+        self,
+    ):
+        pass
+
+    def check_condition(self, action_str: str) -> bool:
+        # We assume the LLM is GOOD enough that when it returns pure natural language
+        # it wants to talk to the user
+        return True
+
+    def parse(self, action_str: str) -> Action:
+        return MessageAction(content=action_str, wait_for_response=True)
diff --git a/openhands/agenthub/tester_agent/agent.py b/openhands/agenthub/tester_agent/agent.py
new file mode 100644
index 000000000000..3dc449f43430
--- /dev/null
+++ b/openhands/agenthub/tester_agent/agent.py
@@ -0,0 +1,201 @@
+import logging
+
+from openhands.agenthub.tester_agent.action_parser import TesterAgentResponseParser
+from openhands.agenthub.tester_agent.prompt import get_prompt
+from openhands.controller.agent import Agent
+from openhands.controller.state.state import State
+from openhands.core.config import AgentConfig
+from openhands.core.config.llm_config import LLMConfig
+from openhands.core.message import Message, TextContent
+from openhands.events.action import Action, AgentFinishAction
+from openhands.events.action.commands import CmdRunAction, IPythonRunCellAction
+from openhands.events.action.message import MessageAction
+from openhands.events.observation.commands import (
+    CmdOutputObservation,
+    IPythonRunCellObservation,
+)
+from openhands.events.observation.error import ErrorObservation
+from openhands.events.observation.observation import Observation
+from openhands.events.observation.reject import UserRejectObservation
+from openhands.llm.llm import LLM
+
+
+class TesterAgent(Agent):
+    VERSION = '1.0'
+    """
+    The Tester Agent is an agent that tries to replicate the issue.
+    """
+
+    action_parser = TesterAgentResponseParser()
+
+    def __init__(self, llm: LLM, config: AgentConfig):
+        """Initialize the Tester Agent with an LLM
+
+        Parameters:
+        - llm (LLM): The llm to be used by this agent
+        - config (AgentConfig): The configuration for this agent
+        """
+        # TODO: Remove this once we have a real LLM config
+        llm_config = LLMConfig(
+            model='deepseek/deepseek-chat', api_key='REDACTED', temperature=0.0
+        )
+        llm = LLM(llm_config)
+        # TODO: Remove this once we have a real AgentConfig
+        config = AgentConfig(llm_config='deepseek')
+        super().__init__(llm, config)
+        # Set up logger
+        self.logger = logging.getLogger(__name__)
+        logging.basicConfig(level=logging.DEBUG)  # Set the logging level
+
+    def get_action_message(self, action: Action) -> Message | None:
+        """Convert an Action to a Message for the LLM conversation.
+
+        Parameters:
+        - action (Action): The action to convert
+
+        Returns:
+        - Message | None: The converted message, or None if action type is not supported
+        """
+        if isinstance(action, CmdRunAction):
+            return Message(
+                role='assistant',
+                content=[
+                    TextContent(
+                        text=f'{action.thought}\n<execute_bash>\n{action.command}\n</execute_bash>'
+                    )
+                ],
+            )
+        elif isinstance(action, IPythonRunCellAction):
+            return Message(
+                role='assistant',
+                content=[
+                    TextContent(
+                        text=f'{action.thought}\n<execute_ipython>\n{action.code}\n</execute_ipython>'
+                    )
+                ],
+            )
+        elif isinstance(action, MessageAction):
+            return Message(
+                role='user' if action.source == 'user' else 'assistant',
+                content=[TextContent(text=action.content)],
+            )
+        elif isinstance(action, AgentFinishAction) and action.source == 'agent':
+            return Message(role='assistant', content=[TextContent(text=action.thought)])
+        return None
+
+    def get_observation_message(self, obs: Observation) -> Message | None:
+        """Convert an Observation to a Message for the LLM conversation.
+
+        Parameters:
+        - obs (Observation): The observation to convert
+
+        Returns:
+        - Message | None: The converted message, or None if observation type is not supported
+        """
+        obs_prefix = 'OBSERVATION:\n'
+        if isinstance(obs, CmdOutputObservation):
+            text = obs_prefix + obs.content
+            text += (
+                f'\n[Command {obs.command_id} finished with exit code {obs.exit_code}]'
+            )
+            return Message(role='user', content=[TextContent(text=text)])
+        elif isinstance(obs, IPythonRunCellObservation):
+            text = obs_prefix + obs.content
+            return Message(role='user', content=[TextContent(text=text)])
+        elif isinstance(obs, ErrorObservation):
+            text = obs_prefix + obs.content
+            text += '\n[Error occurred in processing last action]'
+            return Message(role='user', content=[TextContent(text=text)])
+        elif isinstance(obs, UserRejectObservation):
+            text = obs_prefix + obs.content
+            text += '\n[Last action has been rejected by the user]'
+            return Message(role='user', content=[TextContent(text=text)])
+        else:
+            raise ValueError(f'Unknown observation type: {type(obs)}')
+
+    def step(self, state: State) -> Action:
+        """Performs one step using the Tester Agent.
+        This includes gathering info on previous steps and prompting the model to make a command to execute.
+
+        Parameters:
+        - state (State): used to get updated info
+
+        Returns:
+        - CmdRunAction(command) - bash command to run
+        - IPythonRunCellAction(code) - IPython code to run
+        - MessageAction(content) - Message action to run (e.g. ask for clarification)
+        - AgentFinishAction() - end the interaction
+        """
+        # if we're done, go back
+        latest_user_message = state.history.get_last_user_message()
+        if latest_user_message and latest_user_message.strip() == '/exit':
+            return AgentFinishAction()
+
+        # prepare what we want to send to the LLM
+        messages = self._get_messages(state)
+        params = {
+            'messages': self.llm.format_messages_for_llm(messages),
+            'stop': [
+                '</execute_ipython>',
+                '</execute_bash>',
+            ],
+        }
+
+        response = self.llm.completion(**params)
+
+        return self.action_parser.parse(response)
+
+    def _get_messages(self, state: State) -> list[Message]:
+        task = state.inputs.get('task', '')
+        summary = state.inputs.get('summary', '')
+
+        messages: list[Message] = [
+            Message(
+                role='system',
+                content=[
+                    TextContent(
+                        text=get_prompt(task, summary),
+                        cache_prompt=self.llm.is_caching_prompt_active(),  # Cache system prompt
+                    )
+                ],
+            ),
+        ]
+
+        for event in state.history.get_events():
+            # create a regular message from an event
+            if isinstance(event, Action):
+                message = self.get_action_message(event)
+            elif isinstance(event, Observation):
+                message = self.get_observation_message(event)
+            else:
+                raise ValueError(f'Unknown event type: {type(event)}')
+
+            # add regular message
+            if message:
+                # handle error if the message is the SAME role as the previous message
+                if messages and messages[-1].role == message.role:
+                    messages[-1].content.extend(message.content)
+                else:
+                    messages.append(message)
+
+        # Add caching to the last 2 user messages
+        if self.llm.is_caching_prompt_active():
+            user_turns_processed = 0
+            for message in reversed(messages):
+                if message.role == 'user' and user_turns_processed < 2:
+                    message.content[
+                        -1
+                    ].cache_prompt = True  # Last item inside the message content
+                    user_turns_processed += 1
+
+        # Add environment reminder to the latest user message
+        latest_user_message = next(
+            (m for m in reversed(messages) if m.role == 'user'),
+            None,
+        )
+
+        if latest_user_message:
+            reminder_text = f'\n\nENVIRONMENT REMINDER: You have {state.max_iterations - state.iteration} turns left to complete the task. When finished reply with <finish></finish>.'
+            latest_user_message.content.append(TextContent(text=reminder_text))
+
+        return messages
diff --git a/openhands/agenthub/tester_agent/prompt.py b/openhands/agenthub/tester_agent/prompt.py
new file mode 100644
index 000000000000..7523bc7d16f3
--- /dev/null
+++ b/openhands/agenthub/tester_agent/prompt.py
@@ -0,0 +1,329 @@
+# General Description, the goal is to devise a manager that is able to iterate if the solution has not been found yet.
+# In order to successfully fix an issue there are two phases:
+# 1. Exploring the codebase, finding the root cause of the issue.
+# 2. Implementing the solution.
+# Then the manager needs to check if the issue has been fixed, if not, it needs to iterate.
+general_description = """
+You are a QA Engineer, an expert in testing software.
+You are given an issue and your goal is to understand how to replicate the issue.
+
+Here is the issue you are trying to replicate:
+%(task)s
+
+Some other agents have already gathered information about the codebase.
+You can use this information to understand the codebase and replicate the issue.
+%(summary)s
+
+IMPORTANT: YOU SHOULD NEVER TRY TO IMPLEMENT A SOLUTION. YOUR ONLY GOAL IS TO REPLICATE THE ISSUE.
+As an expert in testing software, you have been equipped with a set of tools
+that will help you replicate the issue:
+- You can execute bash commands wrapped with <execute_bash>, e.g. <execute_bash> ls </execute_bash>.
+- If a bash command returns exit code `-1`, this means the process is not yet finished.
+- You must then send a second <execute_bash>. The second <execute_bash> can be empty
+  (which will retrieve any additional logs), or it can contain text to be sent to STDIN of the running process,
+  or it can contain the text `ctrl+c` to interrupt the process.
+- For commands that may run indefinitely, the output should be redirected to a file and the command run
+  in the background, e.g. <execute_bash> python3 app.py > server.log 2>&1 & </execute_bash>
+- If a command execution result says "Command timed out. Sending SIGINT to the process",
+  you should retry running the command in the background.
+
+You have access to a python interpreter wrapped with <execute_ipython>.
+e.g.:
+<execute_ipython>
+print("Hello World!")
+</execute_ipython>
+
+You can install Python packages using the %pip magic command in an IPython environment by using the following syntax: <execute_ipython> %pip install [package needed] </execute_ipython> and should always import packages and define variables before starting to use them.
+
+Apart from the standard Python library, you can also use the following functions (already imported) in <execute_ipython> environment:
+open_file(path: str, line_number: int | None = 1, context_lines: int | None = 100) -> None:
+    Opens the file at the given path in the editor. IF the file is to be edited, first use `scroll_down` repeatedly to read the full file!
+    If line_number is provided, the window will be moved to include that line.
+    It only shows the first 100 lines by default! `context_lines` is the max number of lines to be displayed, up to 100. Use `scroll_up` and `scroll_down` to view more content up or down.
+    Args:
+    path: str: The path to the file to open, preferred absolute path.
+    line_number: int | None = 1: The line number to move to. Defaults to 1.
+    context_lines: int | None = 100: Only shows this number of lines in the context window (usually from line 1), with line_number as the center (if possible). Defaults to 100.
+
+goto_line(line_number: int) -> None:
+    Moves the window to show the specified line number.
+    Args:
+    line_number: int: The line number to move to.
+
+scroll_down() -> None:
+    Moves the window down by 100 lines.
+    Args:
+    None
+
+scroll_up() -> None:
+    Moves the window up by 100 lines.
+    Args:
+    None
+
+search_dir(search_term: str, dir_path: str = './') -> None:
+    Searches for search_term in all files in dir. If dir is not provided, searches in the current directory.
+    Args:
+    search_term: str: The term to search for.
+    dir_path: str: The path to the directory to search.
+
+search_file(search_term: str, file_path: str | None = None) -> None:
+    Searches for search_term in file. If file is not provided, searches in the current open file.
+    Args:
+    search_term: str: The term to search for.
+    file_path: str | None: The path to the file to search.
+
+find_file(file_name: str, dir_path: str = './') -> None:
+    Finds all files with the given name in the specified directory.
+    Args:
+    file_name: str: The name of the file to find.
+    dir_path: str: The path to the directory to search.
+
+parse_pdf(file_path: str) -> None:
+    Parses the content of a PDF file and prints it.
+    Args:
+    file_path: str: The path to the file to open.
+
+parse_docx(file_path: str) -> None:
+    Parses the content of a DOCX file and prints it.
+    Args:
+    file_path: str: The path to the file to open.
+
+parse_latex(file_path: str) -> None:
+    Parses the content of a LaTex file and prints it.
+    Args:
+    file_path: str: The path to the file to open.
+
+parse_pptx(file_path: str) -> None:
+    Parses the content of a pptx file and prints it.
+    Args:
+    file_path: str: The path to the file to open.
+
+
+IMPORTANT:
+- `open_file` only returns the first 100 lines of the file by default!
+- Indentation is important and code that is not indented correctly will fail and require fixing before it can be run.
+- Any code issued should be less than 50 lines to avoid context being cut off!
+
+
+Create a test that when executed, it will replicate the issue.
+Responses should be concise.
+You should attempt fewer things at a time instead of putting too many commands OR too much code in one "execute" block.
+Include ONLY ONE <execute_ipython>, <execute_bash>, or <execute_browse> per response, unless you is finished with the task or needs more input or action from the user in order to proceed.
+If you is finished with the task you MUST include <finish></finish> in your response.
+IMPORTANT: Execute code using <execute_ipython>, <execute_bash>, or <execute_browse> whenever possible.
+IMPORTANT: You MUST NOT edit or remove any files, but you are allowed to add new files. Code executed in <execute_ipython> WON'T BE SAVED.
+You should utilize full file paths and the `pwd` command to prevent path-related errors.
+You MUST NOT apologize to the user or thank the user after running commands or editing files. It should only address the user in response to an explicit message from the user, or to ask for more information.
+You MUST NOT push any changes to GitHub unless explicitly requested to do so.
+
+When you have successfully managed to replicate the issue in a test,
+you MUST generate a NEW test file that when executed will replicate the issue.
+Then, you MUST run the test and verify that the issue is reproducible.
+If it is not reproducible, remove the new test file and think of a different approach to replicate the issue.
+If it is reproducible, you MUST output the command to run the test file wrapped with <finish></finish> tags.
+
+--- START OF EXAMPLE ---
+
+USER: I need to test if there's an issue with the ModelChain when processing multiple inputs in parallel.
+
+YOU: I'll create a test to verify the ModelChain's behavior with parallel processing.
+
+<execute_ipython>
+import torch
+from openhands.models import ModelChain
+import concurrent.futures
+
+def test_model_chain_parallel():
+    # Create a simple model chain
+    model = ModelChain(
+        backbone="resnet18",
+        num_classes=10
+    )
+
+    # Generate dummy input data
+    batch_size = 4
+    input_data = torch.randn(batch_size, 3, 224, 224)
+
+    # Process inputs in parallel
+    with concurrent.futures.ThreadPoolExecutor(max_workers=2) as executor:
+        futures = []
+        for i in range(batch_size):
+            futures.append(
+                executor.submit(model.forward, input_data[i:i+1])
+            )
+
+        # Get results
+        results = [future.result() for future in futures]
+
+    return results
+
+# Run the test
+results = test_model_chain_parallel()
+print(f"Processed {len(results)} inputs in parallel")
+</execute_ipython>
+
+USER:
+OBSERVATION:
+RuntimeError: Cannot re-initialize CUDA in forked subprocess. To use CUDA with multiprocessing, you must use 'spawn' start method
+
+YOU: Ah, I see the issue is reproducible. Let me create a test file that demonstrates this CUDA initialization problem.
+
+<execute_ipython>
+test_code = '''import torch
+import torch.multiprocessing as mp
+import pytest
+from openhands.models import ModelChain
+
+def process_input(model, input_tensor):
+    return model(input_tensor)
+
+def test_model_chain_multiprocess():
+    # Force CUDA if available
+    device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')
+
+    # Create model and move to device
+    model = ModelChain(
+        backbone="resnet18",
+        num_classes=10
+    ).to(device)
+
+    # Create dummy input
+    input_tensor = torch.randn(1, 3, 224, 224).to(device)
+
+    # Try parallel processing
+    mp.set_start_method('fork')  # This will cause the CUDA error
+    processes = []
+
+    for _ in range(2):
+        p = mp.Process(target=process_input, args=(model, input_tensor))
+        p.start()
+        processes.append(p)
+
+    for p in processes:
+        p.join()'''
+
+with open('test_model_chain_cuda.py', 'w') as f:
+    f.write(test_code)
+</execute_ipython>
+
+YOU: I've created a test file that reproduces the CUDA initialization issue. Let me run it to confirm it works.
+
+<execute_bash>
+python3 -m pytest test_model_chain_cuda.py -v
+</execute_bash>
+
+USER:
+OBSERVATION:
+RuntimeError: Cannot re-initialize CUDA in forked subprocess. To use CUDA with multiprocessing, you must use 'spawn' start method
+
+YOU: Perfect, I've successfully replicated the issue and created a test file that demonstrates the problem.
+
+<finish>
+python3 -m pytest test_model_chain_cuda.py -v
+</finish>
+
+--- END OF EXAMPLE ---
+
+--- START OF EXAMPLE 2 ---
+
+USER: I need to test if the CLI's quiet mode is working properly. The issue is that there's no way to reduce verbosity below the default level.
+
+YOU: I'll create a test to verify the CLI's verbosity behavior.
+
+<execute_ipython>
+import subprocess
+import tempfile
+
+def test_cli_verbosity():
+    # Create a temporary file with some content to test
+    with tempfile.NamedTemporaryFile(mode='w', suffix='.sql') as temp_file:
+        temp_file.write("SELECT  *  FROM table")  # Extra spaces to trigger formatting
+        temp_file.flush()
+
+        # Run command with default verbosity
+        default_result = subprocess.run(
+            ['sqlfluff', 'fix', temp_file.name],
+            capture_output=True,
+            text=True
+        )
+
+        # Run command with --quiet flag (which doesn't exist yet, this should fail)
+        quiet_result = subprocess.run(
+            ['sqlfluff', 'fix', '--quiet', temp_file.name],
+            capture_output=True,
+            text=True
+        )
+
+        print("Default verbosity output length:", len(default_result.stdout.split('\n')))
+        print("Default output:", default_result.stdout)
+        print("\nQuiet mode output length:", len(quiet_result.stdout.split('\n')))
+        print("Quiet output:", quiet_result.stdout)
+
+test_cli_verbosity()
+</execute_ipython>
+
+USER:
+OBSERVATION:
+FileNotFoundError: [Errno 2] No such file or directory: 'sqlfluff'
+
+YOU: Let me create a test file that demonstrates this verbosity issue with our CLI tool.
+
+<execute_ipython>
+test_code = '''import subprocess
+import tempfile
+import pytest
+
+def test_cli_verbosity_levels():
+    # Create a temporary file with content that needs formatting
+    with tempfile.NamedTemporaryFile(mode='w', suffix='.sql') as temp_file:
+        temp_file.write("SELECT  *  FROM table")  # Extra spaces to trigger formatting
+        temp_file.flush()
+
+        # Test default verbosity
+        default_cmd = ['sqlfluff', 'fix', temp_file.name]
+        default_output = subprocess.run(default_cmd, capture_output=True, text=True)
+
+        # Test with quiet flag (should fail as feature doesn't exist)
+        quiet_cmd = ['sqlfluff', 'fix', '--quiet', temp_file.name]
+        quiet_output = subprocess.run(quiet_cmd, capture_output=True, text=True)
+
+        # Verify that default output is more verbose
+        assert len(default_output.stdout.split('\\n')) > 1, "Default output should be verbose"
+
+        # This assertion should fail since --quiet flag doesn't exist
+        with pytest.raises(subprocess.CalledProcessError):
+            subprocess.run(quiet_cmd, check=True, capture_output=True, text=True)'''
+
+with open('test_cli_verbosity.py', 'w') as f:
+    f.write(test_code)
+</execute_ipython>
+
+YOU: Let me run the test to verify it demonstrates the missing quiet mode functionality.
+
+<execute_bash>
+python3 -m pytest test_cli_verbosity.py -v
+</execute_bash>
+
+USER:
+OBSERVATION:
+FileNotFoundError: [Errno 2] No such file or directory: 'sqlfluff'
+
+YOU: Perfect, I've created a test file that demonstrates the lack of quiet mode functionality. The test fails because sqlfluff is not installed, but the test case correctly shows that the CLI tool lacks the ability to reduce verbosity below the default level.
+
+<finish>
+python3 -m pytest test_cli_verbosity.py -v
+</finish>
+
+--- END OF EXAMPLE 2 ---
+
+IMPORTANT: YOU MUST ONLY GENERATE A <finish></finish> TAG IN YOUR RESPONSE with only ONE command to run the test file.
+
+"""
+
+
+def get_prompt(task: str, summary: str) -> str:
+    formatted_prompt = (general_description) % {
+        'task': task,
+        'summary': summary,
+    }
+    return formatted_prompt

From 500112aab722b454b17358a759c8624deb4c36f7 Mon Sep 17 00:00:00 2001
From: AlexCuadron <alex.cl.2000@gmail.com>
Date: Tue, 5 Nov 2024 14:17:32 -0800
Subject: [PATCH 14/18] merge

---
 .github/ISSUE_TEMPLATE/bug_template.yml       |   2 +
 .github/workflows/ghcr-build.yml              |   6 +-
 .github/workflows/openhands-resolver.yml      |   2 +
 .gitignore                                    |   1 +
 README.md                                     |   9 +-
 .../usage/how-to/evaluation-harness.md        |   5 +-
 .../usage/how-to/evaluation-harness.md        |   4 +-
 .../current/usage/how-to/headless-mode.md     |   1 -
 .../usage/how-to/evaluation-harness.md        |   4 +-
 docs/modules/usage/how-to/gui-mode.md         |   9 +
 docs/modules/usage/runtimes.md                |   2 -
 evaluation/EDA/run_infer.py                   |   7 +-
 evaluation/README.md                          |   1 -
 evaluation/agent_bench/run_infer.py           |   5 +-
 evaluation/aider_bench/run_infer.py           |   3 +-
 evaluation/biocoder/run_infer.py              |   3 +-
 evaluation/bird/run_infer.py                  |   5 +-
 evaluation/browsing_delegation/run_infer.py   |   3 +-
 evaluation/discoverybench/README.md           |  37 ++
 .../discoverybench/eval_utils/README.md       |   7 +
 .../discoverybench/eval_utils/__init__.py     |   0
 .../eval_utils/eval_w_subhypo_gen.py          | 538 ++++++++++++++++++
 .../discoverybench/eval_utils/lm_utils.py     |  64 +++
 .../eval_utils/openai_helpers.py              | 190 +++++++
 .../eval_utils/openai_semantic_gen_prompts.py | 151 +++++
 .../eval_utils/response_parser.py             |  52 ++
 evaluation/discoverybench/run_infer.py        | 492 ++++++++++++++++
 .../discoverybench/scripts/run_infer.sh       |  46 ++
 evaluation/gaia/run_infer.py                  |   5 +-
 evaluation/gorilla/run_infer.py               |   5 +-
 evaluation/gpqa/run_infer.py                  |   5 +-
 evaluation/humanevalfix/run_infer.py          |   3 +-
 evaluation/integration_tests/run_infer.py     |  17 +-
 .../tests/t06_github_pr_browsing.py           |  44 ++
 evaluation/logic_reasoning/run_infer.py       |   5 +-
 evaluation/miniwob/run_infer.py               |  53 +-
 evaluation/mint/run_infer.py                  |   9 +-
 evaluation/ml_bench/run_infer.py              |   3 +-
 evaluation/scienceagentbench/Dockerfile       |  17 +
 .../scienceagentbench/Dockerfile.evaluator    |  25 +
 evaluation/scienceagentbench/README.md        |  54 ++
 evaluation/scienceagentbench/post_proc.py     |  30 +
 evaluation/scienceagentbench/run_infer.py     | 292 ++++++++++
 .../scienceagentbench/scripts/run_infer.sh    |  49 ++
 evaluation/swe_bench/run_infer.py             |  26 +-
 evaluation/swe_bench/scripts/run_infer.sh     |  11 +
 evaluation/toolqa/run_infer.py                |   5 +-
 evaluation/utils/shared.py                    |  46 +-
 evaluation/webarena/run_infer.py              |   5 +-
 frontend/.eslintrc                            |   2 +-
 frontend/.gitignore                           |   7 +-
 .../components/chat/chat-interface.test.tsx   |   4 +-
 .../file-explorer/FileExplorer.test.tsx       |  13 +-
 .../utils/extractModelAndProvider.test.ts     |   1 -
 .../utils/organizeModelsAndProviders.test.ts  |   1 -
 frontend/package-lock.json                    | 105 +++-
 frontend/package.json                         |   5 +-
 frontend/playwright.config.ts                 |  79 +++
 frontend/src/api/open-hands.ts                | 135 +----
 frontend/src/api/open-hands.types.ts          |   5 +
 frontend/src/assets/arrow-send.svg            |   2 +-
 .../assets/branding/all-hands-logo-spark.svg  |   2 +-
 frontend/src/assets/branding/github-logo.svg  |   2 +-
 frontend/src/assets/clip.svg                  |   2 +-
 frontend/src/assets/clipboard.svg             |   2 +-
 frontend/src/assets/default-user.svg          |   2 +-
 frontend/src/assets/docs.svg                  |   2 +-
 frontend/src/assets/external-link.svg         |   2 +-
 frontend/src/assets/lightbulb.svg             |   2 +-
 frontend/src/assets/loading-outer.svg         |   2 +-
 frontend/src/assets/message.svg               |   2 +-
 frontend/src/assets/new-project.svg           |   2 +-
 frontend/src/assets/refresh.svg               |   2 +-
 frontend/src/components/AgentStatusBar.tsx    |  27 +-
 .../analytics-consent-form-modal.tsx          |  42 ++
 .../src/components/buttons/ModalButton.tsx    |   3 +
 frontend/src/components/chat-interface.tsx    |   2 +-
 frontend/src/components/chat/message.d.ts     |   3 +-
 frontend/src/components/error-message.tsx     |  35 +-
 frontend/src/components/feedback-form.tsx     |  10 +-
 .../components/file-explorer/FileExplorer.tsx |  82 ++-
 .../src/components/file-explorer/TreeNode.tsx |  19 +-
 .../github-repositories-suggestion-box.tsx    |  94 +++
 .../modals/AccountSettingsModal.tsx           |  13 +
 .../modals/confirmation-modals/BaseModal.tsx  |   8 +-
 .../modals/connect-to-github-modal.tsx        |   1 +
 frontend/src/context/socket.tsx               |  16 +-
 frontend/src/entry.client.tsx                 |  15 +-
 frontend/src/i18n/translation.json            |  18 +
 frontend/src/mocks/handlers.ts                |  33 +-
 .../_oh._index/github-repo-selector.tsx       |  10 +-
 frontend/src/routes/_oh._index/route.tsx      | 104 ++--
 .../_oh.app._index/code-editor-component.tsx  |  33 +-
 frontend/src/routes/_oh.app._index/route.tsx  |  57 +-
 frontend/src/routes/_oh.app.tsx               |  33 +-
 frontend/src/routes/_oh.tsx                   |  28 +-
 frontend/src/routes/oauth.github.callback.tsx |   4 +-
 frontend/src/routes/set-consent.ts            |   9 +
 frontend/src/routes/settings.ts               |   3 +
 frontend/src/services/actions.ts              |  32 +-
 frontend/src/services/api.ts                  |  34 +-
 frontend/src/services/auth.ts                 |  21 +-
 frontend/src/state/chatSlice.ts               |   6 +-
 frontend/src/state/statusSlice.ts             |   6 +-
 frontend/src/types/Message.tsx                |  10 +-
 frontend/src/types/core/observations.ts       |   3 +
 frontend/src/utils/download-workspace.ts      |   7 +-
 frontend/src/utils/get-valid-fallback-host.ts |  19 -
 .../utils/suggestions/non-repo-suggestions.ts |   4 +-
 frontend/src/utils/user-is-authenticated.ts   |  18 +-
 frontend/test-utils.tsx                       |   7 +-
 frontend/tests/fixtures/project.zip           |   0
 frontend/tests/redirect.spec.ts               |  61 ++
 frontend/tsconfig.json                        |   2 +-
 frontend/vite.config.ts                       |   2 +
 openhands/__init__.py                         |  14 +-
 openhands/agenthub/__init__.py                |   4 -
 .../agenthub/browsing_agent/browsing_agent.py |   4 +-
 .../agenthub/codeact_agent/codeact_agent.py   |  32 +-
 .../codeact_agent/function_calling.py         | 153 ++++-
 .../codeact_swe_agent/codeact_swe_agent.py    |   6 +-
 openhands/agenthub/delegator_agent/agent.py   |   8 +-
 openhands/agenthub/dummy_agent/agent.py       |   2 +-
 openhands/agenthub/micro/agent.py             |   8 +-
 openhands/agenthub/planner_agent/prompt.py    |   4 +-
 openhands/agenthub/searcher_agent/__init__.py |   4 -
 .../agenthub/searcher_agent/action_parser.py  | 153 -----
 openhands/agenthub/searcher_agent/agent.py    | 203 -------
 openhands/agenthub/searcher_agent/prompt.py   | 146 -----
 openhands/agenthub/supervisor_agent/agent.py  |   8 +-
 openhands/agenthub/tester_agent/__init__.py   |   4 -
 .../agenthub/tester_agent/action_parser.py    | 158 -----
 openhands/agenthub/tester_agent/agent.py      | 201 -------
 openhands/agenthub/tester_agent/prompt.py     | 329 -----------
 openhands/controller/agent_controller.py      | 303 ++++++----
 openhands/controller/state/state.py           |  51 +-
 openhands/controller/stuck.py                 |   2 +-
 openhands/core/cli.py                         |  59 +-
 openhands/core/config/agent_config.py         |   4 +-
 openhands/core/config/app_config.py           |   2 -
 openhands/core/loop.py                        |  50 ++
 openhands/core/main.py                        |  54 +-
 openhands/core/message.py                     |   2 +
 openhands/events/action/browse.py             |   4 +-
 openhands/events/action/message.py            |   2 +-
 openhands/events/event.py                     |   1 +
 openhands/events/observation/__init__.py      |   3 +-
 openhands/events/observation/browse.py        |  48 +-
 openhands/events/observation/error.py         |  15 +-
 openhands/events/stream.py                    | 102 +++-
 openhands/llm/llm.py                          | 192 ++++---
 openhands/memory/__init__.py                  |   3 +-
 openhands/memory/history.py                   | 224 --------
 openhands/runtime/action_execution_server.py  |   3 +-
 openhands/runtime/base.py                     |  62 +-
 openhands/runtime/browser/browser_env.py      |   6 +-
 openhands/runtime/builder/docker.py           |   2 +-
 openhands/runtime/builder/remote.py           |  35 +-
 openhands/runtime/impl/e2b/e2b_runtime.py     |   4 +-
 openhands/runtime/impl/e2b/sandbox.py         |   4 +-
 .../impl/eventstream/eventstream_runtime.py   | 252 ++++----
 openhands/runtime/impl/modal/modal_runtime.py |   6 +-
 .../runtime/impl/remote/remote_runtime.py     | 356 +++++-------
 .../plugins/agent_skills/file_editor/impl.py  |   6 +-
 openhands/runtime/utils/bash.py               |   8 +-
 openhands/runtime/utils/edit.py               |   9 +-
 openhands/runtime/utils/request.py            |  46 +-
 openhands/runtime/utils/runtime_build.py      |   4 +-
 .../utils/runtime_templates/Dockerfile.j2     |   1 +
 openhands/runtime/utils/tenacity_stop.py      |   5 +-
 openhands/security/analyzer.py                |   3 +-
 openhands/security/invariant/analyzer.py      |   1 +
 openhands/server/github.py                    | 128 +++++
 openhands/server/listen.py                    | 113 ++--
 openhands/server/middleware.py                |   4 +-
 openhands/server/session/agent_session.py     |  44 +-
 openhands/server/session/manager.py           |   6 +-
 openhands/server/session/session.py           |  80 ++-
 openhands/server/sheets_client.py             |  68 +++
 poetry.lock                                   |  22 +-
 pyproject.toml                                |   7 +-
 tests/runtime/test_stress_remote_runtime.py   | 231 ++++++++
 tests/unit/test_agent_controller.py           |  93 ++-
 tests/unit/test_codeact_agent.py              |   2 +-
 tests/unit/test_is_stuck.py                   | 379 ++++++------
 tests/unit/test_llm.py                        |   3 +
 tests/unit/test_memory.py                     |   2 +-
 tests/unit/test_micro_agents.py               |  14 +-
 tests/unit/test_prompt_caching.py             |  92 +--
 189 files changed, 5161 insertions(+), 3177 deletions(-)
 create mode 100644 evaluation/discoverybench/README.md
 create mode 100644 evaluation/discoverybench/eval_utils/README.md
 create mode 100644 evaluation/discoverybench/eval_utils/__init__.py
 create mode 100644 evaluation/discoverybench/eval_utils/eval_w_subhypo_gen.py
 create mode 100644 evaluation/discoverybench/eval_utils/lm_utils.py
 create mode 100644 evaluation/discoverybench/eval_utils/openai_helpers.py
 create mode 100644 evaluation/discoverybench/eval_utils/openai_semantic_gen_prompts.py
 create mode 100644 evaluation/discoverybench/eval_utils/response_parser.py
 create mode 100644 evaluation/discoverybench/run_infer.py
 create mode 100755 evaluation/discoverybench/scripts/run_infer.sh
 create mode 100644 evaluation/integration_tests/tests/t06_github_pr_browsing.py
 create mode 100644 evaluation/scienceagentbench/Dockerfile
 create mode 100644 evaluation/scienceagentbench/Dockerfile.evaluator
 create mode 100644 evaluation/scienceagentbench/README.md
 create mode 100644 evaluation/scienceagentbench/post_proc.py
 create mode 100644 evaluation/scienceagentbench/run_infer.py
 create mode 100755 evaluation/scienceagentbench/scripts/run_infer.sh
 create mode 100644 frontend/playwright.config.ts
 create mode 100644 frontend/src/components/analytics-consent-form-modal.tsx
 create mode 100644 frontend/src/components/github-repositories-suggestion-box.tsx
 create mode 100644 frontend/src/routes/set-consent.ts
 delete mode 100644 frontend/src/utils/get-valid-fallback-host.ts
 create mode 100644 frontend/tests/fixtures/project.zip
 create mode 100644 frontend/tests/redirect.spec.ts
 delete mode 100644 openhands/agenthub/searcher_agent/__init__.py
 delete mode 100644 openhands/agenthub/searcher_agent/action_parser.py
 delete mode 100644 openhands/agenthub/searcher_agent/agent.py
 delete mode 100644 openhands/agenthub/searcher_agent/prompt.py
 delete mode 100644 openhands/agenthub/tester_agent/__init__.py
 delete mode 100644 openhands/agenthub/tester_agent/action_parser.py
 delete mode 100644 openhands/agenthub/tester_agent/agent.py
 delete mode 100644 openhands/agenthub/tester_agent/prompt.py
 create mode 100644 openhands/core/loop.py
 delete mode 100644 openhands/memory/history.py
 create mode 100644 openhands/server/github.py
 create mode 100644 openhands/server/sheets_client.py
 create mode 100644 tests/runtime/test_stress_remote_runtime.py

diff --git a/.github/ISSUE_TEMPLATE/bug_template.yml b/.github/ISSUE_TEMPLATE/bug_template.yml
index ad618f82e5a1..7a6a0ba244f6 100644
--- a/.github/ISSUE_TEMPLATE/bug_template.yml
+++ b/.github/ISSUE_TEMPLATE/bug_template.yml
@@ -31,6 +31,8 @@ body:
       options:
         - Docker command in README
         - Development workflow
+        - app.all-hands.dev
+        - Other
       default: 0
 
   - type: input
diff --git a/.github/workflows/ghcr-build.yml b/.github/workflows/ghcr-build.yml
index 25d05b9a0ca7..a7398961da3c 100644
--- a/.github/workflows/ghcr-build.yml
+++ b/.github/workflows/ghcr-build.yml
@@ -401,7 +401,7 @@ jobs:
           exit 1
   update_pr_description:
     name: Update PR Description
-    if: github.event_name == 'pull_request' && !github.event.pull_request.head.repo.fork
+    if: github.event_name == 'pull_request' && !github.event.pull_request.head.repo.fork && github.actor != 'dependabot[bot]'
     needs: [ghcr_build_runtime]
     runs-on: ubuntu-latest
     steps:
@@ -424,9 +424,9 @@ jobs:
             -p 3000:3000 \
             -v /var/run/docker.sock:/var/run/docker.sock \
             --add-host host.docker.internal:host-gateway \
-            -e SANDBOX_RUNTIME_CONTAINER_IMAGE=ghcr.io/all-hands-ai/runtime:$SHORT_SHA-nikolaik \
+            -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:$SHORT_SHA-nikolaik \
             --name openhands-app-$SHORT_SHA \
-            ghcr.io/all-hands-ai/runtime:$SHORT_SHA"
+            docker.all-hands.dev/all-hands-ai/openhands:$SHORT_SHA"
 
           PR_BODY=$(gh pr view $PR_NUMBER --json body --jq .body)
 
diff --git a/.github/workflows/openhands-resolver.yml b/.github/workflows/openhands-resolver.yml
index fa253905e1f5..1e5360afba0b 100644
--- a/.github/workflows/openhands-resolver.yml
+++ b/.github/workflows/openhands-resolver.yml
@@ -3,6 +3,8 @@ name: Resolve Issues with OpenHands
 on:
   issues:
     types: [labeled]
+  pull_request:
+    types: [labeled]
 
 jobs:
   call-openhands-resolver:
diff --git a/.gitignore b/.gitignore
index a4bc03c4eeb1..0cc7d149d781 100644
--- a/.gitignore
+++ b/.gitignore
@@ -174,6 +174,7 @@ evaluation/bird/data
 evaluation/gaia/data
 evaluation/gorilla/data
 evaluation/toolqa/data
+evaluation/scienceagentbench/benchmark
 
 # frontend
 
diff --git a/README.md b/README.md
index 39e9e746edfc..e67bd0599478 100644
--- a/README.md
+++ b/README.md
@@ -12,7 +12,7 @@
   <a href="https://codecov.io/github/All-Hands-AI/OpenHands?branch=main"><img alt="CodeCov" src="https://img.shields.io/codecov/c/github/All-Hands-AI/OpenHands?style=for-the-badge&color=blue"></a>
   <a href="https://github.com/All-Hands-AI/OpenHands/blob/main/LICENSE"><img src="https://img.shields.io/github/license/All-Hands-AI/OpenHands?style=for-the-badge&color=blue" alt="MIT License"></a>
   <br/>
-  <a href="https://join.slack.com/t/opendevin/shared_invite/zt-2oikve2hu-UDxHeo8nsE69y6T7yFX_BA"><img src="https://img.shields.io/badge/Slack-Join%20Us-red?logo=slack&logoColor=white&style=for-the-badge" alt="Join our Slack community"></a>
+  <a href="https://join.slack.com/t/openhands-ai/shared_invite/zt-2tom0er4l-JeNUGHt_AxpEfIBstbLPiw"><img src="https://img.shields.io/badge/Slack-Join%20Us-red?logo=slack&logoColor=white&style=for-the-badge" alt="Join our Slack community"></a>
   <a href="https://discord.gg/ESHStjSjD4"><img src="https://img.shields.io/badge/Discord-Join%20Us-purple?logo=discord&logoColor=white&style=for-the-badge" alt="Join our Discord community"></a>
   <a href="https://github.com/All-Hands-AI/OpenHands/blob/main/CREDITS.md"><img src="https://img.shields.io/badge/Project-Credits-blue?style=for-the-badge&color=FFE165&logo=github&logoColor=white" alt="Credits"></a>
   <br/>
@@ -40,7 +40,7 @@ system requirements and more information.
 ```bash
 docker pull docker.all-hands.dev/all-hands-ai/runtime:0.12-nikolaik
 
-docker run -it --rm --pull=always \
+docker run -it --pull=always \
     -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.12-nikolaik \
     -v /var/run/docker.sock:/var/run/docker.sock \
     -p 3000:3000 \
@@ -59,7 +59,8 @@ works best, but you have [many options](https://docs.all-hands.dev/modules/usage
 
 You can also [connect OpenHands to your local filesystem](https://docs.all-hands.dev/modules/usage/runtimes),
 run OpenHands in a scriptable [headless mode](https://docs.all-hands.dev/modules/usage/how-to/headless-mode),
-or interact with it via a [friendly CLI](https://docs.all-hands.dev/modules/usage/how-to/cli-mode).
+interact with it via a [friendly CLI](https://docs.all-hands.dev/modules/usage/how-to/cli-mode),
+or run it on tagged issues with [a github action](https://github.com/All-Hands-AI/OpenHands-resolver).
 
 Visit [Installation](https://docs.all-hands.dev/modules/usage/installation) for more information and setup instructions.
 
@@ -92,7 +93,7 @@ For details, please check [CONTRIBUTING.md](./CONTRIBUTING.md).
 Whether you're a developer, a researcher, or simply enthusiastic about OpenHands, we'd love to have you in our community.
 Let's make software engineering better together!
 
-- [Slack workspace](https://join.slack.com/t/opendevin/shared_invite/zt-2oikve2hu-UDxHeo8nsE69y6T7yFX_BA) - Here we talk about research, architecture, and future development.
+- [Slack workspace](https://join.slack.com/t/openhands-ai/shared_invite/zt-2tom0er4l-JeNUGHt_AxpEfIBstbLPiw) - Here we talk about research, architecture, and future development.
 - [Discord server](https://discord.gg/ESHStjSjD4) - This is a community-run server for general discussion, questions, and feedback.
 
 ## 📈 Progress
diff --git a/docs/i18n/fr/docusaurus-plugin-content-docs/current/usage/how-to/evaluation-harness.md b/docs/i18n/fr/docusaurus-plugin-content-docs/current/usage/how-to/evaluation-harness.md
index d027a0ead929..3f191053998f 100644
--- a/docs/i18n/fr/docusaurus-plugin-content-docs/current/usage/how-to/evaluation-harness.md
+++ b/docs/i18n/fr/docusaurus-plugin-content-docs/current/usage/how-to/evaluation-harness.md
@@ -161,7 +161,7 @@ Pour créer un workflow d'évaluation pour votre benchmark, suivez ces étapes :
            instruction=instruction,
            test_result=evaluation_result,
            metadata=metadata,
-           history=state.history.compatibility_for_eval_history_pairs(),
+           history=compatibility_for_eval_history_pairs(state.history),
            metrics=state.metrics.get() if state.metrics else None,
            error=state.last_error if state and state.last_error else None,
        )
@@ -260,7 +260,7 @@ def codeact_user_response(state: State | None) -> str:
         # vérifier si l'agent a essayé de parler à l'utilisateur 3 fois, si oui, faire savoir à l'agent qu'il peut abandonner
         user_msgs = [
             event
-            for event in state.history.get_events()
+            for event in state.history
             if isinstance(event, MessageAction) and event.source == 'user'
         ]
         if len(user_msgs) >= 2:
@@ -279,4 +279,3 @@ Cette fonction fait ce qui suit :
 3. Si l'agent a fait plusieurs tentatives, il lui donne la possibilité d'abandonner
 
 En utilisant cette fonction, vous pouvez garantir un comportement cohérent sur plusieurs exécutions d'évaluation et empêcher l'agent de rester bloqué en attendant une entrée humaine.
-
diff --git a/docs/i18n/zh-Hans/docusaurus-plugin-content-docs/current/usage/how-to/evaluation-harness.md b/docs/i18n/zh-Hans/docusaurus-plugin-content-docs/current/usage/how-to/evaluation-harness.md
index a50bb18502e2..eb99a30ea3fd 100644
--- a/docs/i18n/zh-Hans/docusaurus-plugin-content-docs/current/usage/how-to/evaluation-harness.md
+++ b/docs/i18n/zh-Hans/docusaurus-plugin-content-docs/current/usage/how-to/evaluation-harness.md
@@ -158,7 +158,7 @@ OpenHands 的主要入口点在 `openhands/core/main.py` 中。以下是它工
            instruction=instruction,
            test_result=evaluation_result,
            metadata=metadata,
-           history=state.history.compatibility_for_eval_history_pairs(),
+           history=compatibility_for_eval_history_pairs(state.history),
            metrics=state.metrics.get() if state.metrics else None,
            error=state.last_error if state and state.last_error else None,
        )
@@ -257,7 +257,7 @@ def codeact_user_response(state: State | None) -> str:
         # 检查代理是否已尝试与用户对话 3 次，如果是，让代理知道它可以放弃
         user_msgs = [
             event
-            for event in state.history.get_events()
+            for event in state.history
             if isinstance(event, MessageAction) and event.source == 'user'
         ]
         if len(user_msgs) >= 2:
diff --git a/docs/i18n/zh-Hans/docusaurus-plugin-content-docs/current/usage/how-to/headless-mode.md b/docs/i18n/zh-Hans/docusaurus-plugin-content-docs/current/usage/how-to/headless-mode.md
index 8beacdd208b6..bfcca8386ebe 100644
--- a/docs/i18n/zh-Hans/docusaurus-plugin-content-docs/current/usage/how-to/headless-mode.md
+++ b/docs/i18n/zh-Hans/docusaurus-plugin-content-docs/current/usage/how-to/headless-mode.md
@@ -58,4 +58,3 @@ docker run -it \
     ghcr.io/all-hands-ai/openhands:0.11 \
     python -m openhands.core.main -t "write a bash script that prints hi"
 ```
-
diff --git a/docs/modules/usage/how-to/evaluation-harness.md b/docs/modules/usage/how-to/evaluation-harness.md
index 622f7e5607ba..e4d1e5d15bc7 100644
--- a/docs/modules/usage/how-to/evaluation-harness.md
+++ b/docs/modules/usage/how-to/evaluation-harness.md
@@ -158,7 +158,7 @@ To create an evaluation workflow for your benchmark, follow these steps:
            instruction=instruction,
            test_result=evaluation_result,
            metadata=metadata,
-           history=state.history.compatibility_for_eval_history_pairs(),
+           history=compatibility_for_eval_history_pairs(state.history),
            metrics=state.metrics.get() if state.metrics else None,
            error=state.last_error if state and state.last_error else None,
        )
@@ -257,7 +257,7 @@ def codeact_user_response(state: State | None) -> str:
         # check if the agent has tried to talk to the user 3 times, if so, let the agent know it can give up
         user_msgs = [
             event
-            for event in state.history.get_events()
+            for event in state.history
             if isinstance(event, MessageAction) and event.source == 'user'
         ]
         if len(user_msgs) >= 2:
diff --git a/docs/modules/usage/how-to/gui-mode.md b/docs/modules/usage/how-to/gui-mode.md
index 8726922574a8..df5a070c01e5 100644
--- a/docs/modules/usage/how-to/gui-mode.md
+++ b/docs/modules/usage/how-to/gui-mode.md
@@ -19,6 +19,15 @@ OpenHands provides a user-friendly Graphical User Interface (GUI) mode for inter
 3. Enter the corresponding `API Key` for your chosen provider.
 4. Click "Save" to apply the settings.
 
+### GitHub Token Setup
+
+OpenHands automatically exports a `GITHUB_TOKEN` to the shell environment if it is available. This can happen in two ways:
+
+1. Locally (OSS): The user directly inputs their GitHub token.
+2. Online (SaaS): The token is obtained through GitHub OAuth authentication.
+
+When you reach the `/app` route, the app checks if a token is present. If it finds one, it sets it in the environment for the agent to use.
+
 ### Advanced Settings
 
 1. Toggle `Advanced Options` to access additional settings.
diff --git a/docs/modules/usage/runtimes.md b/docs/modules/usage/runtimes.md
index 92fa04b009ad..3c227ffaf74d 100644
--- a/docs/modules/usage/runtimes.md
+++ b/docs/modules/usage/runtimes.md
@@ -60,7 +60,6 @@ docker run # ...
     -e SANDBOX_REMOTE_RUNTIME_API_URL="https://runtime.app.all-hands.dev" \
     -e SANDBOX_API_KEY="your-all-hands-api-key" \
     -e SANDBOX_KEEP_REMOTE_RUNTIME_ALIVE="true" \
-    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.11-nikolaik \
     # ...
 ```
 
@@ -75,5 +74,4 @@ docker run # ...
     -e RUNTIME=modal \
     -e MODAL_API_TOKEN_ID="your-id" \
     -e MODAL_API_TOKEN_SECRET="your-secret" \
-    -e SANDBOX_RUNTIME_CONTAINER_IMAGE=docker.all-hands.dev/all-hands-ai/runtime:0.11-nikolaik \
 ```
diff --git a/evaluation/EDA/run_infer.py b/evaluation/EDA/run_infer.py
index 2c896939a751..fb5df3b44f01 100644
--- a/evaluation/EDA/run_infer.py
+++ b/evaluation/EDA/run_infer.py
@@ -8,6 +8,7 @@
 from evaluation.utils.shared import (
     EvalMetadata,
     EvalOutput,
+    compatibility_for_eval_history_pairs,
     make_metadata,
     prepare_dataset,
     reset_logger_for_multiprocessing,
@@ -34,7 +35,7 @@ def codeact_user_response_eda(state: State) -> str:
 
     # retrieve the latest model message from history
     if state.history:
-        model_guess = state.history.get_last_agent_message()
+        model_guess = state.get_last_agent_message()
 
     assert game is not None, 'Game is not initialized.'
     msg = game.generate_user_response(model_guess)
@@ -139,7 +140,7 @@ def process_instance(
     if state is None:
         raise ValueError('State should not be None.')
 
-    final_message = state.history.get_last_agent_message()
+    final_message = state.get_last_agent_message()
 
     logger.info(f'Final message: {final_message} | Ground truth: {instance["text"]}')
     test_result = game.reward()
@@ -148,7 +149,7 @@ def process_instance(
     # history is now available as a stream of events, rather than list of pairs of (Action, Observation)
     # for compatibility with the existing output format, we can remake the pairs here
     # remove when it becomes unnecessary
-    histories = state.history.compatibility_for_eval_history_pairs()
+    histories = compatibility_for_eval_history_pairs(state.history)
 
     # Save the output
     output = EvalOutput(
diff --git a/evaluation/README.md b/evaluation/README.md
index 7eb59c7b8d5a..8be0822875f8 100644
--- a/evaluation/README.md
+++ b/evaluation/README.md
@@ -84,4 +84,3 @@ all the preprocessing/evaluation/analysis scripts.
 - Raw data and experimental records should not be stored within this repo.
 - For model outputs, they should be stored at [this huggingface space](https://huggingface.co/spaces/OpenHands/evaluation) for visualization.
 - Important data files of manageable size and analysis scripts (e.g., jupyter notebooks) can be directly uploaded to this repo.
-
diff --git a/evaluation/agent_bench/run_infer.py b/evaluation/agent_bench/run_infer.py
index d6fcc62e0798..acdf60fe4850 100644
--- a/evaluation/agent_bench/run_infer.py
+++ b/evaluation/agent_bench/run_infer.py
@@ -16,6 +16,7 @@
 from evaluation.utils.shared import (
     EvalMetadata,
     EvalOutput,
+    compatibility_for_eval_history_pairs,
     make_metadata,
     prepare_dataset,
     reset_logger_for_multiprocessing,
@@ -242,7 +243,7 @@ def process_instance(
         raw_ans = ''
 
         # retrieve the last agent message or thought
-        for event in state.history.get_events(reverse=True):
+        for event in reversed(state.history):
             if event.source == 'agent':
                 if isinstance(event, AgentFinishAction):
                     raw_ans = event.thought
@@ -271,7 +272,7 @@ def process_instance(
     # history is now available as a stream of events, rather than list of pairs of (Action, Observation)
     # for compatibility with the existing output format, we can remake the pairs here
     # remove when it becomes unnecessary
-    histories = state.history.compatibility_for_eval_history_pairs()
+    histories = compatibility_for_eval_history_pairs(state.history)
 
     metrics = state.metrics.get() if state.metrics else None
 
diff --git a/evaluation/aider_bench/run_infer.py b/evaluation/aider_bench/run_infer.py
index fa1bb9534a83..cddc4bfe7db9 100644
--- a/evaluation/aider_bench/run_infer.py
+++ b/evaluation/aider_bench/run_infer.py
@@ -15,6 +15,7 @@
 from evaluation.utils.shared import (
     EvalMetadata,
     EvalOutput,
+    compatibility_for_eval_history_pairs,
     make_metadata,
     prepare_dataset,
     reset_logger_for_multiprocessing,
@@ -250,7 +251,7 @@ def process_instance(
     # history is now available as a stream of events, rather than list of pairs of (Action, Observation)
     # for compatibility with the existing output format, we can remake the pairs here
     # remove when it becomes unnecessary
-    histories = state.history.compatibility_for_eval_history_pairs()
+    histories = compatibility_for_eval_history_pairs(state.history)
     metrics = state.metrics.get() if state.metrics else None
 
     # Save the output
diff --git a/evaluation/biocoder/run_infer.py b/evaluation/biocoder/run_infer.py
index 4535ccba4e4e..5ab4b3b88313 100644
--- a/evaluation/biocoder/run_infer.py
+++ b/evaluation/biocoder/run_infer.py
@@ -13,6 +13,7 @@
     EvalMetadata,
     EvalOutput,
     codeact_user_response,
+    compatibility_for_eval_history_pairs,
     make_metadata,
     prepare_dataset,
     reset_logger_for_multiprocessing,
@@ -299,7 +300,7 @@ def process_instance(
     # history is now available as a stream of events, rather than list of pairs of (Action, Observation)
     # for compatibility with the existing output format, we can remake the pairs here
     # remove when it becomes unnecessary
-    histories = state.history.compatibility_for_eval_history_pairs()
+    histories = compatibility_for_eval_history_pairs(state.history)
 
     test_result['generated'] = test_result['metadata']['1_copy_change_code']
 
diff --git a/evaluation/bird/run_infer.py b/evaluation/bird/run_infer.py
index adb498cd2eb1..248dbb66181c 100644
--- a/evaluation/bird/run_infer.py
+++ b/evaluation/bird/run_infer.py
@@ -16,6 +16,7 @@
 from evaluation.utils.shared import (
     EvalMetadata,
     EvalOutput,
+    compatibility_for_eval_history_pairs,
     make_metadata,
     prepare_dataset,
     reset_logger_for_multiprocessing,
@@ -46,7 +47,7 @@ def codeact_user_response(state: State) -> str:
         # check if the agent has tried to talk to the user 3 times, if so, let the agent know it can give up
         user_msgs = [
             event
-            for event in state.history.get_events()
+            for event in state.history
             if isinstance(event, MessageAction) and event.source == 'user'
         ]
         if len(user_msgs) > 2:
@@ -431,7 +432,7 @@ def execute_sql(db_path, sql):
     # history is now available as a stream of events, rather than list of pairs of (Action, Observation)
     # for compatibility with the existing output format, we can remake the pairs here
     # remove when it becomes unnecessary
-    histories = state.history.compatibility_for_eval_history_pairs()
+    histories = compatibility_for_eval_history_pairs(state.history)
 
     # Save the output
     output = EvalOutput(
diff --git a/evaluation/browsing_delegation/run_infer.py b/evaluation/browsing_delegation/run_infer.py
index c9fe2ebd18bc..5c1ab8c062e3 100644
--- a/evaluation/browsing_delegation/run_infer.py
+++ b/evaluation/browsing_delegation/run_infer.py
@@ -9,6 +9,7 @@
 from evaluation.utils.shared import (
     EvalMetadata,
     EvalOutput,
+    compatibility_for_eval_history_pairs,
     make_metadata,
     prepare_dataset,
     reset_logger_for_multiprocessing,
@@ -89,7 +90,7 @@ def process_instance(
     # history is now available as a stream of events, rather than list of pairs of (Action, Observation)
     # for compatibility with the existing output format, we can remake the pairs here
     # remove when it becomes unnecessary
-    histories = state.history.compatibility_for_eval_history_pairs()
+    histories = compatibility_for_eval_history_pairs(state.history)
 
     # find the last delegate action
     last_delegate_action = None
diff --git a/evaluation/discoverybench/README.md b/evaluation/discoverybench/README.md
new file mode 100644
index 000000000000..a0d8994709df
--- /dev/null
+++ b/evaluation/discoverybench/README.md
@@ -0,0 +1,37 @@
+# DiscoveryBench with OpenHands
+
+[DiscoveryBench](https://github.com/allenai/discoverybench/) [(Paper)](https://arxiv.org/abs/2407.01725v1) contains 264 tasks collected across 6 diverse domains, such as biology, economics, and sociology. It incorporates discovery workflows from published papers to approximate the real-world challenges faced by researchers.
+
+<p align="center">
+  <a href="[https://github.com/allenai/discoverybench](https://github.com/allenai/discoverybench)">
+    <img src="https://raw.githubusercontent.com/allenai/discoverybench/refs/heads/main/assets/discoverybench-openhands-teaser.png" width="100%" alt="DiscoveryBench Background" />
+  </a>
+</p>
+
+
+## Setup Environment and LLM Configuration
+
+1. Please follow instructions mentioned [here](https://github.com/openlocus/OpenHands/blob/discoverybench-openhands-integration/evaluation/README.md#setup) to setup OpenHands development environment and LLMs locally
+
+2. Execute the bash script to start DiscoveryBench Evaluation
+
+```
+./evaluation/discoverybench/scripts/run_infer.sh [YOUR MODEL CONFIG]
+```
+Replace `[YOUR MODEL CONFIG]` with any model the model that you have set up in `config.toml`
+
+
+## Run Inference on DiscoveryBench Instances
+
+When the `run_infer.sh` script is started, it will automatically pull the latest DiscoveryBench instances & set up the agent environment. The OpenHands agent is invoked to process the task within this environment, producing a hypothesis. We then evaluate it against the “gold” hypothesis provided by DiscoveryBench. The evaluation result, along with the agent chat history is logged to `output.jsonl` under `evaluation_outputs`.
+
+
+```
+./evaluation/discoverybench/scripts/run_infer.sh [MODEL_CONFIG] [GIT_COMMIT] [AGENT] [EVAL_LIMIT] [NUM_WORKERS]
+```
+
+- `MODEL_CONFIG`: Name of the model you want to evaluate with
+- `GIT_COMMIT`: This should be the git commit hash or release tag for OpenHands, e.g., HEAD or a specific tag like 0.6.2.
+- `AGENT`: Use CoderActAgent, right now it only supports that.
+- `EVAL_LIMIT`: Number of samples to evaluate.
+- `NUM_WORKERS`: Number of workers to parallelize the evaluation process.
diff --git a/evaluation/discoverybench/eval_utils/README.md b/evaluation/discoverybench/eval_utils/README.md
new file mode 100644
index 000000000000..13c98ebaa8d2
--- /dev/null
+++ b/evaluation/discoverybench/eval_utils/README.md
@@ -0,0 +1,7 @@
+## DiscoveryBench Evaluation Utils
+
+- **`eval_w_subhypo_gen.py`**: Implements the DiscoveryBench logic for evaluating agent-generated hypotheses.
+- **`lm_utils.py`**: Provides utility functions necessary for the evaluation process.
+- **`openai_helpers.py`**: Includes helper functions for OpenAI-related tasks.
+- **`openai_semantic_gen_prompts.py`**: Contains prompts used for semantic generation.
+- **`response_parser.py`**: Handles the parsing of agent-generated hypotheses.
diff --git a/evaluation/discoverybench/eval_utils/__init__.py b/evaluation/discoverybench/eval_utils/__init__.py
new file mode 100644
index 000000000000..e69de29bb2d1
diff --git a/evaluation/discoverybench/eval_utils/eval_w_subhypo_gen.py b/evaluation/discoverybench/eval_utils/eval_w_subhypo_gen.py
new file mode 100644
index 000000000000..a80df8279cfb
--- /dev/null
+++ b/evaluation/discoverybench/eval_utils/eval_w_subhypo_gen.py
@@ -0,0 +1,538 @@
+import json
+import logging
+
+from openai import OpenAI
+
+from .lm_utils import run_chatgpt_query_multi_turn
+from .openai_helpers import get_response
+
+logging.basicConfig(
+    format='%(asctime)s - %(levelname)s - %(name)s -   %(message)s',
+    datefmt='%m/%d/%Y %H:%M:%S',
+    level=logging.INFO,
+)
+logger = logging.getLogger(__name__)
+
+
+def get_score_from_answer(type, answer):
+    if type == 'context':
+        answer = answer.replace('Answer:', '').strip()
+        if answer.startswith('A)'):
+            return 1.0
+        elif answer.startswith('B)'):
+            return 0.0
+        return -1.0
+
+    elif type == 'var':
+        try:
+            var_json = json.loads(answer)
+            # print(f"var_json:{var_json}")
+            p = 0.0
+            r = 0.0
+            f1 = 0.0
+            if var_json['sizeB']:
+                p = var_json['intersection'] / var_json['sizeB']
+            if var_json['sizeA']:
+                r = var_json['intersection'] / var_json['sizeA']
+            if p > 0.0 and r > 0.0:
+                f1 = (2 * p * r) / (p + r)
+            else:
+                f1 = 0.0
+            eval_rec = {
+                'p': p,
+                'r': r,
+                'f1': f1,
+                'sizeA': var_json['sizeA'],
+                'sizeB': var_json['sizeB'],
+                'intersection': var_json['intersection'],
+                'explanation': var_json['explanation'],
+            }
+            print(f'var_eval: {eval_rec}')
+            return eval_rec
+        except Exception:  # COMMENT: added Exception
+            return {'p': -1.0, 'r': -1.0, 'f1': -1.0}
+    elif type == 'rel':
+        print(answer)
+        rel_json = json.loads(answer)
+        answer_str = rel_json['answer'].strip()
+        if answer_str.startswith('A') or 'very similar' in answer_str:
+            return 1.0
+        elif (
+            answer_str.startswith('B') or 'similar but general than HypoA' in answer_str
+        ):
+            return 0.5
+        elif answer_str.startswith('C') or 'different' in answer_str:
+            return 0.0
+        return -1.0
+    return -1.0
+
+
+def ask_dimension_question(
+    query,
+    gold_hypo,
+    gold_workflow,
+    gen_hypo,
+    gen_workflow,
+    dataset_meta,
+    llm_used,
+    dimension,
+    dataset_type,
+    use_column_metadata=True,
+):
+    dimension_question = ''
+    answer = ''
+    score = 0.0
+    if dimension == 'var':
+        score = {'p': -1.0, 'r': -1.0, 'f1': -1.0}
+    num_tokens = 256
+    num_retries = 1
+    json_response = False
+
+    messages = [
+        {
+            'role': 'system',
+            'content': 'You are an AI assistant that helps evaluate a data-driven hypothesis. You are a helpful assistant who is not talkative. You only respond with the exact answer to a query without additional conversation.',
+        },
+    ]
+    if dimension == 'context':
+        dimension_question = """\
+        Question: Is HypoB defined in the same context as HypoA?
+        (Context refers to assumptions/stratification under which the hypotheses are defined.)
+        Options: A) same   B) different
+        What is your answer?"""
+    elif dimension == 'var':
+        dimension_question = """\
+        Question: For both HypoA and HypoB, what are the different variables found in the hypotheses? \
+        Return your answer as a JSON object in the following format:
+        ```json
+        {{
+        "sizeA": num of variables used in HypoA
+        "sizeB": num of variables used in HypoB
+        "intersection": num of variables common in HypoA and HypoB. Use *fuzzy matching* to determine intersection, accounting for paraphrases or slightly different surface forms
+        "explanation": a short text explanation about the variables
+        }}```
+        Answer:"""
+        num_tokens = 512
+        num_retries = 1
+        json_response = True
+    elif dimension == 'rel':
+        dimension_question = """\
+        Question: Does HypoB exhibit the same relation as HypoA?
+        Compare using following example hierarchy of relationships (based on specificity): \
+        "there exists a relationship" > "positive relationship" > "positive AND (linear OR quadratic)" > "positive AND linear".
+        Options: A) very similar B) similar but general than HypoA C) different
+        Return your answer as a JSON object in the following format:
+        ```json
+        {{
+        "answer": one of the options from A) very similar B) similar but general than HypoA C) different
+        "explanation": a short text explanation about the relationship comparison
+        }}```
+        Answer:"""
+        num_tokens = 512
+        num_retries = 1
+        json_response = True
+
+    datasets_json = prepare_dataset_metadata_json(
+        dataset_meta, dataset_type=dataset_type, use_column_metadata=use_column_metadata
+    )
+
+    dimension_question_str = f"""\
+        You are going to compare two natural-language hypotheses HypoA and HypoB accompanied with optional workflows: WorkflowA for HypoA and WorkflowB for HypoB. \
+        Both the hypotheses answer the natural language query "QUERY" over the dataset(s) described by dataset description(s) and column description(s) below. \
+        Compare HypoA and HypoB in terms of three aspects: Contexts, Variables, and Relations. \
+        E.g., for the hypothesis "From 1995 to 2009, the number of sandhill cranes around the tundra (Indigilka River) surged by an astounding ~10X":
+        * Contexts refer to stratification of the data under which the given hypothesis is True. E.g., "For all women", "From 1995 to 2009".
+        * Variables refer to the set of variables (either dependent or independent) that are mentioned in the hypothesis. E.g., number of sandhill cranes, location.
+        * Relations refer to the form of relation between the variables. E.g., "surged by ~10x".
+
+        Answer following questions for a given pair of hypotheses, HypoA and HypoB, along with an explanation grounded on the QUERY and the DATASET(S).
+
+        Here is the metadata for the task:
+        ```json
+        {{
+        "datasets": {datasets_json},
+        "query": {query},
+        "HypoA": {gold_hypo},
+        "WorkflowA": {gold_workflow},
+        "HypoB": {gen_hypo},
+        "WorkflowB": {gen_workflow}
+        }}
+        ```
+
+        {dimension_question}"""
+
+    messages.append({'role': 'user', 'content': dimension_question_str})
+    for retry in range(num_retries):
+        response = run_chatgpt_query_multi_turn(
+            messages=messages,
+            model_name=llm_used,
+            max_tokens=num_tokens,
+            temperature=0,  # 0 for greedy best decoding
+            json_response=json_response,
+        )
+        if response is not None:  # COMMENT: changed from != to is not
+            break
+
+    if response is not None:  # COMMENT: changed from != to is not
+        answer = response.choices[0].message.content.strip()
+        score = get_score_from_answer(type=dimension, answer=answer)
+
+    return dimension_question, answer, score
+
+
+def prepare_dataset_metadata_json(dataset_meta, dataset_type, use_column_metadata=True):
+    if dataset_meta is None:  # COMMENT: changed from == to is None
+        return [
+            {
+                'dataset_description': '',
+                'columns': [],
+            }
+        ]
+    datasets_json = []
+    if dataset_type == 'real':
+        for d in dataset_meta['datasets']:
+            datasets_json.append(
+                {
+                    'dataset_description': d['description'],
+                    'columns': [
+                        {'name': col['name'], 'description': col['description']}
+                        for col in d['columns']['raw']
+                    ]
+                    if use_column_metadata
+                    else [],
+                }
+            )
+    else:
+        for d in dataset_meta['datasets']:
+            datasets_json.append(
+                {
+                    'dataset_description': d['description'],
+                    'columns': [
+                        {'name': col['name'], 'description': col['description']}
+                        for col in d['columns']
+                    ]
+                    if use_column_metadata
+                    else [],
+                }
+            )
+    return datasets_json
+
+
+def get_sub_hypotheses(
+    query,
+    hypo,
+    workflow,
+    dataset_meta,
+    llm_used,
+    dataset_type,
+    use_column_metadata=True,
+):
+    client = OpenAI()
+    extraction_prompt = """\
+        Given a set of dataset columns, a ground-truth hypothesis, and the analysis workflow used, your task is to extract three dimensions that define the hypothesis: Context, Variables, and Relations. \
+        Here are the definitions for these dimensions:
+        - Contexts: Boundary conditions that limit the scope of a hypothesis. E.g., “for men over \
+        the age of 30”, “in Asia and Europe”. If the context applies to the full dataset, then extract the context from the dataset_descrption.
+        - Variables: Known concepts that interact in a meaningful way under a given context to \
+        produce the hypothesis. E.g., gender, age, income, or "None" if there is no interacting variable.
+        - Relations: Interactions between a given set of variables under a given context to produce \
+        the hypothesis. E.g., “quadratic relationship”, “inversely proportional”, piecewise conditionals, \
+        or "None" if there is no interacting relationship.
+        Make sure to only use the information present in the hypothesis and the workflow. Do not add any new information. \
+        For each dimension, be specific, and do not omit any important details.
+
+        Here is the metadata for the task:
+        ```json
+        {
+        "datasets": %s,
+        "hypothesis": "%s",
+        "workflow": "%s"
+        }
+        ```
+
+        Return your answer as a JSON object in the following format:
+        ```json
+        {
+        "sub_hypo": [
+            {
+                "text": the hypothesis in natural language,
+                "context": a short text description of the context of the hypothesis,
+                "variables": a list of columns involved in the hypothesis,
+                "relations": a short text description of the relationship between the variables of the hypothesis
+            },
+            ...
+        ]
+        }```
+        """
+    datasets_json = prepare_dataset_metadata_json(
+        dataset_meta, dataset_type, use_column_metadata=use_column_metadata
+    )
+    _prompt = extraction_prompt % (datasets_json, hypo, workflow)
+    sub_hypo_json = get_response(client, _prompt, model=llm_used, max_retry=1)
+
+    if sub_hypo_json is not None:  # COMMENT: changed from != to is not
+        # print(f"full hypothesis: {hypo}")
+        print(f'sub_hypo_json: {sub_hypo_json}')
+    else:
+        sub_hypo_json = {
+            'sub_hypo': [],
+        }
+
+    sub_hypo_json['full_hypo'] = hypo
+
+    return sub_hypo_json
+
+
+def match_context_with_gpt(
+    gold_hyp, gold_context, pred_hyp, pred_context, model='gpt-3.5-turbo'
+):
+    prompt = f"""\
+        Given a gold hypothesis, a gold context, a predicted hypothesis, and a predicted context, your task is \
+        to determine if the predicted context semantically matches the ground-truth context. \
+        Here is the definition for Context: Boundary conditions that limit the scope of a sub-hypothesis. E.g., “for men over the age of 30”, “in Asia and Europe”. If the context applies to the full dataset, then the context is derived from the dataset_descrption. \
+        Here is the definition for Context: Boundary conditions that limit the scope of a sub-hypothesis. E.g., “for men over the age of 30”, “in Asia and Europe”. If the context applies to the full dataset, then the context is derived from the dataset_descrption. \
+        If the predicted context matches the gold context, return true, otherwise return false.
+        If both gold and predicted hypotheses are defined over the context of the full dataset, then also return true.
+        If both gold and predicted hypotheses are defined over the context of the full dataset, then also return true.
+
+        Here is the metadata for the task:
+        ```json
+        {{
+            "gold_hypothesis": "{gold_hyp}",
+            "gold_context": "{gold_context}",
+            "predicted_hypothesis": "{pred_hyp}",
+            "predicted_context": "{pred_context}"
+        }}
+        ```
+
+        Return your answer as a JSON object in the following format:
+        ```json
+        {{
+            "match": true or false
+        }}
+        ```"""
+
+    client = OpenAI()
+    output = get_response(client, prompt, model=model)
+    return output.get('match', False)
+
+
+def is_matching_context(gold_hyp, gold_context, pred_hyp, pred_context, llm_used):
+    if gold_context == pred_context:
+        return True
+    if 'None' in [gold_context, pred_context]:
+        return False
+    return match_context_with_gpt(
+        gold_hyp, gold_context, pred_hyp, pred_context, model=llm_used
+    )
+
+
+def run_eval_gold_vs_gen_NL_subhypo(
+    query,
+    gold_hypo,
+    gold_workflow,
+    gen_hypo,
+    gen_workflow,
+    dataset_meta,
+    llm_used,
+    context_score,
+    dataset_type,
+    use_column_metadata=True,
+):
+    # GPT-4 based evaluation to evaluate generated hypothesis in terms of context, variables, relation
+
+    eval_rec = {
+        'query': query,
+        'HypoA': gold_hypo,
+        'WorkflowA': gold_workflow,
+        'HypoB': gen_hypo,
+        'WorkflowB': gen_workflow,
+    }
+
+    for dimension in ['var', 'rel']:
+        question, answer, score = ask_dimension_question(
+            query,
+            gold_hypo,
+            gold_workflow,
+            gen_hypo,
+            gen_workflow,
+            dataset_meta,
+            llm_used,
+            dimension=dimension,
+            dataset_type=dataset_type,
+            use_column_metadata=use_column_metadata,
+        )
+
+        eval_rec[dimension] = {'question': question, 'answer': answer, 'score': score}
+
+    eval_rec['context'] = context_score
+    eval_rec['accuracy_score'] = (
+        1.0
+        * eval_rec['context']['score']
+        * eval_rec['var']['score']['f1']
+        * eval_rec['rel']['score']
+    )
+
+    return eval_rec
+
+
+def run_eval_gold_vs_gen_NL_hypo_workflow(
+    query,
+    gold_hypo,
+    gold_workflow,
+    gen_hypo,
+    gen_workflow,
+    dataset_meta,
+    llm_used,
+    dataset_type,
+    use_column_metadata=True,
+):
+    # Input: Dataset Metadata, Query, Gold {Hg, Wg}, Predicted {Hp, Wp}
+    # Output: eval_rec json includes final_score
+
+    # Procedure:
+    # Dataset Metadata, Query, Gold {Hg, Wg}, Pred {Hg, Wg}
+    # Gold: [Hg1, Hg2] (compute on the fly) Hg1 is a NL form of subhypothesis
+    # Predicted: [Hp1, Hp2] (compute on the fly)
+
+    # Compute Intersection: [(Hg_i, Hp_j), …]  # tuples of (gold,pred) that matched with context (do this w/o explicit extraction)
+    # # filter so that a gold context and a predicted context are only attached to one tuple
+    # Compute recall_context (programmatically)
+
+    # r_v_list = []
+    # For (Hg_i, Hp_j) in the intersection:
+    #             With Hg_i, Hp_j in NL, ask GPT4 → #variables and #intersection and a paragraph explanation and programmatically calculate f1_v
+    # Hg_i, Hp_j in NL, ask GPT4 → matching score (0, 0.5 or 1) : A) very similar B) similar but general than HypoA C) different + explanation
+    # 	r_v_list ← f1_v * score_r
+    # accuracy_score = mean(r_v_list)
+    # score =   [ recall_context * mean over predicted context(context_score * var_score *rel_score )]
+
+    # recall_context = 1.0  # COMMENT: never used
+    eval_rec = {
+        'query': query,
+        'HypoA': gold_hypo,
+        'WorkflowA': gold_workflow,
+        'HypoB': gen_hypo,
+        'WorkflowB': gen_workflow,
+    }
+
+    gold_sub_hypo_json = get_sub_hypotheses(
+        query=query,
+        hypo=gold_hypo,
+        workflow=gold_workflow,
+        dataset_meta=dataset_meta,
+        llm_used=llm_used,
+        dataset_type=dataset_type,
+        use_column_metadata=use_column_metadata,
+    )
+    if len(gold_sub_hypo_json['sub_hypo']) == 0:
+        gold_sub_hypo_json['sub_hypo'] = [
+            {
+                'text': gold_hypo,
+                'context': 'None',
+                'variables': [],
+                'relations': '',
+                'explanation': 'unable to segment',
+            }
+        ]
+    print(f'gold_sub_hypo_json: {gold_sub_hypo_json}')
+
+    gen_sub_hypo_json = get_sub_hypotheses(
+        query=query,
+        hypo=gen_hypo,
+        workflow=gen_workflow,
+        dataset_meta=dataset_meta,
+        llm_used=llm_used,
+        dataset_type=dataset_type,
+        use_column_metadata=use_column_metadata,
+    )
+    if len(gen_sub_hypo_json['sub_hypo']) == 0:
+        gen_sub_hypo_json['sub_hypo'] = [
+            {
+                'text': gen_hypo,
+                'context': 'None',
+                'variables': [],
+                'relations': '',
+                'explanation': 'unable to segment',
+            }
+        ]
+    print(f'gen_sub_hypo_json: {gen_sub_hypo_json}')
+
+    eval_rec['gold_sub_hypo'] = gold_sub_hypo_json
+    eval_rec['gen_sub_hypo'] = gen_sub_hypo_json
+
+    gold_subh_covered = []
+    gen_subh_to_gold_subh = dict()
+    gen_gold_subh_to_context = dict()
+
+    for p_id, gen_subh in enumerate(gen_sub_hypo_json['sub_hypo']):
+        gen_subh_to_gold_subh[p_id] = -1
+
+        for g_id, gold_subh in enumerate(gold_sub_hypo_json['sub_hypo']):
+            if g_id in gold_subh_covered:
+                continue
+
+            # match context
+            context_bool = is_matching_context(
+                gold_subh['text'],
+                gold_subh.get('context', ''),
+                gen_subh['text'],
+                gen_subh.get('context', ''),
+                llm_used,
+            )
+            if context_bool:
+                context_score = 1.0
+            else:
+                context_score = 0.0
+
+            if context_score == 1.0:  # match only when context_score = 1.0
+                gen_subh_to_gold_subh[p_id] = g_id
+                gold_subh_covered.append(g_id)
+                gen_gold_subh_to_context[f'P{p_id}||G{g_id}'] = {
+                    'question': f"""Comapring: GoldH: {gold_subh["text"]}, GoldC: {gold_subh['context']}\nGenH: {gen_subh['text']}, GenC: {gen_subh['context']}""",
+                    'answer': context_bool,
+                    'score': context_score,
+                }
+                break
+
+    print(f'gen_subh_to_gold_subh: {gen_subh_to_gold_subh}')
+    eval_rec['gen_subh_to_gold_subh'] = gen_subh_to_gold_subh
+    eval_rec['gold_subh_covered'] = gold_subh_covered
+    matched_gold_gen_subh_evals = dict()
+    sum_accuracy_score = 0.0
+    for p_id, g_id in gen_subh_to_gold_subh.items():
+        if g_id >= 0:
+            key = f'P{p_id}||G{g_id}'
+            context_score = gen_gold_subh_to_context[key]
+            subh_eval_rec = run_eval_gold_vs_gen_NL_subhypo(
+                query,
+                gold_hypo,
+                gold_workflow,
+                gen_hypo,
+                gen_workflow,
+                dataset_meta,
+                llm_used,
+                context_score,
+                dataset_type=dataset_type,
+                use_column_metadata=use_column_metadata,
+            )
+            sum_accuracy_score += subh_eval_rec['accuracy_score']
+            matched_gold_gen_subh_evals[key] = subh_eval_rec
+
+    eval_rec['matched_gold_gen_subh_evals'] = matched_gold_gen_subh_evals
+    eval_rec['recall_context'] = (
+        len(gold_subh_covered) / len(gold_sub_hypo_json['sub_hypo'])
+        if len(gold_sub_hypo_json['sub_hypo'])
+        else 0.0
+    )
+    mean_accuracy_score = (
+        sum_accuracy_score / len(gen_subh_to_gold_subh)
+        if len(gen_subh_to_gold_subh)
+        else 0.0
+    )
+    eval_rec['mean_accuracy_score'] = mean_accuracy_score
+    final_score = eval_rec['recall_context'] * mean_accuracy_score
+    eval_rec['final_score'] = final_score
+    print(f'eval_rec: {json.dumps(eval_rec, indent=2)}')
+
+    return eval_rec
diff --git a/evaluation/discoverybench/eval_utils/lm_utils.py b/evaluation/discoverybench/eval_utils/lm_utils.py
new file mode 100644
index 000000000000..10486ee82294
--- /dev/null
+++ b/evaluation/discoverybench/eval_utils/lm_utils.py
@@ -0,0 +1,64 @@
+import os
+import sys
+import time
+
+from openai import OpenAI
+from tenacity import (
+    retry,
+    stop_after_attempt,  # type: ignore
+    wait_random_exponential,  # type: ignore
+)
+
+if sys.version_info >= (3, 8):
+    from typing import Literal
+else:
+    from typing_extensions import Literal
+
+
+Model = Literal['gpt-4', 'gpt-3.5-turbo', 'text-davinci-003']
+
+OpenAI.api_key = os.getenv('OPENAI_API_KEY')
+OPENAI_GEN_HYP = {
+    'temperature': 0,
+    'max_tokens': 250,
+    'top_p': 1.0,
+    'frequency_penalty': 0,
+    'presence_penalty': 0,
+}
+
+
+@retry(wait=wait_random_exponential(min=1, max=60), stop=stop_after_attempt(6))
+def run_chatgpt_query_multi_turn(
+    messages,
+    model_name='gpt-4-turbo',  # pass "gpt4" for more recent model output
+    max_tokens=256,
+    temperature=0.0,
+    json_response=False,
+):
+    response = None
+    num_retries = 3
+    retry = 0
+    while retry < num_retries:
+        retry += 1
+        try:
+            client = OpenAI()
+
+            if json_response:
+                response = client.chat.completions.create(
+                    model=model_name,
+                    response_format={'type': 'json_object'},
+                    messages=messages,
+                    **OPENAI_GEN_HYP,
+                )
+            else:
+                response = client.chat.completions.create(
+                    model=model_name, messages=messages, **OPENAI_GEN_HYP
+                )
+            break
+
+        except Exception as e:
+            print(e)
+            print('GPT error. Retrying in 2 seconds...')
+            time.sleep(2)
+
+    return response
diff --git a/evaluation/discoverybench/eval_utils/openai_helpers.py b/evaluation/discoverybench/eval_utils/openai_helpers.py
new file mode 100644
index 000000000000..95ab23cf9c2e
--- /dev/null
+++ b/evaluation/discoverybench/eval_utils/openai_helpers.py
@@ -0,0 +1,190 @@
+import json
+
+
+def OPENAI_TOPIC_GEN_MESSAGES(n=10):
+    return [
+        {
+            'role': 'system',
+            'content': 'You are a helpful assistant who is not talkative. You only respond with the exact answer to a query without additional conversation.',
+        },
+        {
+            'role': 'user',
+            'content': f'Given `n`, come up with a list of `n` distinct topics and their descriptions. The topics can be absolutely anything. Be as creative as possible. Return your answer as a JSON object. \n\nFor example, for `n`=3, a valid answer might be:\n```json\n{{"topics": [\n  {{"id": 1, "topic": "cooking", "description": "Related to recipes, ingredients, chefs, etc."}},\n  {{"id": 2, "topic": "sports", "description": "Related to players, stadiums, trophies, etc."}},\n  {{"id": 3, "topic": "antiquing", "description": "Related to unique items, history, etc."}}\n]}}```\n\nNow, give me a list for `n`={n}. Remember, pick diverse topics from everything possible. No consecutive topics should be broadly similar. Directly respond with the answer JSON object.',
+        },
+    ]
+
+
+OPENAI_GEN_HYP = {
+    'temperature': 1.0,
+    'max_tokens': 4096,
+    'top_p': 1.0,
+    'frequency_penalty': 0,
+    'presence_penalty': 0,
+}
+
+
+def OPENAI_SEMANTICS_GEN_MESSAGES(dependent, relationship, domain, domain_desc):
+    return [
+        {
+            'role': 'system',
+            'content': 'You are a helpful assistant who is not talkative. You only respond with the exact answer to a query without additional conversation.',
+        },
+        {
+            'role': 'user',
+            'content': f'Given the true relationship in a dataset and a given domain, your task is to come up with an interpretation of some real-world concepts that the relationship could be modeling from the provided domain. It\'s okay to be wrong, but suggest something reasonable. Try as much as possible to make sure that the TARGET is actually derivable from the other variables. Give your answer as a JSON object. Here\'s an example:\n\nRelationship for x2 = "(96.4 * x1 ** 3) + (88.72 * x5 ** 2) + (81.96 * x6 ** -2) + (28.13 * x3)  + (97.0) + (0 * x4)"\nDomain="Sales"\nDomain description="Related to product distribution, revenues, marketing, etc."\n\nBased on this, the following real-world concepts might be applicable:\n```json\n{{\n  "dependent": "x2",\n  "relationship": "(96.4 * x1 ** 3) + (88.72 * x5 ** 2) + (81.96 * x6 ** -2) + (28.13 * x3)  + (97.0) + (0 * x4)",\n  "domain": "Sales",\n  "trends": {{\n    "x1": "Positive, cubic factor",\n    "x2": "TARGET",\n    "x3": "Positive, linear factor",\n    "x4": "No relation",\n    "x5": "Positive quadratic factor",\n    "x6": "Positive, inverse quadratic factor"\n  }},\n  "interpretation": {{\n    "x2": {{"description": "Volume of product sales by area", "name": "sales_area", "is_target": true}},\n    "x1": {{"description": "Population by area", "name": "pop_area"}},\n    "x3": {{"description": "Advertising spending", "name": "ad_spend"}},\n    "x4": {{"description": "Gender ratio of marketing team", "name": "gdr_ratio_mkt_team"}},\n    "x5": {{"description": "Intensity of marketing campaign", "name": "mkt_intensity"}}\n  }},\n    "x6": {{"description": "Distance to distribution center", "name": "dist_to_distr_ctr"}}\n}}```\n\nHere\'s a new test question:\nRelationship for {dependent} = "{relationship}"\nDomain = "{domain}"\nDomain description="{domain_desc}"\n\nRespond only with the answer JSON. Make sure that you do not forget to include the TARGET variable in the interpretation object.',
+        },
+    ]
+
+
+def OPENAI_SEMANTICS_GEN_W_MAP_MESSAGES(
+    dependent, relationship, domain, domain_desc, mapping
+):
+    return [
+        {
+            'role': 'system',
+            'content': 'You are a helpful assistant who is not talkative. You only respond with the exact answer to a query without additional conversation.',
+        },
+        {
+            'role': 'user',
+            'content': f'Given a partial mapping from variables to real-world concepts and a true relationship in a dataset, your task is to come up with an interpretation of real-world concepts for the variables without any assigned mapping (those starting with x). Suggest something reasonable. The dependent variable must be derivable only from the other variables in the dependent relationship. Give your answer as a JSON object. Here\'s an example:\n\nExample partial mapping and relationship:\n```json\n{{\n  "domain": "Sales",\n  "domain_description": "Related to product distribution, revenues, marketing, etc.",\n  "variable_mapping": {{\n    "x1": {{"description": "Population by area", "name": "pop_area"}},\n    "x2": {{"description": "Volume of product sales by area", "name": "sales_area"}},\n    "x4": {{"description": "Gender ratio of marketing team", "name": "gdr_ratio_mkt_team"}},\n    "x6": {{"description": "Distance to distribution center", "name": "dist_to_distr_ctr"}}\n  }},\n  "dependent_variable": "sales_area",\n  "dependent_relationship": "(96.4 * pop_area ** 3) + (88.72 * x5 ** 2) + (81.96 * dist_to_distr_ctr ** -2) + (28.13 * x3)  + (97.0)"\n}}```\nBased on this, an example answer would be:\n```json\n{{\n  "dependent_variable": "sales_area",\n  "missing_mapping": ["x3", "x5"],\n  "trends": {{\n    "x3": "Positive, linear factor",\n    "x5": "Positive quadratic factor"\n  }},\n  "interpretation": {{\n    "x3": {{"description": "Advertising spending", "name": "ad_spend"}},\n    "x5": {{"description": "Intensity of marketing campaign", "name": "mkt_intensity"}}\n  }}\n}}```\n\nHere\'s a new test question:\n```json\n{{\n  "domain": "{domain}",\n  "domain_description": "{domain_desc}",\n  "variable_mapping": {json.dumps(mapping, indent=2)},\n  "dependent_variable": "{dependent}",\n  "dependent_relationship": "{relationship}"\n}}```\nRespond only with the answer JSON.',
+        },
+    ]
+
+
+def OPENAI_SEMANTICS_GEN_SUMMARY_MESSAGES(dataset):
+    return [
+        {
+            'role': 'system',
+            'content': 'You are a helpful assistant who is not talkative. You only respond with the exact answer to a query without additional conversation.',
+        },
+        {
+            'role': 'user',
+            'content': f'Given the following descriptions of the columns of a dataset, your task is to come up with a natural language overview of the dataset, which should include (1) what the dataset is about, (2) how the data was collected, (3) when the data was collected, and (3) for what purpose the data was collected. Be specific and creative.\n\nExample dataset:\n```json\n{{  \n  "dataset": {{                                                                                                                                                                                       \n    "x6": {{"description": "Ancient artifact significance score", "name": "artifact_significance_score", "is_target": true}},\n    "x1": {{"description": "Distance to ancient city center", "name": "dist_to_ancient_city_ctr"}},\n    "x2": {{"description": "Quantity of discovered relics", "name": "relic_discovery_qty"}},\n    "x3": {{"description": "Years since last archaeological expedition", "name": "years_since_exp"}},\n    "x4": {{"description": "Number of artifacts in excavation site", "name": "artifact_qty"}},\n    "x5": {{"description": "Soil fertility coefficient", "name": "soil_fertility_coef"}},\n    "x7": {{"description": "Distance to ancient burial grounds", "name": "dist_to_burial_grounds"}},\n    "x8": {{"description": "Population estimate of ancient civilization", "name": "ancient_civilization_pop_estimate"}},\n    "x9": {{"description": "Temperature variation in excavation region", "name": "temp_variation"}}\n  }}\n}}```\nExample description:\nThis dataset is about archaeological explorations and findings linked to ancient civilizations. The data was collected in the form of field metrics during various archaeological expeditions during the late mid-20th century. The purpose of the data collection is to evaluate the significance of ancient artifacts discovered during excavations.\n\nHere is a new test dataset.\n{json.dumps(dataset, indent=2)}\nProvide only the description.',
+        },
+    ]
+
+
+def OPENAI_GEN_HYPO_MESSAGES(dataset):
+    return [
+        {
+            'role': 'system',
+            'content': 'You are a helpful assistant who is not talkative. You only respond with the exact answer to a query without additional conversation.',
+        },
+        {
+            'role': 'user',
+            'content': f'Given a dataset with its descriptions and the true functional relationship between its variables, your task is to generate 3 levels of hypotheses for the stated relationship in plain English. The three levels are "broad", "medium" and "narrow". Make sure that the hypotheses sound natural. *Only include concepts for variables that are present in the provided functional relationship.* Give your answer as a JSON.\n\nFor example, an example dataset might be the following:\n```json\n{{\n  "domain": "cybersecurity",\n  "summary": "This dataset is about measuring cybersecurity threats in a system. The data was collected by monitoring various cybersecurity metrics in a network environment. The purpose of the data collection is to assess and predict potential cybersecurity risks and vulnerabilities.",\n  "variables": [\n    {{\n      "description": "Level of cybersecurity threat",\n      "name": "cybersecurity_threat",\n      "is_target": true\n    }},\n    {{\n      "description": "Number of failed login attempts",\n      "name": "failed_login_attempts"\n    }},\n    {{\n      "description": "Amount of encrypted data",\n      "name": "encrypted_data"\n    }},\n    {{\n      "description": "Frequency of software updates",\n      "name": "software_updates"\n    }},\n    {{\n      "description": "Number of antivirus software installed",\n      "name": "antivirus_software"\n    }},\n    {{\n      "description": "Quality of firewall protection",\n      "name": "firewall_quality"\n    }}\n  ],\n  "relationship": {{\n    "dependent": "cybersecurity_threat",\n    "relation": "-53.5*encrypted_data**2 - 53.85*failed_login_attempts**2 + 67.75*firewall_quality - 92.16 - 36.68/software_updates**3"\n  }}\n}}```\nGiven this dataset, the following is a valid answer:\n```json\n{{\n  "broad": {{\n    "instruction": "Be vague. Only indicate which concepts might be related but not how they are related",\n    "hypothesis": "Threat to cybersecurity is influenced by several factors including the amount of encrypted data, the number of failed login attempts, the quality of the firewall, as well as how often the software is updated."\n  }},\n  "medium": {{\n    "instruction": "Be slightly more specific. For each factor, indicate carefully whether it positively or negatively affects the relationship, but do not indicate what the exponent is.",\n    "hypothesis": "Cybersecurity threat tends to decrease with the amount of data encryption, the number of failed login attempts, as well as the frequency of software updates to some extent, while improvement in the firewall quality has a positive effect."\n  }},\n  "narrow": {{\n    "instruction": "Be specific. Communicate the concepts, whether there is a positive or negative effect (be careful), and the meaning of the exponent",\n    "hypothesis": "The threat to cybersecurity interacts in a complex manner with various factors. As the amount of encrypted data increases, there is a quadratic decrease in threat. Similarly for the number of failed login attempts, there is a negative quadratic relationship. The quality of the firewall protection on the other hand demonstrates a positive and linear relationship. Finally, the frequency of software updates has an inverse cubic relationship to the threat."\n  }},\n}}\n```\n\nBased on this, provide an answer for the following test dataset:\n```json\n{dataset}```\nRespond only with a JSON.',
+        },
+    ]
+
+
+def create_prompt(usr_msg):
+    return [
+        {
+            'role': 'system',
+            'content': 'You are a helpful assistant who is not talkative. You only respond with the exact answer to a query without additional conversation.',
+        },
+        {'role': 'user', 'content': usr_msg},
+    ]
+
+
+def get_response(client, prompt, max_retry=5, model='gpt-3.5-turbo', verbose=False):
+    n_try = 0
+    while n_try < max_retry:
+        response = client.chat.completions.create(
+            model=model, messages=create_prompt(prompt), **OPENAI_GEN_HYP
+        )
+
+        # COMMENT: changed from
+        # response.choices[0].message.content.strip().strip('```json').strip('```')
+        content = response.choices[0].message.content
+        cleaned_content = content.split('```json')[1].split('```')[0].strip()
+        output = cleaned_content
+        try:
+            response_json = json.loads(output)
+            return response_json
+        except ValueError:
+            if verbose:
+                print(f'Bad JSON output:\n\n{output}')
+            n_try += 1
+            if n_try < max_retry:
+                if verbose:
+                    print('Retrying...')
+            else:
+                if verbose:
+                    print('Retry limit reached')
+    return None
+
+
+def get_code_fix(
+    client, code, error, max_retry=5, model='gpt-3.5-turbo', verbose=False
+):
+    prompt = f"""\
+Given the following code snippet and error message, provide a single-line fix for the error. \
+Note that the code is going to be executed using python `eval`. \
+The code should be executable and should not produce the error message. Be as specific as possible.
+
+Here's the code and the error:
+{{
+    "code": "{code}",
+    "error": "{error}"
+}}
+
+Return only a JSON object with the fixed code in the following format:
+```json
+{{
+    "fixed_code": "..."
+}}"""
+    response = get_response(
+        client, prompt, max_retry=max_retry, model=model, verbose=verbose
+    )
+    return response
+
+
+def get_new_hypothesis(
+    client, target, old, expr, cols, model='gpt-3.5-turbo', verbose=False
+):
+    prompt = f"""\
+Given a target column from a dataset, a pandas expression to derive the column from existing columns, a list of \
+existing columns, and a previously written hypothesis text, carefully check if the hypothesis text is consistent with \
+the pandas expression or not. If it is consistent, simply return the hypothesis as it is. If it is not consistent, \
+provide a new natural language hypothesis that is consistent with the pandas expression using only the provided \
+information. Be specific.
+
+Here's the information:
+```json
+{{
+    "target_column": "{target}",
+    "pandas_expression": "{expr}",
+    "existing_columns": {json.dumps(cols, indent=4)}
+    "old_hypothesis": "{old}",
+}}```
+
+Give your answer as a new JSON with the following format:
+```json
+{{
+    "hypothesis": "..."
+}}"""
+    response = get_response(client, prompt, model=model, verbose=verbose)
+    return response
+
+
+def replace_variable(client, expr, old, new, model='gpt-3.5-turbo', verbose=False):
+    prompt = f"""\
+Given a pandas "expression", replace mentions of the "old" column with its "new" value such that the resultant \
+expression is equivalent to the original expression.
+
+Here's the information:
+```json
+{{
+    "expression": "{expr}",
+    "old": "{old}",
+    "new": "{new}"
+}}```
+
+Give your answer as a new JSON with the following format:
+```json
+{{
+    "new_expression": "..."
+}}"""
+    response = get_response(client, prompt, model=model, verbose=verbose)
+    return response
diff --git a/evaluation/discoverybench/eval_utils/openai_semantic_gen_prompts.py b/evaluation/discoverybench/eval_utils/openai_semantic_gen_prompts.py
new file mode 100644
index 000000000000..a0b5438e4c8a
--- /dev/null
+++ b/evaluation/discoverybench/eval_utils/openai_semantic_gen_prompts.py
@@ -0,0 +1,151 @@
+common_hypothesis_features = [
+    '1-2 sentences',
+    'surprising finding',
+    'includes numeric concepts',
+    'includes categorical concepts',
+    'includes binary concepts',
+]
+hypothesis_features = [
+    ['requires within-cluster analysis'],
+    ['requires across-cluster analysis'],
+    ['corresponds to a polynomial relationship of some columns'],
+    ['corresponds to a ratio between some columns'],
+    ['requires temporal analysis'],
+    ['relationship is based on descriptive statistics of some columns'],
+    ['requires concepts based on percentage or percentiles'],
+    ['relationship is only applicable to one cluster in the data and not the others'],
+]
+
+column_features = [
+    [
+        'must have one target column',
+        'must have quantifiable columns',
+        'must have a few categorical columns',
+        'make sure the categorical column values do not contain special characters',
+        'include a few distractor columns',
+    ]
+]
+
+common_pandas_features = [
+    'must be executable using python `eval` to create the target column in variable `df` (pandas dataframe)',
+    "for e.g., df['A']**2 + 3*df['B'] + 9, np.where(df['A'] > 3, 'Yes', 'No'), etc.",
+    'variables in pandas_expression must be from the existing columns listed above',
+    'variables in pandas_expression must NOT contain the target column itself',
+]
+pandas_features = [
+    ['expression is a quadratic polynomial'],
+    ['expression is a cubic polynomial'],
+    ['expression is a ratio of existing columns'],
+    ['expression is derived through logical combination of existing columns'],
+    # workflow
+]
+pandas_features = [common_pandas_features + p for p in pandas_features]
+
+common_derived_features = [
+    '1-2 sentences',
+    'includes numeric concepts',
+    'includes categorical concepts',
+    'includes binary concepts',
+]
+derived_features = [common_derived_features + h for h in hypothesis_features]
+hypothesis_features = [common_hypothesis_features + h for h in hypothesis_features]
+
+PROMPT_HYP = """\
+Given a dataset topic and description, generate an interesting hypothesis based on \
+the provided instructions. Be creative and come up with an unusual finding.
+
+```json
+{
+    "topic": "%s",
+    "description": "%s",
+    "hypothesis_features": %s,
+    "hypothesis": "..."
+}```
+
+Give your answer as a new JSON with the following format:
+```json
+{
+    "hypothesis": "..."
+}
+```"""
+
+PROMPT_COL = """\
+Given a dataset topic, its description, and a true hypothesis that can be determined from it, \
+generate a list of valid columns based on the provided instructions.
+
+```json
+{
+    "topic": "%s",
+    "description": "%s",
+    "hypothesis": "%s",
+    "column_instructions": %s,
+    "columns": [
+        {
+            "col_name": "...",  # should be an "_"-separated string
+            "description": "...",
+            "data_type": "...",  # should be executable using python's `eval` function. E.g., str, float, int, bool
+            "data_range": {...},  # should be either {"min": ..., "max": ...} or {"values": [...]}
+            "is_distractor": true/false,  # boolean indicating whether this is a distractor that could cause confusion during data analysis
+            "is_target": true/false  # boolean indicating whether this is the target variable for the hypothesis; at least one column should be the target
+        },
+        ...
+    ],
+    "pandas_instructions": %s,
+    "pandas_equation_for_hypothesis": {
+        "target_col": "...",
+        "target_col_type": "...",
+        "target_col_range": {...},
+        "independent_cols_in_pandas_expression": [],  # list of column names that will be used to derive the target column
+        "pandas_expression": "..."  # expression to derive df[target_col] using df[ind_col1], df[ind_col2], etc.
+    }
+}```
+
+Give your answer as a new JSON with the "columns" and "pandas_equation_for_hypothesis" keys filled using the following format:
+```json
+{
+    "columns": [...],
+    "pandas_equation_for_hypothesis": {...}
+}
+```"""
+
+PROMPT_DER = """\
+Given a dataset topic, description, a true hypothesis that can be determined from the data, \
+and a target column from the dataset, generate a hypothesis for the target column using new independent columns not present in the existing columns.
+
+```json
+{
+    "topic": "%s",
+    "description": "%s",
+    "hypothesis": "%s",
+    "existing_columns": %s,
+    "target_column": "%s",
+    "new_to_target_instructions": %s,
+    "new_to_target_hypothesis": "...",  # describe a relationship between new columns that explains the target column
+    "new_columns_for_target": [  # do not repeat any of the existing columns in the dataset
+        {
+            "col_name": "...",  # should be an "_"-separated string
+            "description": "...",
+            "data_type": "...",  # should be executable using python's `eval` function. E.g., str, float, int, bool
+            "data_range": {...},  # should be either {"min": ..., "max": ...} or {"values": [...]}
+        },
+        ...
+    ],
+    "pandas_instructions": %s,
+    "pandas_equation_for_new_to_target_hypothesis": {
+        "target_col": "...",
+        "target_col_type": "...",
+        "target_col_range": {...},
+        "independent_cols_in_pandas_expression": [],  # list of column names from new_columns_for_target that will be used to derive target_col
+        "pandas_expression": "..."  # expression to derive df[target_col] using df[ind_col1], df[ind_col2], etc.
+    }
+}```
+
+Give your answer as a new JSON with the "new_to_target_hypothesis", "new_columns_for_target", and \
+"pandas_equation_for_new_to_target_hypothesis" keys filled using the following format:
+```json
+{
+    "new_to_target_hypothesis": "...",
+    "new_columns_for_target": [...],
+    "pandas_equation_for_new_to_target_hypothesis": {...}
+}
+```"""
diff --git a/evaluation/discoverybench/eval_utils/response_parser.py b/evaluation/discoverybench/eval_utils/response_parser.py
new file mode 100644
index 000000000000..b5de82b5df9e
--- /dev/null
+++ b/evaluation/discoverybench/eval_utils/response_parser.py
@@ -0,0 +1,52 @@
+workflow_summary_markers = [
+    'WORKFLOW SUMMARY',
+    'WORKFLOW_SUMMARY',
+    'WORKFLOW-SUMMARY',
+    'Workflow Summary',
+]
+
+final_answer_markers = [
+    'FINAL ANSWER',
+    'FINAL_ANSWER',
+    'FINAL-ANSWER',
+    'Final Answer',
+    'Scientific Hypothesis',
+    'Hypothesis',
+]
+
+next_agent_markers = [
+    'NEXT AGENT',
+    'NEXT-AGENT',
+    'NEXT_AGENT',
+    'FEEDBACK',
+]
+
+
+def extract_between(content, start_markers, end_markers=None):
+    for marker in start_markers:
+        if marker in content:
+            result = content.split(marker, 1)[1]
+            if end_markers:
+                for end_marker in end_markers:
+                    if end_marker in result:
+                        result = result.split(end_marker, 1)[0]
+            return result
+    return ''
+
+
+def extract_gen_hypo_from_logs(content: str):
+    error = ''
+
+    gen_workflow = extract_between(
+        content, workflow_summary_markers, final_answer_markers
+    )
+
+    if not gen_workflow:
+        error += 'No Workflow Summary found in the line. | '
+
+    gen_hypothesis = extract_between(content, final_answer_markers, next_agent_markers)
+
+    if not gen_hypothesis:
+        error += 'No Final Answer in the line.'
+
+    return gen_hypothesis, gen_workflow, error
diff --git a/evaluation/discoverybench/run_infer.py b/evaluation/discoverybench/run_infer.py
new file mode 100644
index 000000000000..72148a64e759
--- /dev/null
+++ b/evaluation/discoverybench/run_infer.py
@@ -0,0 +1,492 @@
+import asyncio
+import json
+import os
+
+import git
+import pandas as pd
+
+from evaluation.discoverybench.eval_utils.eval_w_subhypo_gen import (
+    run_eval_gold_vs_gen_NL_hypo_workflow,
+)
+from evaluation.discoverybench.eval_utils.response_parser import (
+    extract_gen_hypo_from_logs,
+)
+from evaluation.utils.shared import (
+    EvalMetadata,
+    EvalOutput,
+    codeact_user_response,
+    compatibility_for_eval_history_pairs,
+    make_metadata,
+    prepare_dataset,
+    reset_logger_for_multiprocessing,
+    run_evaluation,
+)
+from openhands.controller.state.state import State
+from openhands.core.config import (
+    AgentConfig,
+    AppConfig,
+    SandboxConfig,
+    get_llm_config_arg,
+    parse_arguments,
+)
+from openhands.core.logger import openhands_logger as logger
+from openhands.core.main import create_runtime, run_controller
+from openhands.events.action import AgentFinishAction, CmdRunAction, MessageAction
+from openhands.events.observation import CmdOutputObservation
+from openhands.runtime.base import Runtime
+from openhands.utils.async_utils import call_async_from_sync
+
+EVALUATION_LLM = 'gpt-4-1106-preview'
+
+DATA_FILES = {}
+
+LIBRARIES = [
+    'pandas',
+    'numpy',
+    'scipy',
+    'matplotlib',
+    'seaborn',
+    'scikit-learn',
+    'statsmodels',
+]
+
+AGENT_CLS_TO_FAKE_USER_RESPONSE_FN = {
+    'CodeActAgent': codeact_user_response,
+}
+
+AGENT_CLS_TO_INST_SUFFIX = {
+    'CodeActAgent': 'When you think you have fixed the issue through code changes, please run the following command: <execute_bash> exit </execute_bash>.\n'
+}
+
+
+def get_config(
+    metadata: EvalMetadata,
+) -> AppConfig:
+    config = AppConfig(
+        default_agent=metadata.agent_class,
+        run_as_openhands=False,
+        runtime='eventstream',
+        max_iterations=metadata.max_iterations,
+        sandbox=SandboxConfig(
+            base_container_image='python:3.12-bookworm',
+            enable_auto_lint=True,
+            use_host_network=False,
+        ),
+        # do not mount workspace
+        workspace_base=None,
+        workspace_mount_path=None,
+    )
+    config.set_llm_config(metadata.llm_config)
+    agent_config = AgentConfig(
+        function_calling=False,
+        codeact_enable_jupyter=True,
+        codeact_enable_browsing_delegate=True,
+    )
+    config.set_agent_config(agent_config)
+    return config
+
+
+def get_dv_query_for_real(
+    datasets, question, domain_knowledge=None, workflow_tags=None
+):
+    """
+    Prepare a structured query for the agent to execute on the specified datasets.
+
+    This function constructs a query by compiling metadata from the provided datasets, along with any relevant domain knowledge and workflow tags.
+
+    Args:
+        datasets: List of datasets
+        question: Query to be answered
+        domain_knowledge: Domain knowledge if any
+        workflow_tags: Workflow tags if any
+
+    Returns:
+        query_to_dv: Query to be run on the dataset
+        dataset_meta: Metadata of the dataset
+    """
+
+    dataset_meta = ''
+    for dataset_metadata in datasets:
+        dataset_meta += 'Dataset name: ' + dataset_metadata['name']
+        dataset_meta += 'Dataset description: ' + dataset_metadata['description']
+        dataset_meta += '\nBrief description of columns: '
+        for col in dataset_metadata['columns']['raw']:
+            dataset_meta += col['name'] + ': ' + col['description'] + ', '
+
+    query_to_dv = dataset_meta
+
+    query_to_dv += f'\nQuery: {question}'
+
+    if domain_knowledge:
+        query_to_dv += (
+            '\nAdditionally, we provide some hints that might be useful to solve the task. Domain Knowledge: \n'
+            + domain_knowledge
+            + '.\n'
+        )
+
+    if workflow_tags:
+        query_to_dv += 'The meta tags are: ' + workflow_tags + '.\n'
+
+    query_to_dv += (
+        'In the final answer, please write down a scientific hypothesis in '
+        'natural language, derived from the provided dataset, clearly stating the '
+        'context of hypothesis (if any), variables chosen (if any) and '
+        'relationship between those variables (if any) including any statistical significance.'
+        'Also generate a summary of the full workflow starting from data loading that led to the final answer as WORKFLOW SUMMARY:'
+    )
+
+    # Run the NL query through datavoyager
+    return query_to_dv, dataset_meta
+
+
+def initialize_runtime(runtime: Runtime, data_files: list[str]):
+    """
+    Initialize the runtime for the agent.
+
+    This function is called before the runtime is used to run the agent.
+    """
+    logger.info(f"{'-' * 50} BEGIN Runtime Initialization Fn {'-' * 50}")
+    obs: CmdOutputObservation
+
+    action = CmdRunAction(command='mkdir -p /workspace')
+    logger.info(action, extra={'msg_type': 'ACTION'})
+    obs = runtime.run_action(action)
+    assert obs.exit_code == 0
+
+    action = CmdRunAction(command='cd /workspace')
+    logger.info(action, extra={'msg_type': 'ACTION'})
+    obs = runtime.run_action(action)
+    assert obs.exit_code == 0
+
+    for file in data_files:
+        runtime.copy_to(
+            file,
+            '/workspace',
+        )
+
+    for lib in LIBRARIES:
+        action = CmdRunAction(command=f'pip install {lib}')
+        logger.info(action, extra={'msg_type': 'ACTION'})
+        obs = runtime.run_action(action)
+        assert obs.exit_code == 0
+
+    logger.info(f"{'-' * 50} END Runtime Initialization Fn {'-' * 50}")
+
+
+def get_last_agent_finish_action(state: State) -> AgentFinishAction:
+    for event in reversed(state.history):
+        if isinstance(event, AgentFinishAction):
+            return event
+    return None
+
+
+def get_last_message_action(state: State) -> MessageAction:
+    for event in reversed(state.history):
+        if isinstance(event, MessageAction):
+            return event
+    return None
+
+
+def complete_runtime(state: State):
+    last_agent_finish_action = get_last_agent_finish_action(state)
+    last_agent_message_action = get_last_message_action(state)
+
+    if last_agent_finish_action is not None:
+        final_message_1 = last_agent_finish_action.thought
+        gen_hypo_1, gen_workflow_1, error_1 = extract_gen_hypo_from_logs(
+            final_message_1
+        )
+    else:
+        gen_hypo_1, gen_workflow_1, error_1 = '', '', ''
+
+    if last_agent_message_action is not None:
+        final_message_2 = last_agent_message_action.content
+        gen_hypo_2, gen_workflow_2, error_2 = extract_gen_hypo_from_logs(
+            final_message_2
+        )
+    else:
+        gen_hypo_2, gen_workflow_2, error_2 = '', '', ''
+
+    if gen_hypo_1 and gen_hypo_2:
+        test_result = {
+            'gen_hypo': last_agent_finish_action.thought
+            if last_agent_finish_action
+            else last_agent_message_action.content,
+            'gen_workflow': '',
+            'error': '',
+        }
+        return test_result
+
+    test_result = {
+        'gen_hypo': gen_hypo_1 if gen_hypo_1 else gen_hypo_2,
+        'gen_workflow': gen_workflow_1 if gen_workflow_1 else gen_workflow_2,
+        'error': error_1 if error_1 else error_2,
+    }
+
+    return test_result
+
+
+def process_instance(
+    instance: pd.Series,
+    metadata: EvalMetadata,
+    reset_logger: bool = True,
+):
+    """
+    Process and evaluate a single instance of the dataset.
+
+    This function executes the OpenHands agent
+    for a specific instance of the dataset. It retrieves
+    the agent's results and evaluates them against the gold
+    hypothesis.
+
+    Args:
+        instance: A single row of the dataset
+        metadata: Metadata for the evaluation
+        reset_logger: Whether to reset the logger
+
+    Returns:
+        output: EvalOutput object
+    """
+
+    config = get_config(metadata)
+
+    # use a session id for concurrent evaluation
+    sid = 'ID_' + str(instance.instance_id)
+
+    # Setup the logger properly, so you can run
+    # multi-processing to parallelize the evaluation
+    if reset_logger:
+        log_dir = os.path.join(metadata.eval_output_dir, 'infer_logs')
+        reset_logger_for_multiprocessing(logger, instance.instance_id, log_dir)
+    else:
+        logger.info(f'Starting evaluation for instance {instance.instance_id}.')
+
+    problem_statement, dataset_metadata = get_dv_query_for_real(
+        datasets=instance.datasets,
+        question=instance.query,
+        domain_knowledge=instance.domain_knowledge,
+        workflow_tags=instance.workflow_tags,
+    )
+
+    # Prepare instruction
+    instruction = (
+        f'You are a discovery agent who can execute a python code only once to answer a query based on one or more datasets. The datasets will be present in the current directory.\n\n'
+        'Environment has been set up for you to start working. You may assume all necessary tools and datasets are installed.\n\n'
+        '# Problem Statement\n'
+        f'{problem_statement}\n\n'
+    )
+    instruction += (
+        'IMPORTANT: You should ONLY interact with the environment provided to you AND NEVER ASK FOR HUMAN HELP.\n'
+        'You should NOT modify any existing test case files. If needed, you can add new test cases in a NEW file to reproduce the issue.\n'
+        'You SHOULD INCLUDE PROPER INDENTATION in your edit commands.\n'
+    )
+    # NOTE: You can actually set slightly different instruction for different agents
+    instruction += AGENT_CLS_TO_INST_SUFFIX[metadata.agent_class]
+
+    # Here's how you can run the agent (similar to the `main` function) and get the final task state
+    runtime = create_runtime(config, sid=sid)
+    call_async_from_sync(runtime.connect)
+    initialize_runtime(runtime, instance.data_files)
+
+    state: State | None = asyncio.run(
+        run_controller(
+            config=config,
+            initial_user_action=MessageAction(content=instruction),
+            runtime=runtime,
+            fake_user_response_fn=AGENT_CLS_TO_FAKE_USER_RESPONSE_FN.get(
+                metadata.agent_class
+            ),
+        )
+    )
+
+    if state is None:
+        raise ValueError('State should not be None.')
+
+    metrics = state.metrics.get() if state.metrics else None
+    test_result = complete_runtime(state)
+
+    # history is now available as a stream of events, rather than list of pairs of (Action, Observation)
+    # for compatibility with the existing output format, we can remake the pairs here
+    # remove when it becomes unnecessary
+    histories = compatibility_for_eval_history_pairs(state.history)
+
+    # DiscoveryBench Evaluation
+    eval_rec = run_eval_gold_vs_gen_NL_hypo_workflow(
+        query=instance.query,
+        gold_hypo=instance.gold_hypo,
+        gold_workflow='',
+        gen_hypo=test_result['gen_hypo'],
+        gen_workflow='',
+        dataset_meta=instance.dataset_metadata,
+        llm_used=EVALUATION_LLM,
+        dataset_type='real',
+    )
+
+    test_result['eval_rec'] = eval_rec
+
+    output = EvalOutput(
+        instance_id=str(instance.instance_id),
+        instruction=instruction,
+        metadata=metadata,
+        history=histories,
+        metrics=metrics,
+        error=state.last_error if state and state.last_error else None,
+        test_result=test_result,
+    )
+
+    return output
+
+
+def update_csv_name(name):
+    name = name.replace('-', '_')
+
+    if 'meta_regression' in name:
+        name = name.replace('meta_regression', 'meta-regression')
+    if 'ML_enabled' in name:
+        name = name.replace('ML_enabled', 'ML-enabled')
+
+    return name
+
+
+def list_csv_files(list_of_datasets):
+    res = []
+    for ele in list_of_datasets:
+        for key, value in ele.items():
+            if key == 'name':
+                csv_file_name = update_csv_name(value)
+                res.append(DATA_FILES[csv_file_name])
+    return res
+
+
+def create_dataset(repo_location: str, split: str = 'test'):
+    """
+    Create a dataset from the discoverybench repository
+    by walking through the repository and extracting metadata
+    from the metadata_{}.json files
+
+    Args:
+        repo_location: Location of the repository
+        split: Split of the dataset to use
+
+    Returns:
+        df: DataFrame containing the dataset instances
+    """
+
+    data_dict = {}
+
+    data_location = os.path.join(repo_location, 'discoverybench', 'real', split)
+    answer_key_location = os.path.join(repo_location, 'eval', 'answer_key_real.csv')
+
+    idx = 0
+
+    for root, dirs, files in os.walk(data_location):
+        for file in files:
+            if file.endswith('.json'):
+                if 'metadata' in file:
+                    metadata = json.load(open(os.path.join(root, file)))
+
+                    dataset = root.split('/')[-1]
+                    metadata_id = file.split('_')[-1].split('.')[0]
+                    domain = metadata.get('domain', '')
+                    domain_knowledge = metadata.get('domain_knowledge', '')
+                    workflow_tags = metadata.get('workflow_tags', '')
+                    datasets = metadata.get('datasets', [])
+                    queries = metadata.get('queries', [])
+                    gold_workflow = metadata.get('workflow')
+
+                    # loop through queries list to get queries
+                    # and each query has qid; add that to dictionary
+                    for query in queries[0]:
+                        qid = query.get('qid', '')
+
+                        data = {
+                            'dataset': dataset,
+                            'metadata_id': metadata_id,
+                            'qid': qid,
+                            'domain': domain,
+                            'domain_knowledge': domain_knowledge,
+                            'workflow_tags': workflow_tags,
+                            'datasets': datasets,
+                            'question_type': query['question_type'],
+                            'query': query['question'],
+                            'gold_workflow': gold_workflow,
+                            'dataset_metadata': metadata,
+                        }
+
+                        data_dict[idx] = data
+                        idx += 1
+
+            if file.endswith('.csv'):
+                DATA_FILES[file] = os.path.join(root, file)
+            if file.endswith('.txt'):
+                DATA_FILES[file] = os.path.join(root, file)
+
+    df = pd.DataFrame.from_dict(data_dict, orient='index')
+
+    df['instance_id'] = df.index
+
+    df['data_files'] = df['datasets'].apply(lambda x: list_csv_files(x))
+
+    answer_key = pd.read_csv(answer_key_location)
+
+    answer_key = answer_key.rename(
+        columns={
+            'metadataid': 'metadata_id',
+            'query_id': 'qid',
+            'gold_hypothesis': 'gold_hypothesis',
+        }
+    )
+
+    df['qid'] = df['qid'].astype(int)
+    df['metadata_id'] = df['metadata_id'].astype(int)
+
+    answer_key['qid'] = answer_key['qid'].astype(int)
+    answer_key['metadata_id'] = answer_key['metadata_id'].astype(int)
+
+    df = pd.merge(df, answer_key, on=['dataset', 'metadata_id', 'qid'], how='left')
+
+    return df
+
+
+if __name__ == '__main__':
+    args = parse_arguments()
+
+    # clone git repositor for csv files
+    repo_url = 'https://github.com/allenai/discoverybench.git'
+    repo_location = 'git-discoverybench-allenai'
+
+    try:
+        git.Repo.clone_from(repo_url, repo_location)
+    except git.exc.GitCommandError:
+        print('Repository already exists')
+
+    dataset = create_dataset(repo_location)
+
+    # check if there is any empty csv_file
+    if dataset['data_files'].isnull().any():
+        raise ValueError('Some csv files are missing.')
+
+    llm_config = None
+    if args.llm_config:
+        llm_config = get_llm_config_arg(args.llm_config)
+    if llm_config is None:
+        raise ValueError(f'Could not find LLM config: --llm_config {args.llm_config}')
+
+    metadata = make_metadata(
+        llm_config,
+        'discoverybench-python',
+        args.agent_cls,
+        args.max_iterations,
+        args.eval_note,
+        args.eval_output_dir,
+    )
+    output_file = os.path.join(metadata.eval_output_dir, 'output.jsonl')
+    instances = prepare_dataset(dataset, output_file, args.eval_n_limit)
+
+    run_evaluation(
+        instances,
+        metadata,
+        output_file,
+        args.eval_num_workers,
+        process_instance,
+    )
diff --git a/evaluation/discoverybench/scripts/run_infer.sh b/evaluation/discoverybench/scripts/run_infer.sh
new file mode 100755
index 000000000000..8b9fffd7c579
--- /dev/null
+++ b/evaluation/discoverybench/scripts/run_infer.sh
@@ -0,0 +1,46 @@
+#!/bin/bash
+set -eo pipefail
+
+source "evaluation/utils/version_control.sh"
+
+MODEL_CONFIG=$1
+COMMIT_HASH=$2
+AGENT=$3
+EVAL_LIMIT=$4
+NUM_WORKERS=$5
+
+if [ -z "$NUM_WORKERS" ]; then
+  NUM_WORKERS=1
+  echo "Number of workers not specified, use default $NUM_WORKERS"
+fi
+
+# ################################################################################
+
+checkout_eval_branch
+
+if [ -z "$AGENT" ]; then
+  echo "Agent not specified, use default CodeActAgent"
+  AGENT="CodeActAgent"
+fi
+
+get_agent_version
+
+echo "AGENT: $AGENT"
+echo "AGENT_VERSION: $AGENT_VERSION"
+echo "MODEL_CONFIG: $MODEL_CONFIG"
+
+COMMAND="poetry run python evaluation/discoverybench/run_infer.py \
+  --agent-cls $AGENT \
+  --llm-config $MODEL_CONFIG \
+  --max-iterations 10 \
+  --max-chars 10000000 \
+  --eval-num-workers $NUM_WORKERS \
+  --eval-note $AGENT_VERSION"
+
+if [ -n "$EVAL_LIMIT" ]; then
+  echo "EVAL_LIMIT: $EVAL_LIMIT"
+  COMMAND="$COMMAND --eval-n-limit $EVAL_LIMIT"
+fi
+
+# Run the command
+eval $COMMAND
diff --git a/evaluation/gaia/run_infer.py b/evaluation/gaia/run_infer.py
index c02cd0aee737..1fa0c00e6d6a 100644
--- a/evaluation/gaia/run_infer.py
+++ b/evaluation/gaia/run_infer.py
@@ -12,6 +12,7 @@
     EvalMetadata,
     EvalOutput,
     codeact_user_response,
+    compatibility_for_eval_history_pairs,
     make_metadata,
     prepare_dataset,
     reset_logger_for_multiprocessing,
@@ -166,7 +167,7 @@ def process_instance(
 
     model_answer_raw = ''
     # get the last message or thought from the agent
-    for event in state.history.get_events(reverse=True):
+    for event in reversed(state.history):
         if event.source == 'agent':
             if isinstance(event, AgentFinishAction):
                 model_answer_raw = event.thought
@@ -203,7 +204,7 @@ def process_instance(
     # history is now available as a stream of events, rather than list of pairs of (Action, Observation)
     # for compatibility with the existing output format, we can remake the pairs here
     # remove when it becomes unnecessary
-    histories = state.history.compatibility_for_eval_history_pairs()
+    histories = compatibility_for_eval_history_pairs(state.history)
 
     # Save the output
     output = EvalOutput(
diff --git a/evaluation/gorilla/run_infer.py b/evaluation/gorilla/run_infer.py
index 873cb7f89694..e437f2b6075a 100644
--- a/evaluation/gorilla/run_infer.py
+++ b/evaluation/gorilla/run_infer.py
@@ -10,6 +10,7 @@
     EvalMetadata,
     EvalOutput,
     codeact_user_response,
+    compatibility_for_eval_history_pairs,
     make_metadata,
     prepare_dataset,
     reset_logger_for_multiprocessing,
@@ -101,7 +102,7 @@ def process_instance(
         raise ValueError('State should not be None.')
 
     # retrieve the last message from the agent
-    model_answer_raw = state.history.get_last_agent_message()
+    model_answer_raw = state.get_last_agent_message()
 
     # attempt to parse model_answer
     ast_eval_fn = instance['ast_eval']
@@ -114,7 +115,7 @@ def process_instance(
     # history is now available as a stream of events, rather than list of pairs of (Action, Observation)
     # for compatibility with the existing output format, we can remake the pairs here
     # remove when it becomes unnecessary
-    histories = state.history.compatibility_for_eval_history_pairs()
+    histories = compatibility_for_eval_history_pairs(state.history)
 
     output = EvalOutput(
         instance_id=instance_id,
diff --git a/evaluation/gpqa/run_infer.py b/evaluation/gpqa/run_infer.py
index 8fd4034c9d5e..58db2e404fc8 100644
--- a/evaluation/gpqa/run_infer.py
+++ b/evaluation/gpqa/run_infer.py
@@ -28,6 +28,7 @@
 from evaluation.utils.shared import (
     EvalMetadata,
     EvalOutput,
+    compatibility_for_eval_history_pairs,
     make_metadata,
     prepare_dataset,
     reset_logger_for_multiprocessing,
@@ -244,7 +245,7 @@ def process_instance(
         'C': False,
         'D': False,
     }
-    for event in state.history.get_events(reverse=True):
+    for event in reversed(state.history):
         if (
             isinstance(event, AgentFinishAction)
             and event.source != 'user'
@@ -300,7 +301,7 @@ def process_instance(
         instance_id=str(instance.instance_id),
         instruction=instruction,
         metadata=metadata,
-        history=state.history.compatibility_for_eval_history_pairs(),
+        history=compatibility_for_eval_history_pairs(state.history),
         metrics=metrics,
         error=state.last_error if state and state.last_error else None,
         test_result={
diff --git a/evaluation/humanevalfix/run_infer.py b/evaluation/humanevalfix/run_infer.py
index 25fee65561fc..2aa184758b33 100644
--- a/evaluation/humanevalfix/run_infer.py
+++ b/evaluation/humanevalfix/run_infer.py
@@ -21,6 +21,7 @@
     EvalMetadata,
     EvalOutput,
     codeact_user_response,
+    compatibility_for_eval_history_pairs,
     make_metadata,
     prepare_dataset,
     reset_logger_for_multiprocessing,
@@ -255,7 +256,7 @@ def process_instance(
     # history is now available as a stream of events, rather than list of pairs of (Action, Observation)
     # for compatibility with the existing output format, we can remake the pairs here
     # remove when it becomes unnecessary
-    histories = state.history.compatibility_for_eval_history_pairs()
+    histories = compatibility_for_eval_history_pairs(state.history)
 
     # Save the output
     output = EvalOutput(
diff --git a/evaluation/integration_tests/run_infer.py b/evaluation/integration_tests/run_infer.py
index ddc044088bb1..5e3205fefe2e 100644
--- a/evaluation/integration_tests/run_infer.py
+++ b/evaluation/integration_tests/run_infer.py
@@ -13,6 +13,7 @@
     prepare_dataset,
     reset_logger_for_multiprocessing,
     run_evaluation,
+    update_llm_config_for_completions_logging,
 )
 from openhands.controller.state.state import State
 from openhands.core.config import (
@@ -55,18 +56,14 @@ def get_config(
         workspace_base=None,
         workspace_mount_path=None,
     )
-    if metadata.llm_config.log_completions:
-        metadata.llm_config.log_completions_folder = os.path.join(
-            metadata.eval_output_dir, 'llm_completions', instance_id
+    config.set_llm_config(
+        update_llm_config_for_completions_logging(
+            metadata.llm_config, metadata.eval_output_dir, instance_id
         )
-        logger.info(
-            f'Logging LLM completions for instance {instance_id} to '
-            f'{metadata.llm_config.log_completions_folder}'
-        )
-    config.set_llm_config(metadata.llm_config)
+    )
     agent_config = AgentConfig(
         codeact_enable_jupyter=True,
-        codeact_enable_browsing_delegate=True,
+        codeact_enable_browsing=True,
         codeact_enable_llm_editor=False,
     )
     config.set_agent_config(agent_config)
@@ -132,7 +129,7 @@ def process_instance(
     # # result evaluation
     # # =============================================
 
-    histories = [event_to_dict(event) for event in state.history.get_events()]
+    histories = [event_to_dict(event) for event in state.history]
     test_result: TestResult = test_class.verify_result(runtime, histories)
     metrics = state.metrics.get() if state.metrics else None
 
diff --git a/evaluation/integration_tests/tests/t06_github_pr_browsing.py b/evaluation/integration_tests/tests/t06_github_pr_browsing.py
new file mode 100644
index 000000000000..52ec927cd334
--- /dev/null
+++ b/evaluation/integration_tests/tests/t06_github_pr_browsing.py
@@ -0,0 +1,44 @@
+from evaluation.integration_tests.tests.base import BaseIntegrationTest, TestResult
+from openhands.events.action import AgentFinishAction, MessageAction
+from openhands.events.event import Event
+from openhands.events.observation import AgentDelegateObservation
+from openhands.runtime.base import Runtime
+
+
+class Test(BaseIntegrationTest):
+    INSTRUCTION = 'Look at https://github.com/All-Hands-AI/OpenHands/pull/8, and tell me what is happening there and what did @asadm suggest.'
+
+    @classmethod
+    def initialize_runtime(cls, runtime: Runtime) -> None:
+        pass
+
+    @classmethod
+    def verify_result(cls, runtime: Runtime, histories: list[Event]) -> TestResult:
+        # check if the "The answer is OpenHands is all you need!" is in any message
+        message_actions = [
+            event
+            for event in histories
+            if isinstance(
+                event, (MessageAction, AgentFinishAction, AgentDelegateObservation)
+            )
+        ]
+        for event in message_actions:
+            if isinstance(event, AgentDelegateObservation):
+                content = event.content
+            elif isinstance(event, AgentFinishAction):
+                content = event.outputs.get('content', '')
+            elif isinstance(event, MessageAction):
+                content = event.content
+            else:
+                raise ValueError(f'Unknown event type: {type(event)}')
+
+            if (
+                'non-commercial' in content
+                or 'MIT' in content
+                or 'Apache 2.0' in content
+            ):
+                return TestResult(success=True)
+        return TestResult(
+            success=False,
+            reason=f'The answer is not found in any message. Total messages: {len(message_actions)}. Messages: {message_actions}',
+        )
diff --git a/evaluation/logic_reasoning/run_infer.py b/evaluation/logic_reasoning/run_infer.py
index 5b7d35f21130..116b438b3ee9 100644
--- a/evaluation/logic_reasoning/run_infer.py
+++ b/evaluation/logic_reasoning/run_infer.py
@@ -8,6 +8,7 @@
     EvalMetadata,
     EvalOutput,
     codeact_user_response,
+    compatibility_for_eval_history_pairs,
     make_metadata,
     prepare_dataset,
     reset_logger_for_multiprocessing,
@@ -225,7 +226,7 @@ def process_instance(
         raise ValueError('State should not be None.')
 
     final_message = ''
-    for event in state.history.get_events(reverse=True):
+    for event in reversed(state.history):
         if isinstance(event, AgentFinishAction):
             final_message = event.thought
             break
@@ -247,7 +248,7 @@ def process_instance(
     # history is now available as a stream of events, rather than list of pairs of (Action, Observation)
     # for compatibility with the existing output format, we can remake the pairs here
     # remove when it becomes unnecessary
-    histories = state.history.compatibility_for_eval_history_pairs()
+    histories = compatibility_for_eval_history_pairs(state.history)
 
     # Save the output
     output = EvalOutput(
diff --git a/evaluation/miniwob/run_infer.py b/evaluation/miniwob/run_infer.py
index 9c2aaf1e0963..715bdaa470ae 100644
--- a/evaluation/miniwob/run_infer.py
+++ b/evaluation/miniwob/run_infer.py
@@ -10,10 +10,13 @@
 from evaluation.utils.shared import (
     EvalMetadata,
     EvalOutput,
+    codeact_user_response,
+    compatibility_for_eval_history_pairs,
     make_metadata,
     prepare_dataset,
     reset_logger_for_multiprocessing,
     run_evaluation,
+    update_llm_config_for_completions_logging,
 )
 from openhands.controller.state.state import State
 from openhands.core.config import (
@@ -29,7 +32,10 @@
     CmdRunAction,
     MessageAction,
 )
-from openhands.events.observation import CmdOutputObservation
+from openhands.events.observation import (
+    BrowserOutputObservation,
+    CmdOutputObservation,
+)
 from openhands.runtime.base import Runtime
 from openhands.runtime.browser.browser_env import (
     BROWSER_EVAL_GET_GOAL_ACTION,
@@ -37,7 +43,11 @@
 )
 from openhands.utils.async_utils import call_async_from_sync
 
-SUPPORTED_AGENT_CLS = {'BrowsingAgent'}
+SUPPORTED_AGENT_CLS = {'BrowsingAgent', 'CodeActAgent'}
+
+AGENT_CLS_TO_FAKE_USER_RESPONSE_FN = {
+    'CodeActAgent': codeact_user_response,
+}
 
 
 def get_config(
@@ -47,25 +57,32 @@ def get_config(
     config = AppConfig(
         default_agent=metadata.agent_class,
         run_as_openhands=False,
-        runtime='eventstream',
+        runtime=os.environ.get('RUNTIME', 'eventstream'),
         max_iterations=metadata.max_iterations,
         sandbox=SandboxConfig(
             base_container_image='xingyaoww/od-eval-miniwob:v1.0',
             enable_auto_lint=True,
             use_host_network=False,
             browsergym_eval_env=env_id,
+            api_key=os.environ.get('ALLHANDS_API_KEY', None),
+            remote_runtime_api_url=os.environ.get('SANDBOX_REMOTE_RUNTIME_API_URL'),
+            keep_remote_runtime_alive=False,
         ),
         # do not mount workspace
         workspace_base=None,
         workspace_mount_path=None,
     )
-    config.set_llm_config(metadata.llm_config)
+    config.set_llm_config(
+        update_llm_config_for_completions_logging(
+            metadata.llm_config, metadata.eval_output_dir, env_id
+        )
+    )
     return config
 
 
 def initialize_runtime(
     runtime: Runtime,
-) -> str:
+) -> tuple[str, BrowserOutputObservation]:
     """Initialize the runtime for the agent.
 
     This function is called before the runtime is used to run the agent.
@@ -85,8 +102,14 @@ def initialize_runtime(
     logger.info(obs, extra={'msg_type': 'OBSERVATION'})
     goal = obs.content
 
+    # Run noop to get the initial browser observation (e.g., the page URL & content)
+    action = BrowseInteractiveAction(browser_actions='noop(1000)')
+    logger.info(action, extra={'msg_type': 'ACTION'})
+    obs = runtime.run_action(action)
+    logger.info(obs, extra={'msg_type': 'OBSERVATION'})
+
     logger.info(f"{'-' * 50} END Runtime Initialization Fn {'-' * 50}")
-    return goal
+    return goal, obs
 
 
 def complete_runtime(
@@ -117,7 +140,7 @@ def process_instance(
     metadata: EvalMetadata,
     reset_logger: bool = True,
 ) -> EvalOutput:
-    env_id = instance.id
+    env_id = instance.instance_id
     config = get_config(metadata, env_id)
 
     # Setup the logger properly, so you can run multi-processing to parallelize the evaluation
@@ -129,7 +152,12 @@ def process_instance(
 
     runtime = create_runtime(config)
     call_async_from_sync(runtime.connect)
-    task_str = initialize_runtime(runtime)
+    task_str, obs = initialize_runtime(runtime)
+
+    task_str += (
+        f'\nInitial browser state (output of `noop(1000)`):\n{obs.get_agent_obs_text()}'
+    )
+
     state: State | None = asyncio.run(
         run_controller(
             config=config,
@@ -137,6 +165,9 @@ def process_instance(
                 content=task_str
             ),  # take output from initialize_runtime
             runtime=runtime,
+            fake_user_response_fn=AGENT_CLS_TO_FAKE_USER_RESPONSE_FN[
+                metadata.agent_class
+            ],
         )
     )
 
@@ -152,19 +183,19 @@ def process_instance(
 
     # Instruction is the first message from the USER
     instruction = ''
-    for event in state.history.get_events():
+    for event in state.history:
         if isinstance(event, MessageAction):
             instruction = event.content
             break
 
     return_val = complete_runtime(runtime)
     logger.info(f'Return value from complete_runtime: {return_val}')
-    reward = max(return_val['rewards'])
+    reward = max(return_val['rewards'], default=0)
 
     # history is now available as a stream of events, rather than list of pairs of (Action, Observation)
     # for compatibility with the existing output format, we can remake the pairs here
     # remove when it becomes unnecessary
-    histories = state.history.compatibility_for_eval_history_pairs()
+    histories = compatibility_for_eval_history_pairs(state.history)
 
     # Save the output
     output = EvalOutput(
diff --git a/evaluation/mint/run_infer.py b/evaluation/mint/run_infer.py
index 8017b194d8d8..2165c3c03fe4 100644
--- a/evaluation/mint/run_infer.py
+++ b/evaluation/mint/run_infer.py
@@ -13,6 +13,7 @@
 from evaluation.utils.shared import (
     EvalMetadata,
     EvalOutput,
+    compatibility_for_eval_history_pairs,
     make_metadata,
     prepare_dataset,
     reset_logger_for_multiprocessing,
@@ -28,6 +29,7 @@
 from openhands.core.logger import openhands_logger as logger
 from openhands.core.main import create_runtime, run_controller
 from openhands.events.action import (
+    Action,
     CmdRunAction,
     MessageAction,
 )
@@ -45,7 +47,10 @@ def codeact_user_response_mint(state: State, task: Task, task_config: dict[str,
         task=task,
         task_config=task_config,
     )
-    last_action = state.history.get_last_action()
+    last_action = next(
+        (event for event in reversed(state.history) if isinstance(event, Action)),
+        None,
+    )
     result_state: TaskState = env.step(last_action.message or '')
 
     state.extra_data['task_state'] = result_state
@@ -202,7 +207,7 @@ def process_instance(
     # history is now available as a stream of events, rather than list of pairs of (Action, Observation)
     # for compatibility with the existing output format, we can remake the pairs here
     # remove when it becomes unnecessary
-    histories = state.history.compatibility_for_eval_history_pairs()
+    histories = compatibility_for_eval_history_pairs(state.history)
 
     # Save the output
     output = EvalOutput(
diff --git a/evaluation/ml_bench/run_infer.py b/evaluation/ml_bench/run_infer.py
index deec068f3392..2bb667e3c947 100644
--- a/evaluation/ml_bench/run_infer.py
+++ b/evaluation/ml_bench/run_infer.py
@@ -24,6 +24,7 @@
     EvalMetadata,
     EvalOutput,
     codeact_user_response,
+    compatibility_for_eval_history_pairs,
     make_metadata,
     prepare_dataset,
     reset_logger_for_multiprocessing,
@@ -256,7 +257,7 @@ def process_instance(instance: Any, metadata: EvalMetadata, reset_logger: bool =
     # history is now available as a stream of events, rather than list of pairs of (Action, Observation)
     # for compatibility with the existing output format, we can remake the pairs here
     # remove when it becomes unnecessary
-    histories = state.history.compatibility_for_eval_history_pairs()
+    histories = compatibility_for_eval_history_pairs(state.history)
 
     # Save the output
     output = EvalOutput(
diff --git a/evaluation/scienceagentbench/Dockerfile b/evaluation/scienceagentbench/Dockerfile
new file mode 100644
index 000000000000..70ed92cc4dc8
--- /dev/null
+++ b/evaluation/scienceagentbench/Dockerfile
@@ -0,0 +1,17 @@
+FROM python:3.11-bookworm
+
+
+# For OpenHands agents to explore the dataset directories, please download the full benchmark [here](https://buckeyemailosu-my.sharepoint.com/:u:/g/personal/chen_8336_buckeyemail_osu_edu/EQuA6uJ3CtRHvRfZ2GiN1tYBRVJE4DSUD10MW61fr7HuSQ?e=sCBegG) and unzip it with password `scienceagentbench`.
+# **Please DO NOT redistribute the unzipped data files online.**
+# It will download a benchmark.zip file to the current directory.
+# unzip it and put the benchmark folder under evaluation/scienceagentbench/
+
+RUN mkdir -p /benchmark
+COPY benchmark /benchmark
+
+RUN mkdir -p /workspace
+WORKDIR /workspace
+
+# pushd evaluation/scienceagentbench
+# docker build -t xingyaoww/openhands-eval-scienceagentbench .
+# popd
diff --git a/evaluation/scienceagentbench/Dockerfile.evaluator b/evaluation/scienceagentbench/Dockerfile.evaluator
new file mode 100644
index 000000000000..f8263e1bb0aa
--- /dev/null
+++ b/evaluation/scienceagentbench/Dockerfile.evaluator
@@ -0,0 +1,25 @@
+FROM mambaorg/micromamba:debian12
+
+USER root
+# For https://github.com/OSU-NLP-Group/ScienceAgentBench/tree/main?tab=readme-ov-file#code-generation-with-agents
+
+RUN micromamba create -n sci-agent-eval python=3.10 pip setuptools wheel
+RUN micromamba run -n sci-agent-eval pip install pip-tools
+
+RUN mkdir -p /workspace
+WORKDIR /workspace
+
+RUN apt-get update && apt-get install -y git
+
+RUN git clone https://github.com/OSU-NLP-Group/ScienceAgentBench.git /workspace/
+RUN git checkout 4eddc7db6449a5ade3e37285747c8b208cd54ce7
+
+RUN micromamba create -n sci-agent python=3.10 pip setuptools wheel
+RUN micromamba run -n sci-agent pip install -r requirements.txt
+
+# Replace all occurence of conda with micromamba under the /workspace
+RUN find ./ -type f -exec sed -i 's/conda/micromamba/g' {} \;
+
+# pushd evaluation/scienceagentbench
+# docker build -t xingyaoww/openhands-eval-scienceagentbench-evaluator -f Dockerfile.evaluator .
+# popd
diff --git a/evaluation/scienceagentbench/README.md b/evaluation/scienceagentbench/README.md
new file mode 100644
index 000000000000..3182c2e117be
--- /dev/null
+++ b/evaluation/scienceagentbench/README.md
@@ -0,0 +1,54 @@
+# ScienceAgentBench Evaluation with OpenHands
+
+This folder contains the evaluation harness for [ScienceAgentBench](https://osu-nlp-group.github.io/ScienceAgentBench/) (paper: https://arxiv.org/abs/2410.05080).
+
+## Setup Environment and LLM Configuration
+
+Please follow instruction [here](../README.md#setup) to setup your local development environment and LLM.
+
+## Setup ScienceAgentBench
+
+To prevent benchmark data contamination, we only provide the annotation sheet on [Huggingface](https://huggingface.co/datasets/osunlp/ScienceAgentBench), which includes all necessary *inputs* to run an agent.
+
+## Run Inference on ScienceAgentBench
+
+```bash
+./evaluation/scienceagentbench/scripts/run_infer.sh [model_config] [git-version] [use_knowledge] [agent] [eval_limit] [max_iter] [num_workers] [dataset] [dataset_split]
+
+# Example
+./evaluation/scienceagentbench/scripts/run_infer.sh llm.eval_gpt4o 0.9.3
+```
+
+where `model_config` is mandatory, and the rest are optional.
+
+- `model_config`, e.g. `eval_gpt4_1106_preview`, is the config group name for your
+LLM settings, as defined in your `config.toml`.
+- `git-version`, e.g. `HEAD`, is the git commit hash of the OpenHands version you would
+like to evaluate. It could also be a release tag like `0.6.2`.
+- `use_knowledge`, e.g. `true`, specifies whether allowing the agent to use expert-provided knowledge as additional input or not. By default, it is set to `false`.
+- `agent`, e.g. `CodeActAgent`, is the name of the agent for benchmarks, defaulting
+to `CodeActAgent`.
+- `eval_limit`, e.g. `10`, limits the evaluation to the first `eval_limit` instances. By
+default, the script evaluates the entire SWE-bench_Lite test set (300 issues). Note:
+in order to use `eval_limit`, you must also set `agent`.
+- `max_iter`, e.g. `20`, is the maximum number of iterations for the agent to run. By
+default, it is set to 30.
+- `num_workers`, e.g. `3`, is the number of parallel workers to run the evaluation. By
+default, it is set to 1.
+
+## Evaluate Generated Programs
+
+### Extract Necessary Information from OpenHands Log
+
+After the inference is completed, you may use the following command to extract necessary information from the output log for evaluation:
+
+```bash
+python post_proc.py [log_fname]
+```
+- `log_fname`, e.g. `evaluation/.../output.jsonl`, is the automatically saved trajectory log of an OpenHands agent.
+
+Output will be write to e.g. `evaluation/.../output.converted.jsonl`
+
+### Run evaluation
+
+Please follow the steps [here](https://github.com/OSU-NLP-Group/ScienceAgentBench/tree/main?tab=readme-ov-file#evaluation-of-generated-code) to evaluate the generated programs.
diff --git a/evaluation/scienceagentbench/post_proc.py b/evaluation/scienceagentbench/post_proc.py
new file mode 100644
index 000000000000..46cfbe2b2a7a
--- /dev/null
+++ b/evaluation/scienceagentbench/post_proc.py
@@ -0,0 +1,30 @@
+import json
+from argparse import ArgumentParser
+
+if __name__ == '__main__':
+    parser = ArgumentParser()
+    parser.add_argument(
+        'log_fname',
+        type=str,
+    )
+    args = parser.parse_args()
+
+    fname = args.log_fname
+    out_fname = args.log_fname.replace('.jsonl', '.converted.jsonl')
+
+    log = [json.loads(line) for line in open(fname)]
+
+    simple_log = [
+        json.dumps(
+            {
+                'instance_id': ex['instance_id'],
+                'instruction': ex['instruction'],
+                'test_result': ex['test_result'],
+                'cost': ex['metrics']['accumulated_cost'],
+            }
+        )
+        for ex in log
+    ]
+
+    with open(out_fname, 'w+', encoding='utf-8') as f:
+        f.write('\n'.join(simple_log))
diff --git a/evaluation/scienceagentbench/run_infer.py b/evaluation/scienceagentbench/run_infer.py
new file mode 100644
index 000000000000..93a82855452e
--- /dev/null
+++ b/evaluation/scienceagentbench/run_infer.py
@@ -0,0 +1,292 @@
+import asyncio
+import os
+from typing import Any
+
+import pandas as pd
+from datasets import load_dataset
+from tqdm import tqdm
+
+from evaluation.utils.shared import (
+    EvalMetadata,
+    EvalOutput,
+    codeact_user_response,
+    compatibility_for_eval_history_pairs,
+    make_metadata,
+    prepare_dataset,
+    reset_logger_for_multiprocessing,
+    run_evaluation,
+    update_llm_config_for_completions_logging,
+)
+from openhands.controller.state.state import State
+from openhands.core.config import (
+    AppConfig,
+    SandboxConfig,
+    get_llm_config_arg,
+    get_parser,
+)
+from openhands.core.logger import openhands_logger as logger
+from openhands.core.main import create_runtime, run_controller
+from openhands.events.action import CmdRunAction, MessageAction
+from openhands.events.observation import CmdOutputObservation
+from openhands.runtime.base import Runtime
+from openhands.utils.async_utils import call_async_from_sync
+
+AGENT_CLS_TO_FAKE_USER_RESPONSE_FN = {
+    'CodeActAgent': codeact_user_response,
+}
+
+LOCAL_DATASET_PATH = os.path.join(os.path.dirname(__file__), 'benchmark')
+
+
+def format_task_dict(example, use_knowledge):
+    task = {
+        'instance_id': example['instance_id'],
+        'task_inst': example['task_inst'],
+        'dataset_path': '/benchmark/datasets/'
+        + example['dataset_folder_tree'].split('\n')[0][4:],
+        'dataset_folder_tree': example['dataset_folder_tree'],
+        'dataset_preview': example['dataset_preview'],
+        'pred_program_name': 'pred_' + example['gold_program_name'],
+    }
+
+    if use_knowledge:
+        task['task_inst'] += '\n' + str(example['domain_knowledge'])
+
+    return task
+
+
+def get_config(
+    metadata: EvalMetadata,
+    instance_id: str,
+) -> AppConfig:
+    config = AppConfig(
+        default_agent=metadata.agent_class,
+        run_as_openhands=False,
+        runtime=os.environ.get('RUNTIME', 'eventstream'),
+        max_budget_per_task=4,
+        max_iterations=metadata.max_iterations,
+        sandbox=SandboxConfig(
+            base_container_image='docker.io/xingyaoww/openhands-eval-scienceagentbench',
+            enable_auto_lint=True,
+            use_host_network=False,
+            timeout=300,
+            api_key=os.environ.get('ALLHANDS_API_KEY', None),
+            remote_runtime_api_url=os.environ.get('SANDBOX_REMOTE_RUNTIME_API_URL'),
+            keep_remote_runtime_alive=False,
+        ),
+        # do not mount workspace
+        workspace_base=None,
+        workspace_mount_path=None,
+    )
+    config.set_llm_config(
+        update_llm_config_for_completions_logging(
+            metadata.llm_config,
+            metadata.eval_output_dir,
+            instance_id,
+        )
+    )
+    return config
+
+
+def initialize_runtime(
+    runtime: Runtime,
+    instance: pd.Series,  # this argument is not required
+):
+    """Initialize the runtime for the agent.
+
+    This function is called before the runtime is used to run the agent.
+    """
+    logger.info(f"{'-' * 50} BEGIN Runtime Initialization Fn {'-' * 50}")
+    obs: CmdOutputObservation
+
+    # Set up workspace directories
+    action = CmdRunAction(command='mkdir -p /workspace/pred_programs')
+    logger.info(action, extra={'msg_type': 'ACTION'})
+    obs = runtime.run_action(action)
+    assert obs.exit_code == 0
+
+    action = CmdRunAction(command='mkdir -p /workspace/pred_results')
+    logger.info(action, extra={'msg_type': 'ACTION'})
+    obs = runtime.run_action(action)
+    assert obs.exit_code == 0
+
+    dataset_name = instance['dataset_folder_tree'].split('\n')[0][4:].rstrip('/')
+
+    # Copy the dataset to the workspace
+    dataset_dir = os.path.join(
+        LOCAL_DATASET_PATH,
+        'datasets',
+        dataset_name,
+    )
+    runtime.copy_to(dataset_dir, '/workspace/benchmark/datasets', recursive=True)
+
+    # Check the dataset exists
+    action = CmdRunAction(
+        command='cd /workspace/benchmark/datasets && ls',
+        keep_prompt=False,
+    )
+    obs = runtime.run_action(action)
+    logger.info(obs, extra={'msg_type': 'OBSERVATION'})
+    assert obs.exit_code == 0
+    assert dataset_name in obs.content
+
+    logger.info(f"{'-' * 50} END Runtime Initialization Fn {'-' * 50}")
+
+
+def complete_runtime(
+    runtime: Runtime,
+    instance: pd.Series,
+) -> dict[str, Any]:
+    """Complete the runtime for the agent.
+
+    This function is called before the runtime is used to run the agent.
+    If you need to do something in the sandbox to get the correctness metric after
+    the agent has run, modify this function.
+    """
+    logger.info(f"{'-' * 50} BEGIN Runtime Completion Fn {'-' * 50}")
+    obs: CmdOutputObservation
+
+    test_result = {}
+
+    action = CmdRunAction(command='cd /workspace')
+    logger.info(action, extra={'msg_type': 'ACTION'})
+    obs = runtime.run_action(action)
+
+    assert obs.exit_code == 0
+
+    action = CmdRunAction(
+        command=f'cat pred_programs/{instance.pred_program_name}',
+        keep_prompt=False,
+    )
+    logger.info(action, extra={'msg_type': 'ACTION'})
+    obs = runtime.run_action(action)
+
+    if obs.exit_code == 0:
+        test_result = {'program': obs.content}
+    else:
+        test_result = {'program': 'ERROR'}
+
+    logger.info(f"{'-' * 50} END Runtime Completion Fn {'-' * 50}")
+    return test_result
+
+
+def process_instance(
+    instance: pd.Series,
+    metadata: EvalMetadata,
+    reset_logger: bool = True,
+) -> EvalOutput:
+    instance_id = instance.instance_id.replace('/', '__')
+    config = get_config(metadata, instance_id)
+
+    # Set up the logger properly, so you can run multi-processing to parallelize the evaluation
+    if reset_logger:
+        log_dir = os.path.join(metadata.eval_output_dir, 'infer_logs')
+        reset_logger_for_multiprocessing(logger, instance_id, log_dir)
+    else:
+        logger.info(f'Starting evaluation for instance {instance_id}.')
+
+    instruction = f"""You are an expert Python programming assistant that helps scientist users to write high-quality code to solve their tasks.
+Given a user request, you are expected to write a complete program that accomplishes the requested task and save any outputs to `/workspace/pred_results/` in the correct format.
+
+Here's the user request you need to work on:
+{instance.task_inst}
+
+You can access the dataset at `{instance.dataset_path}`. Here is the directory structure of the dataset:
+```
+{instance.dataset_folder_tree}
+```
+Here are some helpful previews for the dataset file(s):
+{instance.dataset_preview}
+
+Please save your program as `/workspace/pred_programs/{instance.pred_program_name}`.
+Then, please run the program to check and fix any errors.
+Please do NOT run the program in the background.
+If the program uses some packages that are incompatible, please figure out alternative implementations and do NOT restart the environment.
+
+"""
+
+    runtime = create_runtime(config)
+    call_async_from_sync(runtime.connect)
+    initialize_runtime(runtime, instance)
+
+    # Here's how you can run the agent (similar to the `main` function) and get the final task state
+    state: State | None = asyncio.run(
+        run_controller(
+            config=config,
+            initial_user_action=MessageAction(content=instruction),
+            runtime=runtime,
+            fake_user_response_fn=AGENT_CLS_TO_FAKE_USER_RESPONSE_FN.get(
+                metadata.agent_class
+            ),
+        )
+    )
+
+    # ======= Attempt to evaluate the agent's edits =======
+    test_result = complete_runtime(runtime, instance)
+
+    # If you are working on some simpler benchmark that only evaluates the final model output (e.g., in a MessageAction)
+    # You can simply get the LAST `MessageAction` from the returned `state.history` and parse it for evaluation.
+    if state is None:
+        raise ValueError('State should not be None.')
+    metrics = state.metrics.get() if state.metrics else None
+
+    # history is now available as a stream of events, rather than list of pairs of (Action, Observation)
+    # for compatibility with the existing output format, we can remake the pairs here
+    # remove when it becomes unnecessary
+    histories = compatibility_for_eval_history_pairs(state.history)
+
+    # Save the output
+    output = EvalOutput(
+        instance_id=instance.instance_id,
+        instruction=instruction,
+        metadata=metadata,
+        history=histories,
+        metrics=metrics,
+        error=state.last_error if state and state.last_error else None,
+        test_result=test_result,
+    )
+    return output
+
+
+if __name__ == '__main__':
+    parser = get_parser()
+    parser.add_argument(
+        '--use_knowledge',
+        type=str,
+        default='false',
+        choices=['true', 'false'],
+        help='use expert-provided knowledge or not',
+    )
+    args, _ = parser.parse_known_args()
+
+    sab_dataset = load_dataset('osunlp/ScienceAgentBench', split='validation')
+
+    dataset_processed = []
+    for example in tqdm(sab_dataset):
+        dataset_processed.append(
+            format_task_dict(example, args.use_knowledge == 'true')
+        )
+
+    dataset = pd.DataFrame(dataset_processed)
+
+    llm_config = None
+    if args.llm_config:
+        llm_config = get_llm_config_arg(args.llm_config)
+    if llm_config is None:
+        raise ValueError(f'Could not find LLM config: --llm_config {args.llm_config}')
+
+    metadata = make_metadata(
+        llm_config,
+        'ScienceAgentBench',
+        args.agent_cls,
+        args.max_iterations,
+        args.eval_note,
+        args.eval_output_dir,
+    )
+    output_file = os.path.join(metadata.eval_output_dir, 'output.jsonl')
+    dataset['instance_id'] = dataset['instance_id'].apply(str)
+    instances = prepare_dataset(dataset, output_file, args.eval_n_limit)
+
+    run_evaluation(
+        instances, metadata, output_file, args.eval_num_workers, process_instance
+    )
diff --git a/evaluation/scienceagentbench/scripts/run_infer.sh b/evaluation/scienceagentbench/scripts/run_infer.sh
new file mode 100755
index 000000000000..7667e5723789
--- /dev/null
+++ b/evaluation/scienceagentbench/scripts/run_infer.sh
@@ -0,0 +1,49 @@
+#!/bin/bash
+set -eo pipefail
+
+source "evaluation/utils/version_control.sh"
+
+MODEL_CONFIG=$1
+COMMIT_HASH=$2
+USE_KNOWLEDGE=$3
+AGENT=$4
+EVAL_LIMIT=$5
+NUM_WORKERS=$6
+
+if [ -z "$NUM_WORKERS" ]; then
+  NUM_WORKERS=1
+  echo "Number of workers not specified, use default $NUM_WORKERS"
+fi
+checkout_eval_branch
+
+if [ -z "$AGENT" ]; then
+  echo "Agent not specified, use default CodeActAgent"
+  AGENT="CodeActAgent"
+fi
+
+if [ -z "$USE_KNOWLEDGE" ]; then
+  echo "Use knowledge not specified, use default False"
+  USE_KNOWLEDGE=false
+fi
+
+get_agent_version
+
+echo "AGENT: $AGENT"
+echo "AGENT_VERSION: $AGENT_VERSION"
+echo "MODEL_CONFIG: $MODEL_CONFIG"
+
+COMMAND="poetry run python evaluation/scienceagentbench/run_infer.py \
+  --agent-cls $AGENT \
+  --llm-config $MODEL_CONFIG \
+  --use_knowledge $USE_KNOWLEDGE \
+  --max-iterations 30 \
+  --eval-num-workers $NUM_WORKERS \
+  --eval-note $AGENT_VERSION" \
+
+if [ -n "$EVAL_LIMIT" ]; then
+  echo "EVAL_LIMIT: $EVAL_LIMIT"
+  COMMAND="$COMMAND --eval-n-limit $EVAL_LIMIT"
+fi
+
+# Run the command
+eval $COMMAND
diff --git a/evaluation/swe_bench/run_infer.py b/evaluation/swe_bench/run_infer.py
index 7578e7e9562b..2cc0dfd7d9a6 100644
--- a/evaluation/swe_bench/run_infer.py
+++ b/evaluation/swe_bench/run_infer.py
@@ -20,6 +20,7 @@
     prepare_dataset,
     reset_logger_for_multiprocessing,
     run_evaluation,
+    update_llm_config_for_completions_logging,
 )
 from openhands.controller.state.state import State
 from openhands.core.config import (
@@ -40,6 +41,7 @@
 
 USE_HINT_TEXT = os.environ.get('USE_HINT_TEXT', 'false').lower() == 'true'
 USE_INSTANCE_IMAGE = os.environ.get('USE_INSTANCE_IMAGE', 'false').lower() == 'true'
+RUN_WITH_BROWSING = os.environ.get('RUN_WITH_BROWSING', 'false').lower() == 'true'
 
 AGENT_CLS_TO_FAKE_USER_RESPONSE_FN = {
     'CodeActAgent': codeact_user_response,
@@ -89,6 +91,13 @@ def get_instruction(instance: pd.Series, metadata: EvalMetadata):
             '5. Think about edgecases and make sure your fix handles them as well\n'
             "Your thinking should be thorough and so it's fine if it's very long.\n"
         )
+
+    if RUN_WITH_BROWSING:
+        instruction += (
+            '<IMPORTANT!>\n'
+            'You SHOULD NEVER attempt to browse the web. '
+            '</IMPORTANT!>\n'
+        )
     return instruction
 
 
@@ -143,18 +152,14 @@ def get_config(
         workspace_base=None,
         workspace_mount_path=None,
     )
-    if metadata.llm_config.log_completions:
-        metadata.llm_config.log_completions_folder = os.path.join(
-            metadata.eval_output_dir, 'llm_completions', instance['instance_id']
+    config.set_llm_config(
+        update_llm_config_for_completions_logging(
+            metadata.llm_config, metadata.eval_output_dir, instance['instance_id']
         )
-        logger.info(
-            f'Logging LLM completions for instance {instance["instance_id"]} to '
-            f'{metadata.llm_config.log_completions_folder}'
-        )
-    config.set_llm_config(metadata.llm_config)
+    )
     agent_config = AgentConfig(
         codeact_enable_jupyter=False,
-        codeact_enable_browsing_delegate=False,
+        codeact_enable_browsing=RUN_WITH_BROWSING,
         codeact_enable_llm_editor=False,
     )
     config.set_agent_config(agent_config)
@@ -439,7 +444,8 @@ def process_instance(
     if state is None:
         raise ValueError('State should not be None.')
 
-    histories = [event_to_dict(event) for event in state.history.get_events()]
+    # NOTE: this is NO LONGER the event stream, but an agent history that includes delegate agent's events
+    histories = [event_to_dict(event) for event in state.history]
     metrics = state.metrics.get() if state.metrics else None
 
     # Save the output
diff --git a/evaluation/swe_bench/scripts/run_infer.sh b/evaluation/swe_bench/scripts/run_infer.sh
index 54bcbbbc3391..520003635a4e 100755
--- a/evaluation/swe_bench/scripts/run_infer.sh
+++ b/evaluation/swe_bench/scripts/run_infer.sh
@@ -34,6 +34,11 @@ if [ -z "$USE_INSTANCE_IMAGE" ]; then
   USE_INSTANCE_IMAGE=true
 fi
 
+if [ -z "$RUN_WITH_BROWSING" ]; then
+  echo "RUN_WITH_BROWSING not specified, use default false"
+  RUN_WITH_BROWSING=false
+fi
+
 
 if [ -z "$DATASET" ]; then
   echo "DATASET not specified, use default princeton-nlp/SWE-bench_Lite"
@@ -47,6 +52,8 @@ fi
 
 export USE_INSTANCE_IMAGE=$USE_INSTANCE_IMAGE
 echo "USE_INSTANCE_IMAGE: $USE_INSTANCE_IMAGE"
+export RUN_WITH_BROWSING=$RUN_WITH_BROWSING
+echo "RUN_WITH_BROWSING: $RUN_WITH_BROWSING"
 
 get_agent_version
 
@@ -67,6 +74,10 @@ if [ "$USE_HINT_TEXT" = false ]; then
   EVAL_NOTE="$EVAL_NOTE-no-hint"
 fi
 
+if [ "$RUN_WITH_BROWSING" = true ]; then
+  EVAL_NOTE="$EVAL_NOTE-with-browsing"
+fi
+
 if [ -n "$EXP_NAME" ]; then
   EVAL_NOTE="$EVAL_NOTE-$EXP_NAME"
 fi
diff --git a/evaluation/toolqa/run_infer.py b/evaluation/toolqa/run_infer.py
index 5c2c53422785..25633ce6ce23 100644
--- a/evaluation/toolqa/run_infer.py
+++ b/evaluation/toolqa/run_infer.py
@@ -9,6 +9,7 @@
     EvalMetadata,
     EvalOutput,
     codeact_user_response,
+    compatibility_for_eval_history_pairs,
     make_metadata,
     prepare_dataset,
     reset_logger_for_multiprocessing,
@@ -126,7 +127,7 @@ def process_instance(instance: Any, metadata: EvalMetadata, reset_logger: bool =
         raise ValueError('State should not be None.')
 
     # retrieve the last message from the agent
-    model_answer_raw = state.history.get_last_agent_message()
+    model_answer_raw = state.get_last_agent_message()
 
     # attempt to parse model_answer
     correct = eval_answer(str(model_answer_raw), str(answer))
@@ -137,7 +138,7 @@ def process_instance(instance: Any, metadata: EvalMetadata, reset_logger: bool =
     # history is now available as a stream of events, rather than list of pairs of (Action, Observation)
     # for compatibility with the existing output format, we can remake the pairs here
     # remove when it becomes unnecessary
-    histories = state.history.compatibility_for_eval_history_pairs()
+    histories = compatibility_for_eval_history_pairs(state.history)
 
     # Save the output
     output = EvalOutput(
diff --git a/evaluation/utils/shared.py b/evaluation/utils/shared.py
index b8d2ad281ad6..d5a6d6d89de8 100644
--- a/evaluation/utils/shared.py
+++ b/evaluation/utils/shared.py
@@ -18,6 +18,9 @@
 from openhands.core.logger import openhands_logger as logger
 from openhands.events.action import Action
 from openhands.events.action.message import MessageAction
+from openhands.events.event import Event
+from openhands.events.serialization.event import event_to_dict
+from openhands.events.utils import get_pairs_from_events
 
 
 class EvalMetadata(BaseModel):
@@ -112,7 +115,14 @@ def codeact_user_response(
     if state.history:
         # check if the last action has an answer, if so, early exit
         if try_parse is not None:
-            last_action = state.history.get_last_action()
+            last_action = next(
+                (
+                    event
+                    for event in reversed(state.history)
+                    if isinstance(event, Action)
+                ),
+                None,
+            )
             ans = try_parse(last_action)
             if ans is not None:
                 return '/exit'
@@ -120,7 +130,7 @@ def codeact_user_response(
         # check if the agent has tried to talk to the user 3 times, if so, let the agent know it can give up
         user_msgs = [
             event
-            for event in state.history.get_events()
+            for event in state.history
             if isinstance(event, MessageAction) and event.source == 'user'
         ]
         if len(user_msgs) >= 2:
@@ -411,3 +421,35 @@ def reset_logger_for_multiprocessing(
     )
     file_handler.setLevel(logging.INFO)
     logger.addHandler(file_handler)
+
+
+def update_llm_config_for_completions_logging(
+    llm_config: LLMConfig,
+    eval_output_dir: str,
+    instance_id: str,
+) -> LLMConfig:
+    """Update the LLM config for logging completions."""
+    if llm_config.log_completions:
+        llm_config.log_completions_folder = os.path.join(
+            eval_output_dir, 'llm_completions', instance_id
+        )
+        logger.info(
+            f'Logging LLM completions for instance {instance_id} to '
+            f'{llm_config.log_completions_folder}'
+        )
+    return llm_config
+
+
+# history is now available as a filtered stream of events, rather than list of pairs of (Action, Observation)
+# we rebuild the pairs here
+# for compatibility with the existing output format in evaluations
+# remove this when it's no longer necessary
+def compatibility_for_eval_history_pairs(
+    history: list[Event],
+) -> list[tuple[dict, dict]]:
+    history_pairs = []
+
+    for action, observation in get_pairs_from_events(history):
+        history_pairs.append((event_to_dict(action), event_to_dict(observation)))
+
+    return history_pairs
diff --git a/evaluation/webarena/run_infer.py b/evaluation/webarena/run_infer.py
index cfc2bdae493a..531f134fd988 100644
--- a/evaluation/webarena/run_infer.py
+++ b/evaluation/webarena/run_infer.py
@@ -10,6 +10,7 @@
 from evaluation.utils.shared import (
     EvalMetadata,
     EvalOutput,
+    compatibility_for_eval_history_pairs,
     make_metadata,
     prepare_dataset,
     reset_logger_for_multiprocessing,
@@ -166,7 +167,7 @@ def process_instance(
 
     # Instruction is the first message from the USER
     instruction = ''
-    for event in state.history.get_events():
+    for event in state.history:
         if isinstance(event, MessageAction):
             instruction = event.content
             break
@@ -178,7 +179,7 @@ def process_instance(
     # history is now available as a stream of events, rather than list of pairs of (Action, Observation)
     # for compatibility with the existing output format, we can remake the pairs here
     # remove when it becomes unnecessary
-    histories = state.history.compatibility_for_eval_history_pairs()
+    histories = compatibility_for_eval_history_pairs(state.history)
 
     # Save the output
     output = EvalOutput(
diff --git a/frontend/.eslintrc b/frontend/.eslintrc
index c0b7a8c9be2c..d5cb543bd728 100644
--- a/frontend/.eslintrc
+++ b/frontend/.eslintrc
@@ -84,4 +84,4 @@
       }
     }
   ]
-}
\ No newline at end of file
+}
diff --git a/frontend/.gitignore b/frontend/.gitignore
index 44029f58cb4f..d62bd04c1a2f 100644
--- a/frontend/.gitignore
+++ b/frontend/.gitignore
@@ -1,4 +1,9 @@
 # i18n translation files make by script using `make build`
 public/locales/**/*
 src/i18n/declaration.ts
-.env
\ No newline at end of file
+.env
+node_modules/
+/test-results/
+/playwright-report/
+/blob-report/
+/playwright/.cache/
diff --git a/frontend/__tests__/components/chat/chat-interface.test.tsx b/frontend/__tests__/components/chat/chat-interface.test.tsx
index d27897c2d8e3..501389f9897a 100644
--- a/frontend/__tests__/components/chat/chat-interface.test.tsx
+++ b/frontend/__tests__/components/chat/chat-interface.test.tsx
@@ -128,14 +128,14 @@ describe.skip("ChatInterface", () => {
         timestamp: new Date().toISOString(),
       },
       {
-        error: "Woops!",
+        error: true,
+        id: "",
         message: "Something went wrong",
       },
     ];
     renderChatInterface(messages);
 
     const error = screen.getByTestId("error-message");
-    expect(within(error).getByText("Woops!")).toBeInTheDocument();
     expect(within(error).getByText("Something went wrong")).toBeInTheDocument();
   });
 
diff --git a/frontend/__tests__/components/file-explorer/FileExplorer.test.tsx b/frontend/__tests__/components/file-explorer/FileExplorer.test.tsx
index b1faa3c18bf4..a1c0717783e9 100644
--- a/frontend/__tests__/components/file-explorer/FileExplorer.test.tsx
+++ b/frontend/__tests__/components/file-explorer/FileExplorer.test.tsx
@@ -16,13 +16,16 @@ vi.mock("../../services/fileService", async () => ({
 }));
 
 const renderFileExplorerWithRunningAgentState = () =>
-  renderWithProviders(<FileExplorer error={null} />, {
-    preloadedState: {
-      agent: {
-        curAgentState: AgentState.RUNNING,
+  renderWithProviders(
+    <FileExplorer error={null} isOpen onToggle={() => {}} />,
+    {
+      preloadedState: {
+        agent: {
+          curAgentState: AgentState.RUNNING,
+        },
       },
     },
-  });
+  );
 
 describe.skip("FileExplorer", () => {
   afterEach(() => {
diff --git a/frontend/__tests__/utils/extractModelAndProvider.test.ts b/frontend/__tests__/utils/extractModelAndProvider.test.ts
index 6ea84db241db..c1ec4ee838ec 100644
--- a/frontend/__tests__/utils/extractModelAndProvider.test.ts
+++ b/frontend/__tests__/utils/extractModelAndProvider.test.ts
@@ -78,4 +78,3 @@ describe("extractModelAndProvider", () => {
     });
   });
 });
-
diff --git a/frontend/__tests__/utils/organizeModelsAndProviders.test.ts b/frontend/__tests__/utils/organizeModelsAndProviders.test.ts
index 1062309dbf68..aa3c84707432 100644
--- a/frontend/__tests__/utils/organizeModelsAndProviders.test.ts
+++ b/frontend/__tests__/utils/organizeModelsAndProviders.test.ts
@@ -63,4 +63,3 @@ test("organizeModelsAndProviders", () => {
     },
   });
 });
-
diff --git a/frontend/package-lock.json b/frontend/package-lock.json
index 5a1f634dfb44..e7cce105b9a1 100644
--- a/frontend/package-lock.json
+++ b/frontend/package-lock.json
@@ -1,12 +1,12 @@
 {
   "name": "openhands-frontend",
-  "version": "0.12.0",
+  "version": "0.12.3",
   "lockfileVersion": 3,
   "requires": true,
   "packages": {
     "": {
       "name": "openhands-frontend",
-      "version": "0.12.0",
+      "version": "0.12.3",
       "dependencies": {
         "@monaco-editor/react": "^4.6.0",
         "@nextui-org/react": "^2.4.8",
@@ -26,6 +26,7 @@
         "isbot": "^5.1.17",
         "jose": "^5.9.4",
         "monaco-editor": "^0.52.0",
+        "posthog-js": "^1.176.0",
         "react": "^18.3.1",
         "react-dom": "^18.3.1",
         "react-highlight": "^0.15.0",
@@ -45,6 +46,7 @@
         "ws": "^8.18.0"
       },
       "devDependencies": {
+        "@playwright/test": "^1.48.2",
         "@remix-run/dev": "^2.11.2",
         "@remix-run/testing": "^2.11.2",
         "@tailwindcss/typography": "^0.5.15",
@@ -3378,6 +3380,21 @@
         "url": "https://opencollective.com/unts"
       }
     },
+    "node_modules/@playwright/test": {
+      "version": "1.48.2",
+      "resolved": "https://registry.npmjs.org/@playwright/test/-/test-1.48.2.tgz",
+      "integrity": "sha512-54w1xCWfXuax7dz4W2M9uw0gDyh+ti/0K/MxcCUxChFh37kkdxPdfZDw5QBbuPUJHr1CiHJ1hXgSs+GgeQc5Zw==",
+      "dev": true,
+      "dependencies": {
+        "playwright": "1.48.2"
+      },
+      "bin": {
+        "playwright": "cli.js"
+      },
+      "engines": {
+        "node": ">=18"
+      }
+    },
     "node_modules/@polka/url": {
       "version": "1.0.0-next.28",
       "resolved": "https://registry.npmjs.org/@polka/url/-/url-1.0.0-next.28.tgz",
@@ -7864,6 +7881,16 @@
         "node": ">=6.6.0"
       }
     },
+    "node_modules/core-js": {
+      "version": "3.38.1",
+      "resolved": "https://registry.npmjs.org/core-js/-/core-js-3.38.1.tgz",
+      "integrity": "sha512-OP35aUorbU3Zvlx7pjsFdu1rGNnD4pgw/CWoYzRY3t2EzoVT7shKHY1dlAy3f41cGIO7ZDPQimhGFTlEYkG/Hw==",
+      "hasInstallScript": true,
+      "funding": {
+        "type": "opencollective",
+        "url": "https://opencollective.com/core-js"
+      }
+    },
     "node_modules/core-util-is": {
       "version": "1.0.3",
       "resolved": "https://registry.npmjs.org/core-util-is/-/core-util-is-1.0.3.tgz",
@@ -9666,6 +9693,11 @@
         "url": "https://github.com/sponsors/wooorm"
       }
     },
+    "node_modules/fflate": {
+      "version": "0.4.8",
+      "resolved": "https://registry.npmjs.org/fflate/-/fflate-0.4.8.tgz",
+      "integrity": "sha512-FJqqoDBR00Mdj9ppamLa/Y7vxm+PRmNWA67N846RvsoYVMKB4q3y/de5PA7gUmRMYK/8CMz2GDZQmCRN1wBcWA=="
+    },
     "node_modules/file-entry-cache": {
       "version": "6.0.1",
       "resolved": "https://registry.npmjs.org/file-entry-cache/-/file-entry-cache-6.0.1.tgz",
@@ -19406,6 +19438,50 @@
         "pathe": "^1.1.2"
       }
     },
+    "node_modules/playwright": {
+      "version": "1.48.2",
+      "resolved": "https://registry.npmjs.org/playwright/-/playwright-1.48.2.tgz",
+      "integrity": "sha512-NjYvYgp4BPmiwfe31j4gHLa3J7bD2WiBz8Lk2RoSsmX38SVIARZ18VYjxLjAcDsAhA+F4iSEXTSGgjua0rrlgQ==",
+      "dev": true,
+      "dependencies": {
+        "playwright-core": "1.48.2"
+      },
+      "bin": {
+        "playwright": "cli.js"
+      },
+      "engines": {
+        "node": ">=18"
+      },
+      "optionalDependencies": {
+        "fsevents": "2.3.2"
+      }
+    },
+    "node_modules/playwright-core": {
+      "version": "1.48.2",
+      "resolved": "https://registry.npmjs.org/playwright-core/-/playwright-core-1.48.2.tgz",
+      "integrity": "sha512-sjjw+qrLFlriJo64du+EK0kJgZzoQPsabGF4lBvsid+3CNIZIYLgnMj9V6JY5VhM2Peh20DJWIVpVljLLnlawA==",
+      "dev": true,
+      "bin": {
+        "playwright-core": "cli.js"
+      },
+      "engines": {
+        "node": ">=18"
+      }
+    },
+    "node_modules/playwright/node_modules/fsevents": {
+      "version": "2.3.2",
+      "resolved": "https://registry.npmjs.org/fsevents/-/fsevents-2.3.2.tgz",
+      "integrity": "sha512-xiqMQR4xAeHTuB9uWm+fFRcIOgKBMiOBP+eXiyT7jsgVCq1bkVygt00oASowB7EdtpOHaaPgKt812P9ab+DDKA==",
+      "dev": true,
+      "hasInstallScript": true,
+      "optional": true,
+      "os": [
+        "darwin"
+      ],
+      "engines": {
+        "node": "^8.16.0 || ^10.6.0 || >=11.0.0"
+      }
+    },
     "node_modules/possible-typed-array-names": {
       "version": "1.0.0",
       "resolved": "https://registry.npmjs.org/possible-typed-array-names/-/possible-typed-array-names-1.0.0.tgz",
@@ -19653,6 +19729,31 @@
       "resolved": "https://registry.npmjs.org/postcss-value-parser/-/postcss-value-parser-4.2.0.tgz",
       "integrity": "sha512-1NNCs6uurfkVbeXG4S8JFT9t19m45ICnif8zWLd5oPSZ50QnwMfK+H3jv408d4jw/7Bttv5axS5IiHoLaVNHeQ=="
     },
+    "node_modules/posthog-js": {
+      "version": "1.176.0",
+      "resolved": "https://registry.npmjs.org/posthog-js/-/posthog-js-1.176.0.tgz",
+      "integrity": "sha512-T5XKNtRzp7q6CGb7Vc7wAI76rWap9fiuDUPxPsyPBPDkreKya91x9RIsSapAVFafwD1AEin1QMczCmt9Le9BWw==",
+      "dependencies": {
+        "core-js": "^3.38.1",
+        "fflate": "^0.4.8",
+        "preact": "^10.19.3",
+        "web-vitals": "^4.2.0"
+      }
+    },
+    "node_modules/posthog-js/node_modules/web-vitals": {
+      "version": "4.2.4",
+      "resolved": "https://registry.npmjs.org/web-vitals/-/web-vitals-4.2.4.tgz",
+      "integrity": "sha512-r4DIlprAGwJ7YM11VZp4R884m0Vmgr6EAKe3P+kO0PPj3Unqyvv59rczf6UiGcb9Z8QxZVcqKNwv/g0WNdWwsw=="
+    },
+    "node_modules/preact": {
+      "version": "10.24.3",
+      "resolved": "https://registry.npmjs.org/preact/-/preact-10.24.3.tgz",
+      "integrity": "sha512-Z2dPnBnMUfyQfSQ+GBdsGa16hz35YmLmtTLhM169uW944hYL6xzTYkJjC07j+Wosz733pMWx0fgON3JNw1jJQA==",
+      "funding": {
+        "type": "opencollective",
+        "url": "https://opencollective.com/preact"
+      }
+    },
     "node_modules/prelude-ls": {
       "version": "1.2.1",
       "resolved": "https://registry.npmjs.org/prelude-ls/-/prelude-ls-1.2.1.tgz",
diff --git a/frontend/package.json b/frontend/package.json
index 819c024f0429..3825ad4ae251 100644
--- a/frontend/package.json
+++ b/frontend/package.json
@@ -1,6 +1,6 @@
 {
   "name": "openhands-frontend",
-  "version": "0.12.0",
+  "version": "0.12.3",
   "private": true,
   "type": "module",
   "engines": {
@@ -25,6 +25,7 @@
     "isbot": "^5.1.17",
     "jose": "^5.9.4",
     "monaco-editor": "^0.52.0",
+    "posthog-js": "^1.176.0",
     "react": "^18.3.1",
     "react-dom": "^18.3.1",
     "react-highlight": "^0.15.0",
@@ -49,6 +50,7 @@
     "build": "npm run make-i18n && tsc && remix vite:build",
     "start": "npx sirv-cli build/ --single",
     "test": "vitest run",
+    "test:e2e": "playwright test",
     "test:coverage": "npm run make-i18n && vitest run --coverage",
     "dev_wsl": "VITE_WATCH_USE_POLLING=true vite",
     "preview": "vite preview",
@@ -70,6 +72,7 @@
     ]
   },
   "devDependencies": {
+    "@playwright/test": "^1.48.2",
     "@remix-run/dev": "^2.11.2",
     "@remix-run/testing": "^2.11.2",
     "@tailwindcss/typography": "^0.5.15",
diff --git a/frontend/playwright.config.ts b/frontend/playwright.config.ts
new file mode 100644
index 000000000000..53a48004433d
--- /dev/null
+++ b/frontend/playwright.config.ts
@@ -0,0 +1,79 @@
+import { defineConfig, devices } from "@playwright/test";
+
+/**
+ * Read environment variables from file.
+ * https://github.com/motdotla/dotenv
+ */
+// import dotenv from 'dotenv';
+// import path from 'path';
+// dotenv.config({ path: path.resolve(__dirname, '.env') });
+
+/**
+ * See https://playwright.dev/docs/test-configuration.
+ */
+export default defineConfig({
+  testDir: "./tests",
+  /* Run tests in files in parallel */
+  fullyParallel: true,
+  /* Fail the build on CI if you accidentally left test.only in the source code. */
+  forbidOnly: !!process.env.CI,
+  /* Retry on CI only */
+  retries: process.env.CI ? 2 : 0,
+  /* Opt out of parallel tests on CI. */
+  workers: process.env.CI ? 1 : undefined,
+  /* Reporter to use. See https://playwright.dev/docs/test-reporters */
+  reporter: "html",
+  /* Shared settings for all the projects below. See https://playwright.dev/docs/api/class-testoptions. */
+  use: {
+    /* Base URL to use in actions like `await page.goto('/')`. */
+    baseURL: "http://127.0.0.1:3000",
+
+    /* Collect trace when retrying the failed test. See https://playwright.dev/docs/trace-viewer */
+    trace: "on-first-retry",
+  },
+
+  /* Configure projects for major browsers */
+  projects: [
+    {
+      name: "chromium",
+      use: { ...devices["Desktop Chrome"] },
+    },
+
+    {
+      name: "firefox",
+      use: { ...devices["Desktop Firefox"] },
+    },
+
+    {
+      name: "webkit",
+      use: { ...devices["Desktop Safari"] },
+    },
+
+    /* Test against mobile viewports. */
+    // {
+    //   name: 'Mobile Chrome',
+    //   use: { ...devices['Pixel 5'] },
+    // },
+    // {
+    //   name: 'Mobile Safari',
+    //   use: { ...devices['iPhone 12'] },
+    // },
+
+    /* Test against branded browsers. */
+    // {
+    //   name: 'Microsoft Edge',
+    //   use: { ...devices['Desktop Edge'], channel: 'msedge' },
+    // },
+    // {
+    //   name: 'Google Chrome',
+    //   use: { ...devices['Desktop Chrome'], channel: 'chrome' },
+    // },
+  ],
+
+  /* Run your local dev server before starting the tests */
+  webServer: {
+    command: "npm run dev:mock -- --port 3000",
+    url: "http://127.0.0.1:3000",
+    reuseExistingServer: !process.env.CI,
+  },
+});
diff --git a/frontend/src/api/open-hands.ts b/frontend/src/api/open-hands.ts
index 0ef84c0278c2..6981848c7b4d 100644
--- a/frontend/src/api/open-hands.ts
+++ b/frontend/src/api/open-hands.ts
@@ -1,4 +1,4 @@
-import { getValidFallbackHost } from "#/utils/get-valid-fallback-host";
+import { request } from "#/services/api";
 import {
   SaveFileSuccessResponse,
   FileUploadSuccessResponse,
@@ -9,36 +9,13 @@ import {
   GetConfigResponse,
 } from "./open-hands.types";
 
-/**
- * Generate the base URL of the OpenHands API
- * @returns Base URL of the OpenHands API
- */
-const generateBaseURL = () => {
-  const fallback = getValidFallbackHost();
-  const baseUrl = import.meta.env.VITE_BACKEND_BASE_URL || fallback;
-
-  if (typeof window === "undefined") {
-    return `http://${baseUrl}`;
-  }
-  return `${window.location.protocol}//${baseUrl}`;
-};
-
-/**
- * Class to interact with the OpenHands API
- */
 class OpenHands {
-  /**
-   * Base URL of the OpenHands API
-   */
-  static BASE_URL = generateBaseURL();
-
   /**
    * Retrieve the list of models available
    * @returns List of models available
    */
   static async getModels(): Promise<string[]> {
-    const response = await fetch(`${OpenHands.BASE_URL}/api/options/models`);
-    return response.json();
+    return request("/api/options/models");
   }
 
   /**
@@ -46,8 +23,7 @@ class OpenHands {
    * @returns List of agents available
    */
   static async getAgents(): Promise<string[]> {
-    const response = await fetch(`${OpenHands.BASE_URL}/api/options/agents`);
-    return response.json();
+    return request(`/api/options/agents`);
   }
 
   /**
@@ -55,178 +31,123 @@ class OpenHands {
    * @returns List of security analyzers available
    */
   static async getSecurityAnalyzers(): Promise<string[]> {
-    const response = await fetch(
-      `${OpenHands.BASE_URL}/api/options/security-analyzers`,
-    );
-    return response.json();
+    return request(`/api/options/security-analyzers`);
   }
 
   static async getConfig(): Promise<GetConfigResponse> {
-    const response = await fetch("config.json", {
-      headers: {
-        "Cache-Control": "no-cache",
-      },
-    });
-    return response.json();
+    return request("/config.json");
   }
 
   /**
    * Retrieve the list of files available in the workspace
-   * @param token User token provided by the server
    * @param path Path to list files from
    * @returns List of files available in the given path. If path is not provided, it lists all the files in the workspace
    */
-  static async getFiles(token: string, path?: string): Promise<string[]> {
-    const url = new URL(`${OpenHands.BASE_URL}/api/list-files`);
-    if (path) url.searchParams.append("path", path);
-
-    const response = await fetch(url.toString(), {
-      headers: {
-        Authorization: `Bearer ${token}`,
-      },
-    });
-
-    return response.json();
+  static async getFiles(path?: string): Promise<string[]> {
+    let url = "/api/list-files";
+    if (path) url += `?path=${encodeURIComponent(path)}`;
+    return request(url);
   }
 
   /**
    * Retrieve the content of a file
-   * @param token User token provided by the server
    * @param path Full path of the file to retrieve
    * @returns Content of the file
    */
-  static async getFile(token: string, path: string): Promise<string> {
-    const url = new URL(`${OpenHands.BASE_URL}/api/select-file`);
-    url.searchParams.append("file", path);
-    const response = await fetch(url.toString(), {
-      headers: {
-        Authorization: `Bearer ${token}`,
-      },
-    });
-
-    const data = await response.json();
+  static async getFile(path: string): Promise<string> {
+    const url = `/api/select-file?file=${encodeURIComponent(path)}`;
+    const data = await request(url);
     return data.code;
   }
 
   /**
    * Save the content of a file
-   * @param token User token provided by the server
    * @param path Full path of the file to save
    * @param content Content to save in the file
    * @returns Success message or error message
    */
   static async saveFile(
-    token: string,
     path: string,
     content: string,
   ): Promise<SaveFileSuccessResponse | ErrorResponse> {
-    const response = await fetch(`${OpenHands.BASE_URL}/api/save-file`, {
+    return request(`/api/save-file`, {
       method: "POST",
       body: JSON.stringify({ filePath: path, content }),
       headers: {
-        Authorization: `Bearer ${token}`,
         "Content-Type": "application/json",
       },
     });
-
-    return response.json();
   }
 
   /**
    * Upload a file to the workspace
-   * @param token User token provided by the server
    * @param file File to upload
    * @returns Success message or error message
    */
   static async uploadFiles(
-    token: string,
     file: File[],
   ): Promise<FileUploadSuccessResponse | ErrorResponse> {
     const formData = new FormData();
     file.forEach((f) => formData.append("files", f));
 
-    const response = await fetch(`${OpenHands.BASE_URL}/api/upload-files`, {
+    return request(`/api/upload-files`, {
       method: "POST",
       body: formData,
-      headers: {
-        Authorization: `Bearer ${token}`,
-      },
     });
-
-    return response.json();
   }
 
   /**
    * Get the blob of the workspace zip
-   * @param token User token provided by the server
    * @returns Blob of the workspace zip
    */
-  static async getWorkspaceZip(token: string): Promise<Blob> {
-    const response = await fetch(`${OpenHands.BASE_URL}/api/zip-directory`, {
-      headers: {
-        Authorization: `Bearer ${token}`,
-      },
-    });
-
+  static async getWorkspaceZip(): Promise<Blob> {
+    const response = await request(`/api/zip-directory`, {}, false, true);
     return response.blob();
   }
 
   /**
    * Send feedback to the server
-   * @param token User token provided by the server
    * @param data Feedback data
    * @returns The stored feedback data
    */
-  static async sendFeedback(
-    token: string,
-    data: Feedback,
-  ): Promise<FeedbackResponse> {
-    const response = await fetch(`${OpenHands.BASE_URL}/api/submit-feedback`, {
+  static async submitFeedback(data: Feedback): Promise<FeedbackResponse> {
+    return request(`/api/submit-feedback`, {
       method: "POST",
       body: JSON.stringify(data),
       headers: {
-        Authorization: `Bearer ${token}`,
         "Content-Type": "application/json",
       },
     });
-
-    return response.json();
   }
 
   /**
-   * Get the GitHub access token
    * @param code Code provided by GitHub
    * @returns GitHub access token
    */
   static async getGitHubAccessToken(
     code: string,
   ): Promise<GitHubAccessTokenResponse> {
-    const response = await fetch(`${OpenHands.BASE_URL}/api/github/callback`, {
+    return request(`/api/github/callback`, {
       method: "POST",
       body: JSON.stringify({ code }),
       headers: {
         "Content-Type": "application/json",
       },
     });
-
-    return response.json();
   }
 
   /**
-   * Check if the user is authenticated
-   * @param login The user's GitHub login handle
-   * @returns Whether the user is authenticated
+   * Authenticate with GitHub token
+   * @returns Response with authentication status and user info if successful
    */
-  static async isAuthenticated(login: string): Promise<boolean> {
-    const response = await fetch(`${OpenHands.BASE_URL}/api/authenticate`, {
-      method: "POST",
-      body: JSON.stringify({ login }),
-      headers: {
-        "Content-Type": "application/json",
+  static async authenticate(): Promise<Response> {
+    return request(
+      `/api/authenticate`,
+      {
+        method: "POST",
       },
-    });
-
-    return response.status === 200;
+      true,
+    );
   }
 }
 
diff --git a/frontend/src/api/open-hands.types.ts b/frontend/src/api/open-hands.types.ts
index 9da1a339b4d2..a562267363d4 100644
--- a/frontend/src/api/open-hands.types.ts
+++ b/frontend/src/api/open-hands.types.ts
@@ -27,6 +27,11 @@ export interface GitHubAccessTokenResponse {
   access_token: string;
 }
 
+export interface AuthenticationResponse {
+  message: string;
+  login?: string; // Only present when allow list is enabled
+}
+
 export interface Feedback {
   version: string;
   email: string;
diff --git a/frontend/src/assets/arrow-send.svg b/frontend/src/assets/arrow-send.svg
index b353657406d0..a42795073c91 100644
--- a/frontend/src/assets/arrow-send.svg
+++ b/frontend/src/assets/arrow-send.svg
@@ -2,4 +2,4 @@
   <path fill-rule="evenodd" clip-rule="evenodd"
     d="M11.5304 6.46978L10.4697 7.53044L6.75006 3.81077L6.75006 17.0001H5.25006L5.25006 3.81077L1.53039 7.53044L0.469727 6.46978L6.00006 0.939453L11.5304 6.46978Z"
     fill="white" />
-</svg>
\ No newline at end of file
+</svg>
diff --git a/frontend/src/assets/branding/all-hands-logo-spark.svg b/frontend/src/assets/branding/all-hands-logo-spark.svg
index 439dff9778cb..bb4070944af8 100644
--- a/frontend/src/assets/branding/all-hands-logo-spark.svg
+++ b/frontend/src/assets/branding/all-hands-logo-spark.svg
@@ -32,4 +32,4 @@
       <rect width="69" height="46" fill="white" transform="translate(0.5)" />
     </clipPath>
   </defs>
-</svg>
\ No newline at end of file
+</svg>
diff --git a/frontend/src/assets/branding/github-logo.svg b/frontend/src/assets/branding/github-logo.svg
index 975e5fa3ca12..fcf918efacfe 100644
--- a/frontend/src/assets/branding/github-logo.svg
+++ b/frontend/src/assets/branding/github-logo.svg
@@ -2,4 +2,4 @@
   <path
     d="M15.359 21V17.319C15.3974 16.8654 15.3314 16.4095 15.1651 15.9814C14.9989 15.5534 14.7363 15.1631 14.3949 14.8364C17.6154 14.5035 21 13.3716 21 8.17826C20.9997 6.85027 20.4489 5.57321 19.4615 4.61139C19.9291 3.44954 19.896 2.16532 19.3692 1.02548C19.3692 1.02548 18.159 0.692576 15.359 2.43321C13.0082 1.84237 10.5302 1.84237 8.17949 2.43321C5.37949 0.692576 4.16923 1.02548 4.16923 1.02548C3.64244 2.16532 3.60938 3.44954 4.07692 4.61139C3.08218 5.58034 2.53079 6.86895 2.53846 8.2068C2.53846 13.3621 5.92308 14.494 9.14359 14.865C8.80615 15.1883 8.54591 15.574 8.3798 15.9968C8.2137 16.4196 8.14544 16.8701 8.17949 17.319V21M8.17949 18.1465C3.05128 19.5732 3.05128 15.7686 1 15.293L8.17949 18.1465Z"
     stroke="white" stroke-linecap="round" stroke-linejoin="round" />
-</svg>
\ No newline at end of file
+</svg>
diff --git a/frontend/src/assets/clip.svg b/frontend/src/assets/clip.svg
index aaebcbc4dfbe..26a6acb485a9 100644
--- a/frontend/src/assets/clip.svg
+++ b/frontend/src/assets/clip.svg
@@ -2,4 +2,4 @@
   <path fill-rule="evenodd" clip-rule="evenodd"
     d="M10.5 3.75C9.25736 3.75 8.25 4.75736 8.25 6V16.5C8.25 18.5711 9.92893 20.25 12 20.25C14.0711 20.25 15.75 18.5711 15.75 16.5V7H17.25V16.5C17.25 19.3995 14.8995 21.75 12 21.75C9.1005 21.75 6.75 19.3995 6.75 16.5V6C6.75 3.92893 8.42893 2.25 10.5 2.25C12.5711 2.25 14.25 3.92893 14.25 6V16C14.25 17.2426 13.2426 18.25 12 18.25C10.7574 18.25 9.75 17.2426 9.75 16V7H11.25V16C11.25 16.4142 11.5858 16.75 12 16.75C12.4142 16.75 12.75 16.4142 12.75 16V6C12.75 4.75736 11.7426 3.75 10.5 3.75Z"
     fill="white" />
-</svg>
\ No newline at end of file
+</svg>
diff --git a/frontend/src/assets/clipboard.svg b/frontend/src/assets/clipboard.svg
index 6da359d3806c..abf4e5a3155a 100644
--- a/frontend/src/assets/clipboard.svg
+++ b/frontend/src/assets/clipboard.svg
@@ -32,4 +32,4 @@
       <rect width="54" height="75" fill="white" />
     </clipPath>
   </defs>
-</svg>
\ No newline at end of file
+</svg>
diff --git a/frontend/src/assets/default-user.svg b/frontend/src/assets/default-user.svg
index b67e64cf786a..620ab2f3f9c6 100644
--- a/frontend/src/assets/default-user.svg
+++ b/frontend/src/assets/default-user.svg
@@ -2,4 +2,4 @@
   <path
     d="M13.0919 10.5917C13.9089 9.94891 14.5052 9.06746 14.7979 8.06997C15.0906 7.07249 15.0652 6.00858 14.7251 5.02625C14.385 4.04391 13.7471 3.19202 12.9003 2.58907C12.0535 1.98612 11.0398 1.66211 10.0002 1.66211C8.9607 1.66211 7.947 1.98612 7.10018 2.58907C6.25336 3.19202 5.61553 4.04391 5.27542 5.02625C4.93531 6.00858 4.90984 7.07249 5.20254 8.06997C5.49525 9.06746 6.09158 9.94891 6.90858 10.5917C5.50864 11.1526 4.28715 12.0828 3.37432 13.2833C2.46149 14.4838 1.89154 15.9094 1.72524 17.4084C1.7132 17.5178 1.72284 17.6285 1.7536 17.7342C1.78435 17.8399 1.83563 17.9386 1.9045 18.0245C2.04359 18.1979 2.24589 18.309 2.46691 18.3334C2.68792 18.3577 2.90954 18.2932 3.08301 18.1541C3.25648 18.015 3.3676 17.8127 3.39191 17.5917C3.5749 15.9627 4.35165 14.4582 5.57376 13.3657C6.79587 12.2732 8.37766 11.6692 10.0169 11.6692C11.6562 11.6692 13.2379 12.2732 14.4601 13.3657C15.6822 14.4582 16.4589 15.9627 16.6419 17.5917C16.6646 17.7965 16.7623 17.9856 16.9162 18.1225C17.0701 18.2595 17.2692 18.3346 17.4752 18.3334H17.5669C17.7854 18.3082 17.985 18.1978 18.1224 18.0261C18.2597 17.8544 18.3237 17.6353 18.3002 17.4167C18.1332 15.9135 17.5601 14.4842 16.6426 13.2819C15.7251 12.0795 14.4977 11.1496 13.0919 10.5917ZM10.0002 10C9.34097 10 8.69651 9.80453 8.14834 9.43825C7.60018 9.07198 7.17294 8.55139 6.92064 7.9423C6.66835 7.33321 6.60234 6.66299 6.73096 6.01639C6.85957 5.36979 7.17704 4.77584 7.64322 4.30967C8.10939 3.84349 8.70334 3.52602 9.34994 3.39741C9.99654 3.26879 10.6668 3.3348 11.2759 3.58709C11.8849 3.83938 12.4055 4.26662 12.7718 4.81479C13.1381 5.36295 13.3336 6.00742 13.3336 6.66669C13.3336 7.55074 12.9824 8.39859 12.3573 9.02371C11.7321 9.64883 10.8843 10 10.0002 10Z"
     fill="#262626" />
-</svg>
\ No newline at end of file
+</svg>
diff --git a/frontend/src/assets/docs.svg b/frontend/src/assets/docs.svg
index 23475a3da1c5..33eb3a374c61 100644
--- a/frontend/src/assets/docs.svg
+++ b/frontend/src/assets/docs.svg
@@ -24,4 +24,4 @@
       <rect width="23.33" height="18.67" fill="white" transform="translate(2.33496 4.66504)" />
     </clipPath>
   </defs>
-</svg>
\ No newline at end of file
+</svg>
diff --git a/frontend/src/assets/external-link.svg b/frontend/src/assets/external-link.svg
index 96294735b2bf..e55411671193 100644
--- a/frontend/src/assets/external-link.svg
+++ b/frontend/src/assets/external-link.svg
@@ -4,4 +4,4 @@
   <path
     d="M4.33333 3.66667C3.59695 3.66667 3 4.26362 3 5V11.6667C3 12.403 3.59695 13 4.33333 13H11C11.7364 13 12.3333 12.403 12.3333 11.6667V9.66667H11.3333V11.6667C11.3333 11.8508 11.1841 12 11 12H4.33333C4.14924 12 4 11.8508 4 11.6667V5C4 4.81591 4.14924 4.66667 4.33333 4.66667H6.33333V3.66667H4.33333Z"
     fill="#EEEEEE" />
-</svg>
\ No newline at end of file
+</svg>
diff --git a/frontend/src/assets/lightbulb.svg b/frontend/src/assets/lightbulb.svg
index 9f61d275efc1..aa703e60abbc 100644
--- a/frontend/src/assets/lightbulb.svg
+++ b/frontend/src/assets/lightbulb.svg
@@ -2,4 +2,4 @@
   <path
     d="M11.3931 1.88021C10.78 1.37593 10.062 1.01489 9.29157 0.823438C8.52113 0.631981 7.71767 0.614934 6.9398 0.773542C5.90405 0.982756 4.95379 1.49492 4.20959 2.24506C3.46538 2.9952 2.96078 3.9495 2.7598 4.98688C2.61303 5.76469 2.6397 6.56531 2.83791 7.33163C3.03611 8.09795 3.40097 8.8111 3.90647 9.42021C4.37559 9.94959 4.64449 10.6266 4.66647 11.3335V13.3335C4.66647 13.864 4.87718 14.3727 5.25225 14.7478C5.62732 15.1228 6.13603 15.3335 6.66647 15.3335H9.33313C9.86356 15.3335 10.3723 15.1228 10.7473 14.7478C11.1224 14.3727 11.3331 13.864 11.3331 13.3335V11.4602C11.3555 10.6797 11.6423 9.92983 12.1465 9.33354C13.0299 8.24073 13.4463 6.84342 13.3052 5.44529C13.1642 4.04716 12.477 2.7612 11.3931 1.86687V1.88021ZM9.9998 13.3335C9.9998 13.5104 9.92956 13.6799 9.80454 13.8049C9.67951 13.93 9.50994 14.0002 9.33313 14.0002H6.66647C6.48965 14.0002 6.32009 13.93 6.19506 13.8049C6.07004 13.6799 5.9998 13.5104 5.9998 13.3335V12.6669H9.9998V13.3335ZM11.1131 8.50688C10.4428 9.30194 10.0517 10.2949 9.9998 11.3335H8.66647V9.33354C8.66647 9.15673 8.59623 8.98716 8.4712 8.86214C8.34618 8.73711 8.17661 8.66688 7.9998 8.66688C7.82299 8.66688 7.65342 8.73711 7.52839 8.86214C7.40337 8.98716 7.33313 9.15673 7.33313 9.33354V11.3335H5.9998C5.98221 10.3123 5.60443 9.33005 4.93313 8.56021C4.49023 8.02954 4.19239 7.39316 4.06867 6.71311C3.94495 6.03306 3.99957 5.33255 4.22719 4.6799C4.45481 4.02724 4.84768 3.4447 5.36748 2.98909C5.88728 2.53348 6.51627 2.22034 7.19313 2.08021C7.77483 1.96044 8.37591 1.9717 8.95271 2.11319C9.52952 2.25467 10.0676 2.52282 10.5278 2.89818C10.9881 3.27353 11.359 3.74666 11.6136 4.28322C11.8682 4.81978 12.0001 5.4063 11.9998 6.00021C12.0047 6.91345 11.6912 7.79985 11.1131 8.50688Z"
     fill="white" />
-</svg>
\ No newline at end of file
+</svg>
diff --git a/frontend/src/assets/loading-outer.svg b/frontend/src/assets/loading-outer.svg
index da669d7da4cb..aebe42c8e528 100644
--- a/frontend/src/assets/loading-outer.svg
+++ b/frontend/src/assets/loading-outer.svg
@@ -1,4 +1,4 @@
 <svg width="66" height="66" viewBox="0 0 66 66" fill="none" xmlns="http://www.w3.org/2000/svg">
   <path d="M63 33C63 16.4315 49.5685 3 33 3C16.4315 3 3 16.4315 3 33C3 49.5685 16.4315 63 33 63"
     stroke="#007AFF" stroke-width="6" stroke-linecap="round" />
-</svg>
\ No newline at end of file
+</svg>
diff --git a/frontend/src/assets/message.svg b/frontend/src/assets/message.svg
index cf1e6fc409de..3bf0a6e3e29b 100644
--- a/frontend/src/assets/message.svg
+++ b/frontend/src/assets/message.svg
@@ -2,4 +2,4 @@
   <path
     d="M19.8335 8.16634H8.16683C7.85741 8.16634 7.56066 8.28926 7.34187 8.50805C7.12308 8.72684 7.00016 9.02359 7.00016 9.33301C7.00016 9.64243 7.12308 9.93917 7.34187 10.158C7.56066 10.3768 7.85741 10.4997 8.16683 10.4997H19.8335C20.1429 10.4997 20.4397 10.3768 20.6585 10.158C20.8772 9.93917 21.0002 9.64243 21.0002 9.33301C21.0002 9.02359 20.8772 8.72684 20.6585 8.50805C20.4397 8.28926 20.1429 8.16634 19.8335 8.16634ZM19.8335 12.833H8.16683C7.85741 12.833 7.56066 12.9559 7.34187 13.1747C7.12308 13.3935 7.00016 13.6903 7.00016 13.9997C7.00016 14.3091 7.12308 14.6058 7.34187 14.8246C7.56066 15.0434 7.85741 15.1663 8.16683 15.1663H19.8335C20.1429 15.1663 20.4397 15.0434 20.6585 14.8246C20.8772 14.6058 21.0002 14.3091 21.0002 13.9997C21.0002 13.6903 20.8772 13.3935 20.6585 13.1747C20.4397 12.9559 20.1429 12.833 19.8335 12.833ZM22.1668 2.33301H5.8335C4.90524 2.33301 4.015 2.70176 3.35862 3.35813C2.70225 4.01451 2.3335 4.90475 2.3335 5.83301V17.4997C2.3335 18.4279 2.70225 19.3182 3.35862 19.9745C4.015 20.6309 4.90524 20.9997 5.8335 20.9997H19.3552L23.6718 25.328C23.7808 25.4361 23.9101 25.5217 24.0523 25.5797C24.1944 25.6378 24.3466 25.6672 24.5002 25.6663C24.6532 25.6703 24.805 25.6383 24.9435 25.573C25.1565 25.4855 25.3389 25.3369 25.4677 25.1458C25.5964 24.9548 25.6657 24.73 25.6668 24.4997V5.83301C25.6668 4.90475 25.2981 4.01451 24.6417 3.35813C23.9853 2.70176 23.0951 2.33301 22.1668 2.33301ZM23.3335 21.688L20.6618 19.0047C20.5528 18.8965 20.4235 18.811 20.2814 18.7529C20.1392 18.6949 19.987 18.6655 19.8335 18.6663H5.8335C5.52408 18.6663 5.22733 18.5434 5.00854 18.3246C4.78975 18.1058 4.66683 17.8091 4.66683 17.4997V5.83301C4.66683 5.52359 4.78975 5.22684 5.00854 5.00805C5.22733 4.78926 5.52408 4.66634 5.8335 4.66634H22.1668C22.4762 4.66634 22.773 4.78926 22.9918 5.00805C23.2106 5.22684 23.3335 5.52359 23.3335 5.83301V21.688Z"
     fill="white" />
-</svg>
\ No newline at end of file
+</svg>
diff --git a/frontend/src/assets/new-project.svg b/frontend/src/assets/new-project.svg
index e7573f625a4f..550d656e353f 100644
--- a/frontend/src/assets/new-project.svg
+++ b/frontend/src/assets/new-project.svg
@@ -3,4 +3,4 @@
   <path
     d="M20.4165 13.0837H14.9165V7.58366C14.9165 7.34054 14.8199 7.10739 14.648 6.93548C14.4761 6.76357 14.243 6.66699 13.9998 6.66699C13.7567 6.66699 13.5236 6.76357 13.3517 6.93548C13.1797 7.10739 13.0832 7.34054 13.0832 7.58366V13.0837H7.58317C7.34006 13.0837 7.1069 13.1802 6.93499 13.3521C6.76308 13.5241 6.6665 13.7572 6.6665 14.0003C6.6665 14.2434 6.76308 14.4766 6.93499 14.6485C7.1069 14.8204 7.34006 14.917 7.58317 14.917H13.0832V20.417C13.0832 20.6601 13.1797 20.8933 13.3517 21.0652C13.5236 21.2371 13.7567 21.3337 13.9998 21.3337C14.243 21.3337 14.4761 21.2371 14.648 21.0652C14.8199 20.8933 14.9165 20.6601 14.9165 20.417V14.917H20.4165C20.6596 14.917 20.8928 14.8204 21.0647 14.6485C21.2366 14.4766 21.3332 14.2434 21.3332 14.0003C21.3332 13.7572 21.2366 13.5241 21.0647 13.3521C20.8928 13.1802 20.6596 13.0837 20.4165 13.0837Z"
     fill="black" />
-</svg>
\ No newline at end of file
+</svg>
diff --git a/frontend/src/assets/refresh.svg b/frontend/src/assets/refresh.svg
index ae9e2bac3841..4bba1daa1adb 100644
--- a/frontend/src/assets/refresh.svg
+++ b/frontend/src/assets/refresh.svg
@@ -2,4 +2,4 @@
   <path
     d="M4.26446 0.763647C4.39463 0.633472 4.60569 0.633472 4.73586 0.763647L5.73586 1.76365C5.86604 1.89382 5.86604 2.10488 5.73586 2.23505L4.73586 3.23505C4.60569 3.36523 4.39463 3.36523 4.26446 3.23505C4.13429 3.10488 4.13429 2.89382 4.26446 2.76365L4.69542 2.33268H4.16683C2.98426 2.33268 2.00016 3.31678 2.00016 4.49935C2.00016 5.68192 2.98426 6.66602 4.16683 6.66602C5.3494 6.66602 6.3335 5.68192 6.3335 4.49935C6.3335 4.31525 6.48273 4.16602 6.66683 4.16602C6.85092 4.16602 7.00016 4.31525 7.00016 4.49935C7.00016 6.05011 5.71759 7.33268 4.16683 7.33268C2.61607 7.33268 1.3335 6.05011 1.3335 4.49935C1.3335 2.94859 2.61607 1.66602 4.16683 1.66602H4.69542L4.26446 1.23505C4.13429 1.10488 4.13429 0.893821 4.26446 0.763647Z"
     fill="white" />
-</svg>
\ No newline at end of file
+</svg>
diff --git a/frontend/src/components/AgentStatusBar.tsx b/frontend/src/components/AgentStatusBar.tsx
index c337a838f32e..7de9ae0397e8 100644
--- a/frontend/src/components/AgentStatusBar.tsx
+++ b/frontend/src/components/AgentStatusBar.tsx
@@ -1,6 +1,7 @@
 import React, { useEffect } from "react";
 import { useTranslation } from "react-i18next";
 import { useSelector } from "react-redux";
+import toast from "react-hot-toast";
 import { I18nKey } from "#/i18n/declaration";
 import { RootState } from "#/store";
 import AgentState from "#/types/AgentState";
@@ -16,7 +17,7 @@ enum IndicatorColor {
 }
 
 function AgentStatusBar() {
-  const { t } = useTranslation();
+  const { t, i18n } = useTranslation();
   const { curAgentState } = useSelector((state: RootState) => state.agent);
   const { curStatusMessage } = useSelector((state: RootState) => state.status);
 
@@ -94,15 +95,27 @@ function AgentStatusBar() {
   const [statusMessage, setStatusMessage] = React.useState<string>("");
 
   React.useEffect(() => {
-    if (curAgentState === AgentState.LOADING) {
-      const trimmedCustomMessage = curStatusMessage.status.trim();
-      if (trimmedCustomMessage) {
-        setStatusMessage(t(trimmedCustomMessage));
-        return;
+    let message = curStatusMessage.message || "";
+    if (curStatusMessage?.id) {
+      const id = curStatusMessage.id.trim();
+      if (i18n.exists(id)) {
+        message = t(curStatusMessage.id.trim()) || message;
       }
     }
+    if (curStatusMessage?.type === "error") {
+      toast.error(message);
+      return;
+    }
+    if (curAgentState === AgentState.LOADING && message.trim()) {
+      setStatusMessage(message);
+    } else {
+      setStatusMessage(AgentStatusMap[curAgentState].message);
+    }
+  }, [curStatusMessage.id]);
+
+  React.useEffect(() => {
     setStatusMessage(AgentStatusMap[curAgentState].message);
-  }, [curAgentState, curStatusMessage.status]);
+  }, [curAgentState]);
 
   return (
     <div className="flex flex-col items-center">
diff --git a/frontend/src/components/analytics-consent-form-modal.tsx b/frontend/src/components/analytics-consent-form-modal.tsx
new file mode 100644
index 000000000000..e122b9e8a9bf
--- /dev/null
+++ b/frontend/src/components/analytics-consent-form-modal.tsx
@@ -0,0 +1,42 @@
+import { useFetcher } from "@remix-run/react";
+import { ModalBackdrop } from "./modals/modal-backdrop";
+import ModalBody from "./modals/ModalBody";
+import ModalButton from "./buttons/ModalButton";
+import {
+  BaseModalTitle,
+  BaseModalDescription,
+} from "./modals/confirmation-modals/BaseModal";
+
+export function AnalyticsConsentFormModal() {
+  const fetcher = useFetcher({ key: "set-consent" });
+
+  return (
+    <ModalBackdrop>
+      <fetcher.Form
+        method="POST"
+        action="/set-consent"
+        className="flex flex-col gap-2"
+      >
+        <ModalBody>
+          <BaseModalTitle title="Your Privacy Preferences" />
+          <BaseModalDescription>
+            We use tools to understand how our application is used to improve
+            your experience. You can enable or disable analytics. Your
+            preferences will be stored and can be updated anytime.
+          </BaseModalDescription>
+
+          <label className="flex gap-2 items-center self-start">
+            <input name="analytics" type="checkbox" defaultChecked />
+            Send anonymous usage data
+          </label>
+
+          <ModalButton
+            type="submit"
+            text="Confirm Preferences"
+            className="bg-primary text-white w-full hover:opacity-80"
+          />
+        </ModalBody>
+      </fetcher.Form>
+    </ModalBackdrop>
+  );
+}
diff --git a/frontend/src/components/buttons/ModalButton.tsx b/frontend/src/components/buttons/ModalButton.tsx
index 29ec2ae5c566..e011fef76075 100644
--- a/frontend/src/components/buttons/ModalButton.tsx
+++ b/frontend/src/components/buttons/ModalButton.tsx
@@ -2,6 +2,7 @@ import clsx from "clsx";
 import React from "react";
 
 interface ModalButtonProps {
+  testId?: string;
   variant?: "default" | "text-like";
   onClick?: () => void;
   text: string;
@@ -13,6 +14,7 @@ interface ModalButtonProps {
 }
 
 function ModalButton({
+  testId,
   variant = "default",
   onClick,
   text,
@@ -24,6 +26,7 @@ function ModalButton({
 }: ModalButtonProps) {
   return (
     <button
+      data-testid={testId}
       type={type === "submit" ? "submit" : "button"}
       disabled={disabled}
       onClick={onClick}
diff --git a/frontend/src/components/chat-interface.tsx b/frontend/src/components/chat-interface.tsx
index 25a570736989..9e9af0739628 100644
--- a/frontend/src/components/chat-interface.tsx
+++ b/frontend/src/components/chat-interface.tsx
@@ -73,7 +73,7 @@ export function ChatInterface() {
           isErrorMessage(message) ? (
             <ErrorMessage
               key={index}
-              error={message.error}
+              id={message.id}
               message={message.message}
             />
           ) : (
diff --git a/frontend/src/components/chat/message.d.ts b/frontend/src/components/chat/message.d.ts
index e7248fbd6483..b2ccf43c994f 100644
--- a/frontend/src/components/chat/message.d.ts
+++ b/frontend/src/components/chat/message.d.ts
@@ -6,6 +6,7 @@ type Message = {
 };
 
 type ErrorMessage = {
-  error: string;
+  error: boolean;
+  id?: string;
   message: string;
 };
diff --git a/frontend/src/components/error-message.tsx b/frontend/src/components/error-message.tsx
index cded8c3729f4..86454539f778 100644
--- a/frontend/src/components/error-message.tsx
+++ b/frontend/src/components/error-message.tsx
@@ -1,14 +1,41 @@
+import { useState, useEffect } from "react";
+import { useTranslation } from "react-i18next";
+
 interface ErrorMessageProps {
-  error: string;
+  id?: string;
   message: string;
 }
 
-export function ErrorMessage({ error, message }: ErrorMessageProps) {
+export function ErrorMessage({ id, message }: ErrorMessageProps) {
+  const { t, i18n } = useTranslation();
+  const [showDetails, setShowDetails] = useState(true);
+  const [headline, setHeadline] = useState("");
+  const [details, setDetails] = useState(message);
+
+  useEffect(() => {
+    if (id && i18n.exists(id)) {
+      setHeadline(t(id));
+      setDetails(message);
+      setShowDetails(false);
+    }
+  }, [id, message, i18n.language]);
+
   return (
     <div className="flex gap-2 items-center justify-start border-l-2 border-danger pl-2 my-2 py-2">
       <div className="text-sm leading-4 flex flex-col gap-2">
-        <p className="text-danger font-bold">{error}</p>
-        <p className="text-neutral-300">{message}</p>
+        {headline && <p className="text-danger font-bold">{headline}</p>}
+        {headline && (
+          <button
+            type="button"
+            onClick={() => setShowDetails(!showDetails)}
+            className="cursor-pointer text-left"
+          >
+            {showDetails
+              ? t("ERROR_MESSAGE$HIDE_DETAILS")
+              : t("ERROR_MESSAGE$SHOW_DETAILS")}
+          </button>
+        )}
+        {showDetails && <p className="text-neutral-300">{details}</p>}
       </div>
     </div>
   );
diff --git a/frontend/src/components/feedback-form.tsx b/frontend/src/components/feedback-form.tsx
index 4e1ddde63548..078e4b0ccca6 100644
--- a/frontend/src/components/feedback-form.tsx
+++ b/frontend/src/components/feedback-form.tsx
@@ -1,8 +1,8 @@
 import React from "react";
 import hotToast from "react-hot-toast";
 import ModalButton from "./buttons/ModalButton";
-import { request } from "#/services/api";
 import { Feedback } from "#/api/open-hands.types";
+import OpenHands from "#/api/open-hands";
 
 const FEEDBACK_VERSION = "1.0";
 const VIEWER_PAGE = "https://www.all-hands.dev/share";
@@ -71,13 +71,7 @@ export function FeedbackForm({ onClose, polarity }: FeedbackFormProps) {
       token: "",
     };
 
-    const response = await request("/api/submit-feedback", {
-      method: "POST",
-      body: JSON.stringify(feedback),
-      headers: {
-        "Content-Type": "application/json",
-      },
-    });
+    const response = await OpenHands.submitFeedback(feedback);
     const { message, feedback_id, password } = response.body; // eslint-disable-line
     const link = `${VIEWER_PAGE}?share_id=${feedback_id}`;
     shareFeedbackToast(message, link, password);
diff --git a/frontend/src/components/file-explorer/FileExplorer.tsx b/frontend/src/components/file-explorer/FileExplorer.tsx
index c6e2c249feff..8db4460b1ae6 100644
--- a/frontend/src/components/file-explorer/FileExplorer.tsx
+++ b/frontend/src/components/file-explorer/FileExplorer.tsx
@@ -91,14 +91,15 @@ function ExplorerActions({
 }
 
 interface FileExplorerProps {
+  isOpen: boolean;
+  onToggle: () => void;
   error: string | null;
 }
 
-function FileExplorer({ error }: FileExplorerProps) {
+function FileExplorer({ error, isOpen, onToggle }: FileExplorerProps) {
   const { revalidate } = useRevalidator();
 
   const { paths, setPaths } = useFiles();
-  const [isHidden, setIsHidden] = React.useState(false);
   const [isDragging, setIsDragging] = React.useState(false);
 
   const { curAgentState } = useSelector((state: RootState) => state.agent);
@@ -117,52 +118,47 @@ function FileExplorer({ error }: FileExplorerProps) {
       return;
     }
     dispatch(setRefreshID(Math.random()));
-    // TODO: Get token from data loader
-    const token = localStorage.getItem("token");
-    if (token) OpenHands.getFiles(token).then(setPaths);
+    OpenHands.getFiles().then(setPaths);
     revalidate();
   };
 
   const uploadFileData = async (files: FileList) => {
     try {
-      const token = localStorage.getItem("token");
-      if (token) {
-        const result = await OpenHands.uploadFiles(token, Array.from(files));
+      const result = await OpenHands.uploadFiles(Array.from(files));
 
-        if (isOpenHandsErrorResponse(result)) {
-          // Handle error response
-          toast.error(
-            `upload-error-${new Date().getTime()}`,
-            result.error || t(I18nKey.EXPLORER$UPLOAD_ERROR_MESSAGE),
-          );
-          return;
-        }
+      if (isOpenHandsErrorResponse(result)) {
+        // Handle error response
+        toast.error(
+          `upload-error-${new Date().getTime()}`,
+          result.error || t(I18nKey.EXPLORER$UPLOAD_ERROR_MESSAGE),
+        );
+        return;
+      }
 
-        const uploadedCount = result.uploaded_files.length;
-        const skippedCount = result.skipped_files.length;
+      const uploadedCount = result.uploaded_files.length;
+      const skippedCount = result.skipped_files.length;
 
-        if (uploadedCount > 0) {
-          toast.success(
-            `upload-success-${new Date().getTime()}`,
-            t(I18nKey.EXPLORER$UPLOAD_SUCCESS_MESSAGE, {
-              count: uploadedCount,
-            }),
-          );
-        }
-
-        if (skippedCount > 0) {
-          const message = t(I18nKey.EXPLORER$UPLOAD_PARTIAL_SUCCESS_MESSAGE, {
-            count: skippedCount,
-          });
-          toast.info(message);
-        }
+      if (uploadedCount > 0) {
+        toast.success(
+          `upload-success-${new Date().getTime()}`,
+          t(I18nKey.EXPLORER$UPLOAD_SUCCESS_MESSAGE, {
+            count: uploadedCount,
+          }),
+        );
+      }
 
-        if (uploadedCount === 0 && skippedCount === 0) {
-          toast.info(t(I18nKey.EXPLORER$NO_FILES_UPLOADED_MESSAGE));
-        }
+      if (skippedCount > 0) {
+        const message = t(I18nKey.EXPLORER$UPLOAD_PARTIAL_SUCCESS_MESSAGE, {
+          count: skippedCount,
+        });
+        toast.info(message);
+      }
 
-        refreshWorkspace();
+      if (uploadedCount === 0 && skippedCount === 0) {
+        toast.info(t(I18nKey.EXPLORER$NO_FILES_UPLOADED_MESSAGE));
       }
+
+      refreshWorkspace();
     } catch (e) {
       // Handle unexpected errors (network issues, etc.)
       toast.error(
@@ -211,7 +207,7 @@ function FileExplorer({ error }: FileExplorerProps) {
       <div
         className={twMerge(
           "bg-neutral-800 h-full border-r-1 border-r-neutral-600 flex flex-col",
-          isHidden ? "w-12" : "w-60",
+          !isOpen ? "w-12" : "w-60",
         )}
       >
         <div className="flex flex-col relative h-full px-3 py-2">
@@ -219,17 +215,17 @@ function FileExplorer({ error }: FileExplorerProps) {
             <div
               className={twMerge(
                 "flex items-center",
-                isHidden ? "justify-center" : "justify-between",
+                !isOpen ? "justify-center" : "justify-between",
               )}
             >
-              {!isHidden && (
+              {isOpen && (
                 <div className="text-neutral-300 font-bold text-sm">
                   {t(I18nKey.EXPLORER$LABEL_WORKSPACE)}
                 </div>
               )}
               <ExplorerActions
-                isHidden={isHidden}
-                toggleHidden={() => setIsHidden((prev) => !prev)}
+                isHidden={!isOpen}
+                toggleHidden={onToggle}
                 onRefresh={refreshWorkspace}
                 onUpload={selectFileInput}
               />
@@ -237,7 +233,7 @@ function FileExplorer({ error }: FileExplorerProps) {
           </div>
           {!error && (
             <div className="overflow-auto flex-grow">
-              <div style={{ display: isHidden ? "none" : "block" }}>
+              <div style={{ display: !isOpen ? "none" : "block" }}>
                 <ExplorerTree files={paths} />
               </div>
             </div>
diff --git a/frontend/src/components/file-explorer/TreeNode.tsx b/frontend/src/components/file-explorer/TreeNode.tsx
index fd44cd88cd72..b3aa3c28335c 100644
--- a/frontend/src/components/file-explorer/TreeNode.tsx
+++ b/frontend/src/components/file-explorer/TreeNode.tsx
@@ -59,14 +59,11 @@ function TreeNode({ path, defaultOpen = false }: TreeNodeProps) {
       return;
     }
 
-    const token = localStorage.getItem("token");
-    if (token) {
-      try {
-        const newChildren = await OpenHands.getFiles(token, path);
-        setChildren(newChildren);
-      } catch (error) {
-        toast.error("Failed to fetch files");
-      }
+    try {
+      const newChildren = await OpenHands.getFiles(path);
+      setChildren(newChildren);
+    } catch (error) {
+      toast.error("Failed to fetch files");
     }
   };
 
@@ -77,15 +74,13 @@ function TreeNode({ path, defaultOpen = false }: TreeNodeProps) {
   }, [refreshID, isOpen]);
 
   const handleClick = async () => {
-    const token = localStorage.getItem("token");
-
     if (isDirectory) {
       setIsOpen((prev) => !prev);
-    } else if (token) {
+    } else {
       const code = modifiedFiles[path] || files[path];
 
       try {
-        const fetchedCode = await OpenHands.getFile(token, path);
+        const fetchedCode = await OpenHands.getFile(path);
         setSelectedPath(path);
         if (!code || fetchedCode !== files[path]) {
           setFileContent(path, fetchedCode);
diff --git a/frontend/src/components/github-repositories-suggestion-box.tsx b/frontend/src/components/github-repositories-suggestion-box.tsx
new file mode 100644
index 000000000000..c39abe11a9ea
--- /dev/null
+++ b/frontend/src/components/github-repositories-suggestion-box.tsx
@@ -0,0 +1,94 @@
+import React from "react";
+import {
+  isGitHubErrorReponse,
+  retrieveAllGitHubUserRepositories,
+} from "#/api/github";
+import { SuggestionBox } from "#/routes/_oh._index/suggestion-box";
+import { ConnectToGitHubModal } from "./modals/connect-to-github-modal";
+import { ModalBackdrop } from "./modals/modal-backdrop";
+import { GitHubRepositorySelector } from "#/routes/_oh._index/github-repo-selector";
+import ModalButton from "./buttons/ModalButton";
+import GitHubLogo from "#/assets/branding/github-logo.svg?react";
+
+interface GitHubAuthProps {
+  onConnectToGitHub: () => void;
+  repositories: GitHubRepository[];
+  isLoggedIn: boolean;
+}
+
+function GitHubAuth({
+  onConnectToGitHub,
+  repositories,
+  isLoggedIn,
+}: GitHubAuthProps) {
+  if (isLoggedIn) {
+    return <GitHubRepositorySelector repositories={repositories} />;
+  }
+
+  return (
+    <ModalButton
+      text="Connect to GitHub"
+      icon={<GitHubLogo width={20} height={20} />}
+      className="bg-[#791B80] w-full"
+      onClick={onConnectToGitHub}
+    />
+  );
+}
+
+interface GitHubRepositoriesSuggestionBoxProps {
+  repositories: Awaited<
+    ReturnType<typeof retrieveAllGitHubUserRepositories>
+  > | null;
+  gitHubAuthUrl: string | null;
+  user: GitHubErrorReponse | GitHubUser | null;
+}
+
+export function GitHubRepositoriesSuggestionBox({
+  repositories,
+  gitHubAuthUrl,
+  user,
+}: GitHubRepositoriesSuggestionBoxProps) {
+  const [connectToGitHubModalOpen, setConnectToGitHubModalOpen] =
+    React.useState(false);
+
+  const handleConnectToGitHub = () => {
+    if (gitHubAuthUrl) {
+      window.location.href = gitHubAuthUrl;
+    } else {
+      setConnectToGitHubModalOpen(true);
+    }
+  };
+
+  if (isGitHubErrorReponse(repositories)) {
+    return (
+      <SuggestionBox
+        title="Error Fetching Repositories"
+        content={
+          <p className="text-danger text-center">{repositories.message}</p>
+        }
+      />
+    );
+  }
+
+  return (
+    <>
+      <SuggestionBox
+        title="Open a Repo"
+        content={
+          <GitHubAuth
+            isLoggedIn={!!user && !isGitHubErrorReponse(user)}
+            repositories={repositories || []}
+            onConnectToGitHub={handleConnectToGitHub}
+          />
+        }
+      />
+      {connectToGitHubModalOpen && (
+        <ModalBackdrop onClose={() => setConnectToGitHubModalOpen(false)}>
+          <ConnectToGitHubModal
+            onClose={() => setConnectToGitHubModalOpen(false)}
+          />
+        </ModalBackdrop>
+      )}
+    </>
+  );
+}
diff --git a/frontend/src/components/modals/AccountSettingsModal.tsx b/frontend/src/components/modals/AccountSettingsModal.tsx
index 1acdacc0319d..0ca6df56be32 100644
--- a/frontend/src/components/modals/AccountSettingsModal.tsx
+++ b/frontend/src/components/modals/AccountSettingsModal.tsx
@@ -14,12 +14,14 @@ interface AccountSettingsModalProps {
   onClose: () => void;
   selectedLanguage: string;
   gitHubError: boolean;
+  analyticsConsent: string | null;
 }
 
 function AccountSettingsModal({
   onClose,
   selectedLanguage,
   gitHubError,
+  analyticsConsent,
 }: AccountSettingsModalProps) {
   const data = useRouteLoaderData<typeof clientLoader>("routes/_oh");
   const settingsFetcher = useFetcher<typeof settingsClientAction>({
@@ -32,6 +34,7 @@ function AccountSettingsModal({
     const formData = new FormData(event.currentTarget);
     const language = formData.get("language")?.toString();
     const ghToken = formData.get("ghToken")?.toString();
+    const analytics = formData.get("analytics")?.toString() === "on";
 
     const accountForm = new FormData();
     const loginForm = new FormData();
@@ -44,6 +47,7 @@ function AccountSettingsModal({
       accountForm.append("language", languageKey ?? "en");
     }
     if (ghToken) loginForm.append("ghToken", ghToken);
+    accountForm.append("analytics", analytics.toString());
 
     settingsFetcher.submit(accountForm, {
       method: "POST",
@@ -101,6 +105,15 @@ function AccountSettingsModal({
           )}
         </div>
 
+        <label className="flex gap-2 items-center self-start">
+          <input
+            name="analytics"
+            type="checkbox"
+            defaultChecked={analyticsConsent === "true"}
+          />
+          Enable analytics
+        </label>
+
         <div className="flex flex-col gap-2 w-full">
           <ModalButton
             disabled={
diff --git a/frontend/src/components/modals/confirmation-modals/BaseModal.tsx b/frontend/src/components/modals/confirmation-modals/BaseModal.tsx
index f7dbb1123054..6c353e190021 100644
--- a/frontend/src/components/modals/confirmation-modals/BaseModal.tsx
+++ b/frontend/src/components/modals/confirmation-modals/BaseModal.tsx
@@ -21,13 +21,17 @@ export function BaseModalTitle({ title }: BaseModalTitleProps) {
 }
 
 interface BaseModalDescriptionProps {
-  description: React.ReactNode;
+  description?: React.ReactNode;
+  children?: React.ReactNode;
 }
 
 export function BaseModalDescription({
   description,
+  children,
 }: BaseModalDescriptionProps) {
-  return <span className="text-xs text-[#A3A3A3]">{description}</span>;
+  return (
+    <span className="text-xs text-[#A3A3A3]">{children || description}</span>
+  );
 }
 
 interface BaseModalProps {
diff --git a/frontend/src/components/modals/connect-to-github-modal.tsx b/frontend/src/components/modals/connect-to-github-modal.tsx
index e0315d14c84f..165dab852105 100644
--- a/frontend/src/components/modals/connect-to-github-modal.tsx
+++ b/frontend/src/components/modals/connect-to-github-modal.tsx
@@ -53,6 +53,7 @@ export function ConnectToGitHubModal({ onClose }: ConnectToGitHubModalProps) {
 
         <div className="flex flex-col gap-2 w-full">
           <ModalButton
+            testId="connect-to-github"
             type="submit"
             text="Connect"
             disabled={fetcher.state === "submitting"}
diff --git a/frontend/src/context/socket.tsx b/frontend/src/context/socket.tsx
index 259acd2e3e28..14c6cfeefe0d 100644
--- a/frontend/src/context/socket.tsx
+++ b/frontend/src/context/socket.tsx
@@ -1,7 +1,6 @@
 import React from "react";
 import { Data } from "ws";
 import EventLogger from "#/utils/event-logger";
-import { getValidFallbackHost } from "#/utils/get-valid-fallback-host";
 
 interface WebSocketClientOptions {
   token: string | null;
@@ -46,12 +45,17 @@ function SocketProvider({ children }: SocketProviderProps) {
       );
     }
 
-    const fallback = getValidFallbackHost();
-    const baseUrl = import.meta.env.VITE_BACKEND_BASE_URL || fallback;
+    const baseUrl =
+      import.meta.env.VITE_BACKEND_BASE_URL || window?.location.host;
     const protocol = window.location.protocol === "https:" ? "wss:" : "ws:";
-    const ws = new WebSocket(
-      `${protocol}//${baseUrl}/ws${options?.token ? `?token=${options.token}` : ""}`,
-    );
+    const sessionToken = options?.token || "NO_JWT"; // not allowed to be empty or duplicated
+    const ghToken = localStorage.getItem("ghToken") || "NO_GITHUB";
+
+    const ws = new WebSocket(`${protocol}//${baseUrl}/ws`, [
+      "openhands",
+      sessionToken,
+      ghToken,
+    ]);
 
     ws.addEventListener("open", (event) => {
       setIsConnected(true);
diff --git a/frontend/src/entry.client.tsx b/frontend/src/entry.client.tsx
index 3e87b2736e23..8a6d4fac2dfc 100644
--- a/frontend/src/entry.client.tsx
+++ b/frontend/src/entry.client.tsx
@@ -6,13 +6,25 @@
  */
 
 import { RemixBrowser } from "@remix-run/react";
-import { startTransition, StrictMode } from "react";
+import React, { startTransition, StrictMode } from "react";
 import { hydrateRoot } from "react-dom/client";
 import { Provider } from "react-redux";
+import posthog from "posthog-js";
 import { SocketProvider } from "./context/socket";
 import "./i18n";
 import store from "./store";
 
+function PosthogInit() {
+  React.useEffect(() => {
+    posthog.init("phc_3ESMmY9SgqEAGBB6sMGK5ayYHkeUuknH2vP6FmWH9RA", {
+      api_host: "https://us.i.posthog.com",
+      person_profiles: "identified_only",
+    });
+  }, []);
+
+  return null;
+}
+
 async function prepareApp() {
   if (
     process.env.NODE_ENV === "development" &&
@@ -34,6 +46,7 @@ prepareApp().then(() =>
         <SocketProvider>
           <Provider store={store}>
             <RemixBrowser />
+            <PosthogInit />
           </Provider>
         </SocketProvider>
       </StrictMode>,
diff --git a/frontend/src/i18n/translation.json b/frontend/src/i18n/translation.json
index 795c60e051f2..6db2520b6b8a 100644
--- a/frontend/src/i18n/translation.json
+++ b/frontend/src/i18n/translation.json
@@ -1441,6 +1441,12 @@
     "fr": "Privé",
     "tr": "Özel"
   },
+  "ERROR_MESSAGE$SHOW_DETAILS": {
+    "en": "Show details"
+  },
+  "ERROR_MESSAGE$HIDE_DETAILS": {
+    "en": "Hide details"
+  },
   "STATUS$STARTING_RUNTIME": {
     "en": "Starting Runtime...",
     "zh-CN": "启动运行时...",
@@ -1510,5 +1516,17 @@
     "ar": "في انتظار جاهزية العميل...",
     "fr": "En attente que le client soit prêt...",
     "tr": "İstemcinin hazır olması bekleniyor..."
+  },
+  "STATUS$ERROR_LLM_AUTHENTICATION": {
+    "en": "Error authenticating with the LLM provider. Please check your API key"
+  },
+  "STATUS$ERROR_RUNTIME_DISCONNECTED": {
+    "en": "There was an error while connecting to the runtime. Please refresh the page."
+  },
+  "AGENT_ERROR$BAD_ACTION": {
+    "en": "Agent tried to execute a malformed action."
+  },
+  "AGENT_ERROR$ACTION_TIMEOUT": {
+    "en": "Action timed out."
   }
 }
diff --git a/frontend/src/mocks/handlers.ts b/frontend/src/mocks/handlers.ts
index 91a1c5d7acbc..97a7f9cf9c84 100644
--- a/frontend/src/mocks/handlers.ts
+++ b/frontend/src/mocks/handlers.ts
@@ -1,7 +1,7 @@
 import { delay, http, HttpResponse } from "msw";
 
 const openHandsHandlers = [
-  http.get("http://localhost:3000/api/options/models", async () => {
+  http.get("/api/options/models", async () => {
     await delay();
     return HttpResponse.json([
       "gpt-3.5-turbo",
@@ -10,17 +10,17 @@ const openHandsHandlers = [
     ]);
   }),
 
-  http.get("http://localhost:3000/api/options/agents", async () => {
+  http.get("/api/options/agents", async () => {
     await delay();
     return HttpResponse.json(["CodeActAgent", "CoActAgent"]);
   }),
 
-  http.get("http://localhost:3000/api/options/security-analyzers", async () => {
+  http.get("/api/options/security-analyzers", async () => {
     await delay();
     return HttpResponse.json(["mock-invariant"]);
   }),
 
-  http.get("http://localhost:3000/api/list-files", async ({ request }) => {
+  http.get("http://localhost:3001/api/list-files", async ({ request }) => {
     await delay();
 
     const token = request.headers
@@ -32,11 +32,11 @@ const openHandsHandlers = [
     return HttpResponse.json(["file1.ts", "dir1/file2.ts", "file3.ts"]);
   }),
 
-  http.post("http://localhost:3000/api/save-file", () =>
+  http.post("http://localhost:3001/api/save-file", () =>
     HttpResponse.json(null, { status: 200 }),
   ),
 
-  http.get("http://localhost:3000/api/select-file", async ({ request }) => {
+  http.get("http://localhost:3001/api/select-file", async ({ request }) => {
     await delay();
 
     const token = request.headers
@@ -58,7 +58,7 @@ const openHandsHandlers = [
     return HttpResponse.json(null, { status: 404 });
   }),
 
-  http.post("http://localhost:3000/api/submit-feedback", async () => {
+  http.post("http://localhost:3001/api/submit-feedback", async () => {
     await delay(1200);
 
     return HttpResponse.json({
@@ -70,7 +70,9 @@ const openHandsHandlers = [
 
 export const handlers = [
   ...openHandsHandlers,
-  http.get("https://api.github.com/user/repos", ({ request }) => {
+  http.get("https://api.github.com/user/repos", async ({ request }) => {
+    if (import.meta.env.MODE !== "test") await delay(3500);
+
     const token = request.headers
       .get("Authorization")
       ?.replace("Bearer", "")
@@ -85,7 +87,20 @@ export const handlers = [
       { id: 2, full_name: "octocat/earth" },
     ]);
   }),
-  http.post("http://localhost:3000/api/submit-feedback", async () =>
+  http.get("https://api.github.com/user", () => {
+    const user: GitHubUser = {
+      id: 1,
+      login: "octocat",
+      avatar_url: "https://avatars.githubusercontent.com/u/583231?v=4",
+    };
+
+    return HttpResponse.json(user);
+  }),
+  http.post("http://localhost:3001/api/submit-feedback", async () =>
     HttpResponse.json({ statusCode: 200 }, { status: 200 }),
   ),
+  http.post("https://us.i.posthog.com/e", async () =>
+    HttpResponse.json(null, { status: 200 }),
+  ),
+  http.get("/config.json", () => HttpResponse.json({ APP_MODE: "oss" })),
 ];
diff --git a/frontend/src/routes/_oh._index/github-repo-selector.tsx b/frontend/src/routes/_oh._index/github-repo-selector.tsx
index 73dc03cf7d97..370bd3a613e4 100644
--- a/frontend/src/routes/_oh._index/github-repo-selector.tsx
+++ b/frontend/src/routes/_oh._index/github-repo-selector.tsx
@@ -1,5 +1,6 @@
 import { Autocomplete, AutocompleteItem } from "@nextui-org/react";
 import { useDispatch } from "react-redux";
+import { useNavigate } from "react-router-dom";
 import { setSelectedRepository } from "#/state/initial-query-slice";
 
 interface GitHubRepositorySelectorProps {
@@ -9,6 +10,7 @@ interface GitHubRepositorySelectorProps {
 export function GitHubRepositorySelector({
   repositories,
 }: GitHubRepositorySelectorProps) {
+  const navigate = useNavigate();
   const dispatch = useDispatch();
 
   const handleRepoSelection = (id: string | null) => {
@@ -16,6 +18,7 @@ export function GitHubRepositorySelector({
     if (repo) {
       // set query param
       dispatch(setSelectedRepository(repo.full_name));
+      navigate("/app");
     }
   };
 
@@ -26,6 +29,7 @@ export function GitHubRepositorySelector({
 
   return (
     <Autocomplete
+      data-testid="github-repo-selector"
       name="repo"
       aria-label="GitHub Repository"
       placeholder="Select a GitHub project"
@@ -39,7 +43,11 @@ export function GitHubRepositorySelector({
       clearButtonProps={{ onClick: handleClearSelection }}
     >
       {repositories.map((repo) => (
-        <AutocompleteItem key={repo.id} value={repo.id}>
+        <AutocompleteItem
+          data-testid="github-repo-item"
+          key={repo.id}
+          value={repo.id}
+        >
           {repo.full_name}
         </AutocompleteItem>
       ))}
diff --git a/frontend/src/routes/_oh._index/route.tsx b/frontend/src/routes/_oh._index/route.tsx
index edc22c6dfca4..5f1df1b6c0a1 100644
--- a/frontend/src/routes/_oh._index/route.tsx
+++ b/frontend/src/routes/_oh._index/route.tsx
@@ -1,54 +1,24 @@
 import {
+  Await,
   ClientActionFunctionArgs,
   ClientLoaderFunctionArgs,
-  json,
+  defer,
   redirect,
   useLoaderData,
+  useNavigate,
   useRouteLoaderData,
 } from "@remix-run/react";
-import React from "react";
+import React, { Suspense } from "react";
 import { SuggestionBox } from "./suggestion-box";
 import { TaskForm } from "./task-form";
 import { HeroHeading } from "./hero-heading";
-import { GitHubRepositorySelector } from "./github-repo-selector";
-import {
-  isGitHubErrorReponse,
-  retrieveAllGitHubUserRepositories,
-} from "#/api/github";
-import ModalButton from "#/components/buttons/ModalButton";
-import GitHubLogo from "#/assets/branding/github-logo.svg?react";
-import { ConnectToGitHubModal } from "#/components/modals/connect-to-github-modal";
-import { ModalBackdrop } from "#/components/modals/modal-backdrop";
+import { retrieveAllGitHubUserRepositories } from "#/api/github";
 import store from "#/store";
 import { setInitialQuery } from "#/state/initial-query-slice";
 import { clientLoader as rootClientLoader } from "#/routes/_oh";
 import OpenHands from "#/api/open-hands";
 import { generateGitHubAuthUrl } from "#/utils/generate-github-auth-url";
-
-interface GitHubAuthProps {
-  onConnectToGitHub: () => void;
-  repositories: GitHubRepository[];
-  isLoggedIn: boolean;
-}
-
-function GitHubAuth({
-  onConnectToGitHub,
-  repositories,
-  isLoggedIn,
-}: GitHubAuthProps) {
-  if (isLoggedIn) {
-    return <GitHubRepositorySelector repositories={repositories} />;
-  }
-
-  return (
-    <ModalButton
-      text="Connect to GitHub"
-      icon={<GitHubLogo width={20} height={20} />}
-      className="bg-[#791B80] w-full"
-      onClick={onConnectToGitHub}
-    />
-  );
-}
+import { GitHubRepositoriesSuggestionBox } from "#/components/github-repositories-suggestion-box";
 
 export const clientLoader = async ({ request }: ClientLoaderFunctionArgs) => {
   let isSaas = false;
@@ -67,12 +37,12 @@ export const clientLoader = async ({ request }: ClientLoaderFunctionArgs) => {
   const token = localStorage.getItem("token");
   if (token) return redirect("/app");
 
-  let repositories: GitHubRepository[] = [];
+  let repositories: ReturnType<
+    typeof retrieveAllGitHubUserRepositories
+  > | null = null;
   if (ghToken) {
-    const data = await retrieveAllGitHubUserRepositories(ghToken);
-    if (!isGitHubErrorReponse(data)) {
-      repositories = data;
-    }
+    const data = retrieveAllGitHubUserRepositories(ghToken);
+    repositories = data;
   }
 
   let githubAuthUrl: string | null = null;
@@ -81,7 +51,7 @@ export const clientLoader = async ({ request }: ClientLoaderFunctionArgs) => {
     githubAuthUrl = generateGitHubAuthUrl(githubClientId, requestUrl);
   }
 
-  return json({ repositories, githubAuthUrl });
+  return defer({ repositories, githubAuthUrl });
 };
 
 export const clientAction = async ({ request }: ClientActionFunctionArgs) => {
@@ -93,40 +63,40 @@ export const clientAction = async ({ request }: ClientActionFunctionArgs) => {
 };
 
 function Home() {
+  const navigate = useNavigate();
   const rootData = useRouteLoaderData<typeof rootClientLoader>("routes/_oh");
   const { repositories, githubAuthUrl } = useLoaderData<typeof clientLoader>();
-  const [connectToGitHubModalOpen, setConnectToGitHubModalOpen] =
-    React.useState(false);
   const [importedFile, setImportedFile] = React.useState<File | null>(null);
 
-  const handleConnectToGitHub = () => {
-    if (githubAuthUrl) {
-      window.location.href = githubAuthUrl;
-    } else {
-      setConnectToGitHubModalOpen(true);
-    }
-  };
-
   return (
-    <div className="bg-root-secondary h-full rounded-xl flex flex-col items-center justify-center relative overflow-y-auto">
+    <div
+      data-testid="root-index"
+      className="bg-root-secondary h-full rounded-xl flex flex-col items-center justify-center relative overflow-y-auto"
+    >
       <HeroHeading />
       <div className="flex flex-col gap-16 w-[600px] items-center">
         <div className="flex flex-col gap-2 w-full">
           <TaskForm importedProjectZip={importedFile} />
         </div>
         <div className="flex gap-4 w-full">
-          <SuggestionBox
-            title="Open a Repo"
-            content={
-              <GitHubAuth
-                isLoggedIn={
-                  !!rootData?.user && !isGitHubErrorReponse(rootData.user)
-                }
-                repositories={repositories}
-                onConnectToGitHub={handleConnectToGitHub}
+          <Suspense
+            fallback={
+              <SuggestionBox
+                title="Open a Repo"
+                content="Loading repositories..."
               />
             }
-          />
+          >
+            <Await resolve={repositories}>
+              {(resolvedRepositories) => (
+                <GitHubRepositoriesSuggestionBox
+                  repositories={resolvedRepositories}
+                  gitHubAuthUrl={githubAuthUrl}
+                  user={rootData?.user || null}
+                />
+              )}
+            </Await>
+          </Suspense>
           <SuggestionBox
             title={importedFile ? "Project Loaded" : "+ Import Project"}
             content={
@@ -148,6 +118,7 @@ function Home() {
                       if (event.target.files) {
                         const zip = event.target.files[0];
                         setImportedFile(zip);
+                        navigate("/app");
                       } else {
                         // TODO: handle error
                       }
@@ -159,13 +130,6 @@ function Home() {
           />
         </div>
       </div>
-      {connectToGitHubModalOpen && (
-        <ModalBackdrop onClose={() => setConnectToGitHubModalOpen(false)}>
-          <ConnectToGitHubModal
-            onClose={() => setConnectToGitHubModalOpen(false)}
-          />
-        </ModalBackdrop>
-      )}
     </div>
   );
 }
diff --git a/frontend/src/routes/_oh.app._index/code-editor-component.tsx b/frontend/src/routes/_oh.app._index/code-editor-component.tsx
index cf94ed863a41..8182805193a0 100644
--- a/frontend/src/routes/_oh.app._index/code-editor-component.tsx
+++ b/frontend/src/routes/_oh.app._index/code-editor-component.tsx
@@ -1,18 +1,21 @@
-import { Editor, Monaco } from "@monaco-editor/react";
+import { Editor, EditorProps } from "@monaco-editor/react";
 import React from "react";
 import { useTranslation } from "react-i18next";
 import { VscCode } from "react-icons/vsc";
-import { type editor } from "monaco-editor";
 import toast from "react-hot-toast";
 import { I18nKey } from "#/i18n/declaration";
 import { useFiles } from "#/context/files";
 import OpenHands from "#/api/open-hands";
 
 interface CodeEditorCompoonentProps {
+  onMount: EditorProps["onMount"];
   isReadOnly: boolean;
 }
 
-function CodeEditorCompoonent({ isReadOnly }: CodeEditorCompoonentProps) {
+function CodeEditorCompoonent({
+  onMount,
+  isReadOnly,
+}: CodeEditorCompoonentProps) {
   const { t } = useTranslation();
   const {
     files,
@@ -22,22 +25,6 @@ function CodeEditorCompoonent({ isReadOnly }: CodeEditorCompoonentProps) {
     saveFileContent: saveNewFileContent,
   } = useFiles();
 
-  const handleEditorDidMount = React.useCallback(
-    (editor: editor.IStandaloneCodeEditor, monaco: Monaco): void => {
-      monaco.editor.defineTheme("my-theme", {
-        base: "vs-dark",
-        inherit: true,
-        rules: [],
-        colors: {
-          "editor.background": "#171717",
-        },
-      });
-
-      monaco.editor.setTheme("my-theme");
-    },
-    [],
-  );
-
   const handleEditorChange = (value: string | undefined) => {
     if (selectedPath && value) modifyFileContent(selectedPath, value);
   };
@@ -49,8 +36,7 @@ function CodeEditorCompoonent({ isReadOnly }: CodeEditorCompoonentProps) {
 
         if (content) {
           try {
-            const token = localStorage.getItem("token")?.toString();
-            if (token) await OpenHands.saveFile(token, selectedPath, content);
+            await OpenHands.saveFile(selectedPath, content);
           } catch (error) {
             toast.error("Failed to save file");
           }
@@ -68,7 +54,7 @@ function CodeEditorCompoonent({ isReadOnly }: CodeEditorCompoonentProps) {
     return (
       <div
         data-testid="code-editor-empty-message"
-        className="flex flex-col items-center text-neutral-400"
+        className="flex flex-col h-full items-center justify-center text-neutral-400"
       >
         <VscCode size={100} />
         {t(I18nKey.CODE_EDITOR$EMPTY_MESSAGE)}
@@ -79,7 +65,6 @@ function CodeEditorCompoonent({ isReadOnly }: CodeEditorCompoonentProps) {
   return (
     <Editor
       data-testid="code-editor"
-      height="100%"
       path={selectedPath ?? undefined}
       defaultValue=""
       value={
@@ -87,7 +72,7 @@ function CodeEditorCompoonent({ isReadOnly }: CodeEditorCompoonentProps) {
           ? modifiedFiles[selectedPath] || files[selectedPath]
           : undefined
       }
-      onMount={handleEditorDidMount}
+      onMount={onMount}
       onChange={handleEditorChange}
       options={{ readOnly: isReadOnly }}
     />
diff --git a/frontend/src/routes/_oh.app._index/route.tsx b/frontend/src/routes/_oh.app._index/route.tsx
index ba20e003f797..6ef5f5762ae8 100644
--- a/frontend/src/routes/_oh.app._index/route.tsx
+++ b/frontend/src/routes/_oh.app._index/route.tsx
@@ -1,12 +1,13 @@
 import React from "react";
 import { useSelector } from "react-redux";
-import { json, useLoaderData, useRouteError } from "@remix-run/react";
+import { json, useRouteError } from "@remix-run/react";
 import toast from "react-hot-toast";
+import { editor } from "monaco-editor";
+import { EditorProps } from "@monaco-editor/react";
 import { RootState } from "#/store";
 import AgentState from "#/types/AgentState";
 import FileExplorer from "#/components/file-explorer/FileExplorer";
 import OpenHands from "#/api/open-hands";
-import { useSocket } from "#/context/socket";
 import CodeEditorCompoonent from "./code-editor-component";
 import { useFiles } from "#/context/files";
 import { EditorActions } from "#/components/editor-actions";
@@ -28,8 +29,7 @@ export function ErrorBoundary() {
 }
 
 function CodeEditor() {
-  const { token } = useLoaderData<typeof clientLoader>();
-  const { runtimeActive } = useSocket();
+  const { curAgentState } = useSelector((state: RootState) => state.agent);
   const {
     setPaths,
     selectedPath,
@@ -37,6 +37,27 @@ function CodeEditor() {
     saveFileContent: saveNewFileContent,
     discardChanges,
   } = useFiles();
+  const [fileExplorerIsOpen, setFileExplorerIsOpen] = React.useState(true);
+  const editorRef = React.useRef<editor.IStandaloneCodeEditor | null>(null);
+
+  const toggleFileExplorer = () => {
+    setFileExplorerIsOpen((prev) => !prev);
+    editorRef.current?.layout({ width: 0, height: 0 });
+  };
+
+  const handleEditorDidMount: EditorProps["onMount"] = (e, monaco) => {
+    editorRef.current = e;
+
+    monaco.editor.defineTheme("oh-dark", {
+      base: "vs-dark",
+      inherit: true,
+      rules: [],
+      colors: {
+        "editor.background": "#171717",
+      },
+    });
+    monaco.editor.setTheme("oh-dark");
+  };
 
   const [errors, setErrors] = React.useState<{ getFiles: string | null }>({
     getFiles: null,
@@ -47,15 +68,14 @@ function CodeEditor() {
   );
 
   React.useEffect(() => {
-    // only retrieve files if connected to WS to prevent requesting before runtime is ready
-    if (runtimeActive && token) {
-      OpenHands.getFiles(token)
+    if (curAgentState === AgentState.INIT) {
+      OpenHands.getFiles()
         .then(setPaths)
         .catch(() => {
           setErrors({ getFiles: "Failed to retrieve files" });
         });
     }
-  }, [runtimeActive, token]);
+  }, [curAgentState]);
 
   // Code editing is only allowed when the agent is paused, finished, or awaiting user input (server rules)
   const isEditingAllowed = React.useMemo(
@@ -69,9 +89,9 @@ function CodeEditor() {
   const handleSave = async () => {
     if (selectedPath) {
       const content = modifiedFiles[selectedPath];
-      if (content && token) {
+      if (content) {
         try {
-          await OpenHands.saveFile(token, selectedPath, content);
+          await OpenHands.saveFile(selectedPath, content);
           saveNewFileContent(selectedPath);
         } catch (error) {
           toast.error("Failed to save file");
@@ -85,9 +105,13 @@ function CodeEditor() {
   };
 
   return (
-    <div className="flex h-full w-full bg-neutral-900 relative">
-      <FileExplorer error={errors.getFiles} />
-      <div className="flex flex-col min-h-0 w-full">
+    <div className="flex h-full bg-neutral-900 relative">
+      <FileExplorer
+        isOpen={fileExplorerIsOpen}
+        onToggle={toggleFileExplorer}
+        error={errors.getFiles}
+      />
+      <div className="w-full">
         {selectedPath && (
           <div className="flex w-full items-center justify-between self-end p-2">
             <span className="text-sm text-neutral-500">{selectedPath}</span>
@@ -98,9 +122,10 @@ function CodeEditor() {
             />
           </div>
         )}
-        <div className="flex grow items-center justify-center">
-          <CodeEditorCompoonent isReadOnly={!isEditingAllowed} />
-        </div>
+        <CodeEditorCompoonent
+          onMount={handleEditorDidMount}
+          isReadOnly={!isEditingAllowed}
+        />
       </div>
     </div>
   );
diff --git a/frontend/src/routes/_oh.app.tsx b/frontend/src/routes/_oh.app.tsx
index 9af5ad278bb3..50c933b64d9e 100644
--- a/frontend/src/routes/_oh.app.tsx
+++ b/frontend/src/routes/_oh.app.tsx
@@ -72,9 +72,8 @@ const isAgentStateChange = (
 
 export const clientLoader = async () => {
   const ghToken = localStorage.getItem("ghToken");
-
   try {
-    const isAuthed = await userIsAuthenticated(ghToken);
+    const isAuthed = await userIsAuthenticated();
     if (!isAuthed) {
       clearSession();
       return redirect("/");
@@ -185,21 +184,6 @@ function App() {
     if (q) addIntialQueryToChat(q, files);
   }, [settings]);
 
-  const handleError = (message: string) => {
-    const [error, ...rest] = message.split(":");
-    const details = rest.join(":");
-    if (!details) {
-      dispatch(
-        addErrorMessage({
-          error: "An error has occured",
-          message: error,
-        }),
-      );
-    } else {
-      dispatch(addErrorMessage({ error, message: details }));
-    }
-  };
-
   const handleMessage = React.useCallback(
     (message: MessageEvent<WebSocket.Data>) => {
       // set token received from the server
@@ -225,7 +209,12 @@ function App() {
         return;
       }
       if (isErrorObservation(parsed)) {
-        handleError(parsed.message);
+        dispatch(
+          addErrorMessage({
+            id: parsed.extras?.error_id,
+            message: parsed.message,
+          }),
+        );
         return;
       }
 
@@ -290,21 +279,21 @@ function App() {
 
   React.useEffect(() => {
     (async () => {
-      if (runtimeActive && token && importedProjectZip) {
+      if (runtimeActive && importedProjectZip) {
         // upload files action
         try {
           const blob = base64ToBlob(importedProjectZip);
           const file = new File([blob], "imported-project.zip", {
             type: blob.type,
           });
-          await OpenHands.uploadFiles(token, [file]);
+          await OpenHands.uploadFiles([file]);
           dispatch(setImportedProjectZip(null));
         } catch (error) {
           toast.error("Failed to upload project files.");
         }
       }
     })();
-  }, [runtimeActive, token, importedProjectZip]);
+  }, [runtimeActive, importedProjectZip]);
 
   const {
     isOpen: securityModalIsOpen,
@@ -315,7 +304,7 @@ function App() {
   return (
     <div className="flex flex-col h-full gap-3">
       <div className="flex h-full overflow-auto gap-3">
-        <Container className="w-[375px] max-h-full">
+        <Container className="w-[390px] max-h-full">
           <ChatInterface />
         </Container>
 
diff --git a/frontend/src/routes/_oh.tsx b/frontend/src/routes/_oh.tsx
index b4b1b35cb098..b6594f7e73bf 100644
--- a/frontend/src/routes/_oh.tsx
+++ b/frontend/src/routes/_oh.tsx
@@ -10,6 +10,8 @@ import {
   Outlet,
   ClientLoaderFunctionArgs,
 } from "@remix-run/react";
+import posthog from "posthog-js";
+import { useDispatch } from "react-redux";
 import { retrieveGitHubUser, isGitHubErrorReponse } from "#/api/github";
 import OpenHands from "#/api/open-hands";
 import CogTooth from "#/assets/cog-tooth";
@@ -28,6 +30,9 @@ import DocsIcon from "#/assets/docs.svg?react";
 import { userIsAuthenticated } from "#/utils/user-is-authenticated";
 import { generateGitHubAuthUrl } from "#/utils/generate-github-auth-url";
 import { WaitlistModal } from "#/components/waitlist-modal";
+import { AnalyticsConsentFormModal } from "#/components/analytics-consent-form-modal";
+import { setCurrentAgentState } from "#/state/agentSlice";
+import AgentState from "#/types/AgentState";
 
 export const clientLoader = async ({ request }: ClientLoaderFunctionArgs) => {
   try {
@@ -41,12 +46,20 @@ export const clientLoader = async ({ request }: ClientLoaderFunctionArgs) => {
 
   let token = localStorage.getItem("token");
   const ghToken = localStorage.getItem("ghToken");
+  const analyticsConsent = localStorage.getItem("analytics-consent");
+  const userConsents = analyticsConsent === "true";
 
-  let isAuthed: boolean = false;
+  if (!userConsents) {
+    posthog.opt_out_capturing();
+  } else {
+    posthog.opt_in_capturing();
+  }
+
+  let isAuthed = false;
   let githubAuthUrl: string | null = null;
 
   try {
-    isAuthed = await userIsAuthenticated(ghToken);
+    isAuthed = await userIsAuthenticated();
     if (!isAuthed && window.__GITHUB_CLIENT_ID__) {
       const requestUrl = new URL(request.url);
       githubAuthUrl = generateGitHubAuthUrl(
@@ -79,6 +92,7 @@ export const clientLoader = async ({ request }: ClientLoaderFunctionArgs) => {
     user,
     settingsIsUpdated,
     settings,
+    analyticsConsent,
   });
 };
 
@@ -132,9 +146,11 @@ export default function MainApp() {
     githubAuthUrl,
     settingsIsUpdated,
     settings,
+    analyticsConsent,
   } = useLoaderData<typeof clientLoader>();
   const logoutFetcher = useFetcher({ key: "logout" });
   const endSessionFetcher = useFetcher({ key: "end-session" });
+  const dispatch = useDispatch();
 
   const [accountSettingsModalOpen, setAccountSettingsModalOpen] =
     React.useState(false);
@@ -204,6 +220,7 @@ export default function MainApp() {
 
   const handleEndSession = () => {
     setStartNewProjectModalIsOpen(false);
+    dispatch(setCurrentAgentState(AgentState.LOADING));
     // call new session action and redirect to '/'
     endSessionFetcher.submit(new FormData(), {
       method: "POST",
@@ -212,7 +229,10 @@ export default function MainApp() {
   };
 
   return (
-    <div className="bg-root-primary p-3 h-screen min-w-[1024px] overflow-x-hidden flex gap-3">
+    <div
+      data-testid="root-layout"
+      className="bg-root-primary p-3 h-screen min-w-[1024px] overflow-x-hidden flex gap-3"
+    >
       <aside className="px-1 flex flex-col gap-1">
         <div className="w-[34px] h-[34px] flex items-center justify-center">
           {navigation.state === "loading" && <LoadingSpinner size="small" />}
@@ -304,6 +324,7 @@ export default function MainApp() {
             onClose={handleAccountSettingsModalClose}
             selectedLanguage={settings.LANGUAGE}
             gitHubError={isGitHubErrorReponse(user)}
+            analyticsConsent={analyticsConsent}
           />
         </ModalBackdrop>
       )}
@@ -328,6 +349,7 @@ export default function MainApp() {
       {!isAuthed && (
         <WaitlistModal ghToken={ghToken} githubAuthUrl={githubAuthUrl} />
       )}
+      {!analyticsConsent && <AnalyticsConsentFormModal />}
     </div>
   );
 }
diff --git a/frontend/src/routes/oauth.github.callback.tsx b/frontend/src/routes/oauth.github.callback.tsx
index 582984c708bb..b5bdca37e910 100644
--- a/frontend/src/routes/oauth.github.callback.tsx
+++ b/frontend/src/routes/oauth.github.callback.tsx
@@ -11,11 +11,11 @@ export const clientLoader = async ({ request }: ClientLoaderFunctionArgs) => {
   const code = url.searchParams.get("code");
 
   if (code) {
-    // request to the server to exchange the code for a token
     const { access_token: accessToken } =
       await OpenHands.getGitHubAccessToken(code);
-    // set the token in local storage
+
     localStorage.setItem("ghToken", accessToken);
+
     return redirect("/");
   }
 
diff --git a/frontend/src/routes/set-consent.ts b/frontend/src/routes/set-consent.ts
new file mode 100644
index 000000000000..1f190ad5a76d
--- /dev/null
+++ b/frontend/src/routes/set-consent.ts
@@ -0,0 +1,9 @@
+import { ClientActionFunctionArgs, json } from "@remix-run/react";
+
+export const clientAction = async ({ request }: ClientActionFunctionArgs) => {
+  const formData = await request.formData();
+  const userConsents = formData.get("analytics") === "on";
+  localStorage.setItem("analytics-consent", userConsents.toString());
+
+  return json(null);
+};
diff --git a/frontend/src/routes/settings.ts b/frontend/src/routes/settings.ts
index 92cb3d7e58f0..30a9167ca743 100644
--- a/frontend/src/routes/settings.ts
+++ b/frontend/src/routes/settings.ts
@@ -28,6 +28,9 @@ export const clientAction = async ({ request }: ClientActionFunctionArgs) => {
     const LANGUAGE = formData.get("language")?.toString();
     if (LANGUAGE) saveSettings({ LANGUAGE });
 
+    const ANALYTICS = formData.get("analytics")?.toString() ?? "false";
+    localStorage.setItem("analytics-consent", ANALYTICS);
+
     return json({ success: true });
   }
 
diff --git a/frontend/src/services/actions.ts b/frontend/src/services/actions.ts
index 46b6aad85130..ccdff694e877 100644
--- a/frontend/src/services/actions.ts
+++ b/frontend/src/services/actions.ts
@@ -1,4 +1,8 @@
-import { addAssistantMessage, addUserMessage } from "#/state/chatSlice";
+import {
+  addAssistantMessage,
+  addUserMessage,
+  addErrorMessage,
+} from "#/state/chatSlice";
 import { setCode, setActiveFilepath } from "#/state/codeSlice";
 import { appendJupyterInput } from "#/state/jupyterSlice";
 import {
@@ -119,13 +123,19 @@ export function handleActionMessage(message: ActionMessage) {
 }
 
 export function handleStatusMessage(message: StatusMessage) {
-  const msg = message.status == null ? "" : message.status.trim();
-  store.dispatch(
-    setCurStatusMessage({
-      ...message,
-      status: msg,
-    }),
-  );
+  if (message.type === "info") {
+    store.dispatch(
+      setCurStatusMessage({
+        ...message,
+      }),
+    );
+  } else if (message.type === "error") {
+    store.dispatch(
+      addErrorMessage({
+        ...message,
+      }),
+    );
+  }
 }
 
 export function handleAssistantMessage(data: string | SocketMessage) {
@@ -139,9 +149,11 @@ export function handleAssistantMessage(data: string | SocketMessage) {
 
   if ("action" in socketMessage) {
     handleActionMessage(socketMessage);
-  } else if ("status" in socketMessage) {
+  } else if ("observation" in socketMessage) {
+    handleObservationMessage(socketMessage);
+  } else if ("status_update" in socketMessage) {
     handleStatusMessage(socketMessage);
   } else {
-    handleObservationMessage(socketMessage);
+    console.error("Unknown message type", socketMessage);
   }
 }
diff --git a/frontend/src/services/api.ts b/frontend/src/services/api.ts
index aca92986c341..ab6006b6ab7f 100644
--- a/frontend/src/services/api.ts
+++ b/frontend/src/services/api.ts
@@ -1,14 +1,26 @@
-import { getToken } from "./auth";
+import { getToken, getGitHubToken } from "./auth";
 import toast from "#/utils/toast";
 
 const WAIT_FOR_AUTH_DELAY_MS = 500;
 
+const UNAUTHED_ROUTE_PREFIXES = [
+  "/api/authenticate",
+  "/api/options/",
+  "/config.json",
+  "/api/github/callback",
+];
+
 export async function request(
   url: string,
   options: RequestInit = {},
   disableToast: boolean = false,
+  returnResponse: boolean = false,
+  maxRetries: number = 3,
   /* eslint-disable-next-line @typescript-eslint/no-explicit-any */
 ): Promise<any> {
+  if (maxRetries < 0) {
+    throw new Error("Max retries exceeded");
+  }
   const onFail = (msg: string) => {
     if (!disableToast) {
       toast.error("api", msg);
@@ -16,12 +28,17 @@ export async function request(
     throw new Error(msg);
   };
 
-  const needsAuth = !url.startsWith("/api/options/");
+  const needsAuth = !UNAUTHED_ROUTE_PREFIXES.some((prefix) =>
+    url.startsWith(prefix),
+  );
   const token = getToken();
+  const githubToken = getGitHubToken();
   if (!token && needsAuth) {
     return new Promise((resolve) => {
       setTimeout(() => {
-        resolve(request(url, options, disableToast));
+        resolve(
+          request(url, options, disableToast, returnResponse, maxRetries - 1),
+        );
       }, WAIT_FOR_AUTH_DELAY_MS);
     });
   }
@@ -32,6 +49,13 @@ export async function request(
       Authorization: `Bearer ${token}`,
     };
   }
+  if (githubToken) {
+    // eslint-disable-next-line no-param-reassign
+    options.headers = {
+      ...(options.headers || {}),
+      "X-GitHub-Token": githubToken,
+    };
+  }
 
   let response = null;
   try {
@@ -48,6 +72,10 @@ export async function request(
     onFail(`Error fetching ${url}: ${response?.statusText}`);
   }
 
+  if (returnResponse) {
+    return response;
+  }
+
   try {
     return await (response && response.json());
   } catch (e) {
diff --git a/frontend/src/services/auth.ts b/frontend/src/services/auth.ts
index a7d8cfa490b9..b1bb9ef90285 100644
--- a/frontend/src/services/auth.ts
+++ b/frontend/src/services/auth.ts
@@ -1,4 +1,5 @@
 const TOKEN_KEY = "token";
+const GITHUB_TOKEN_KEY = "ghToken";
 
 const getToken = (): string => localStorage.getItem(TOKEN_KEY) ?? "";
 
@@ -10,4 +11,22 @@ const setToken = (token: string): void => {
   localStorage.setItem(TOKEN_KEY, token);
 };
 
-export { getToken, setToken, clearToken };
+const getGitHubToken = (): string =>
+  localStorage.getItem(GITHUB_TOKEN_KEY) ?? "";
+
+const setGitHubToken = (token: string): void => {
+  localStorage.setItem(GITHUB_TOKEN_KEY, token);
+};
+
+const clearGitHubToken = (): void => {
+  localStorage.removeItem(GITHUB_TOKEN_KEY);
+};
+
+export {
+  getToken,
+  setToken,
+  clearToken,
+  getGitHubToken,
+  setGitHubToken,
+  clearGitHubToken,
+};
diff --git a/frontend/src/state/chatSlice.ts b/frontend/src/state/chatSlice.ts
index 46f156ebddbd..7d77901fee71 100644
--- a/frontend/src/state/chatSlice.ts
+++ b/frontend/src/state/chatSlice.ts
@@ -39,10 +39,10 @@ export const chatSlice = createSlice({
 
     addErrorMessage(
       state,
-      action: PayloadAction<{ error: string; message: string }>,
+      action: PayloadAction<{ id?: string; message: string }>,
     ) {
-      const { error, message } = action.payload;
-      state.messages.push({ error, message });
+      const { id, message } = action.payload;
+      state.messages.push({ id, message, error: true });
     },
 
     clearMessages(state) {
diff --git a/frontend/src/state/statusSlice.ts b/frontend/src/state/statusSlice.ts
index b0b503d6c6b4..6f5158c9f6d1 100644
--- a/frontend/src/state/statusSlice.ts
+++ b/frontend/src/state/statusSlice.ts
@@ -2,8 +2,10 @@ import { createSlice, PayloadAction } from "@reduxjs/toolkit";
 import { StatusMessage } from "#/types/Message";
 
 const initialStatusMessage: StatusMessage = {
-  status: "",
-  is_error: false,
+  status_update: true,
+  type: "info",
+  id: "",
+  message: "",
 };
 
 export const statusSlice = createSlice({
diff --git a/frontend/src/types/Message.tsx b/frontend/src/types/Message.tsx
index d4d365d5904b..85b1d970641b 100644
--- a/frontend/src/types/Message.tsx
+++ b/frontend/src/types/Message.tsx
@@ -33,10 +33,8 @@ export interface ObservationMessage {
 }
 
 export interface StatusMessage {
-  // TODO not implemented yet
-  // Whether the status is an error, default is false
-  is_error: boolean;
-
-  // A status message to display to the user
-  status: string;
+  status_update: true;
+  type: string;
+  id: string;
+  message: string;
 }
diff --git a/frontend/src/types/core/observations.ts b/frontend/src/types/core/observations.ts
index 9de2a70e8b19..21bafddf21d2 100644
--- a/frontend/src/types/core/observations.ts
+++ b/frontend/src/types/core/observations.ts
@@ -54,6 +54,9 @@ export interface BrowseObservation extends OpenHandsObservationEvent<"browse"> {
 
 export interface ErrorObservation extends OpenHandsObservationEvent<"error"> {
   source: "user";
+  extras: {
+    error_id?: string;
+  };
 }
 
 export type OpenHandsObservation =
diff --git a/frontend/src/utils/download-workspace.ts b/frontend/src/utils/download-workspace.ts
index 1bbf30612d34..dc79141d5f5d 100644
--- a/frontend/src/utils/download-workspace.ts
+++ b/frontend/src/utils/download-workspace.ts
@@ -4,12 +4,7 @@ import OpenHands from "#/api/open-hands";
  * Downloads the current workspace as a .zip file.
  */
 export const downloadWorkspace = async () => {
-  const token = localStorage.getItem("token");
-  if (!token) {
-    throw new Error("No token found");
-  }
-
-  const blob = await OpenHands.getWorkspaceZip(token);
+  const blob = await OpenHands.getWorkspaceZip();
 
   const url = URL.createObjectURL(blob);
   const link = document.createElement("a");
diff --git a/frontend/src/utils/get-valid-fallback-host.ts b/frontend/src/utils/get-valid-fallback-host.ts
deleted file mode 100644
index 6b1482520ec6..000000000000
--- a/frontend/src/utils/get-valid-fallback-host.ts
+++ /dev/null
@@ -1,19 +0,0 @@
-/**
- * Get the valid fallback host. Returns the host unless it is localhost, in which case it returns localhost:3000
- * @returns Valid fallback host
- *
- * @example
- * // If the host is localhost (e.g., localhost:5173), it returns localhost:3000
- * const host = getValidFallbackHost(); // localhost:3000
- *
- * // If the host is not localhost, it returns the host
- * const host = getValidFallbackHost(); // sub.example.com
- */
-export const getValidFallbackHost = () => {
-  if (typeof window !== "undefined") {
-    return window.location.host;
-  }
-
-  // Fallback is localhost:3000 because that is the default port for the server
-  return "localhost:3000";
-};
diff --git a/frontend/src/utils/suggestions/non-repo-suggestions.ts b/frontend/src/utils/suggestions/non-repo-suggestions.ts
index da7207dc7847..d9cf04992ac1 100644
--- a/frontend/src/utils/suggestions/non-repo-suggestions.ts
+++ b/frontend/src/utils/suggestions/non-repo-suggestions.ts
@@ -1,6 +1,6 @@
 const KEY_1 = "Build an app to view pull requests";
-const VALUE_1 = `I want to create a React app to view all of the open pull 
-requests that exist on all of my team's github repos. Here 
+const VALUE_1 = `I want to create a React app to view all of the open pull
+requests that exist on all of my team's github repos. Here
 are some details:
 
 1. Please initialize the app using vite and react-ts.
diff --git a/frontend/src/utils/user-is-authenticated.ts b/frontend/src/utils/user-is-authenticated.ts
index 467c03076223..360b9041cf23 100644
--- a/frontend/src/utils/user-is-authenticated.ts
+++ b/frontend/src/utils/user-is-authenticated.ts
@@ -1,16 +1,12 @@
-import { retrieveGitHubUser, isGitHubErrorReponse } from "#/api/github";
 import OpenHands from "#/api/open-hands";
 
-export const userIsAuthenticated = async (ghToken: string | null) => {
-  if (window.__APP_MODE__ !== "saas") return true;
+export const userIsAuthenticated = async () => {
+  if (window.__APP_MODE__ === "oss") return true;
 
-  let user: GitHubUser | GitHubErrorReponse | null = null;
-  if (ghToken) user = await retrieveGitHubUser(ghToken);
-
-  if (user && !isGitHubErrorReponse(user)) {
-    const isAuthed = await OpenHands.isAuthenticated(user.login);
-    return isAuthed;
+  try {
+    await OpenHands.authenticate();
+    return true;
+  } catch (error) {
+    return false;
   }
-
-  return false;
 };
diff --git a/frontend/test-utils.tsx b/frontend/test-utils.tsx
index 669403b68427..b88ee1063bbb 100644
--- a/frontend/test-utils.tsx
+++ b/frontend/test-utils.tsx
@@ -6,6 +6,7 @@ import { configureStore } from "@reduxjs/toolkit";
 // eslint-disable-next-line import/no-extraneous-dependencies
 import { RenderOptions, render } from "@testing-library/react";
 import { AppStore, RootState, rootReducer } from "./src/store";
+import { SocketProvider } from "#/context/socket";
 
 const setupStore = (preloadedState?: Partial<RootState>): AppStore =>
   configureStore({
@@ -32,7 +33,11 @@ export function renderWithProviders(
   }: ExtendedRenderOptions = {},
 ) {
   function Wrapper({ children }: PropsWithChildren<object>): JSX.Element {
-    return <Provider store={store}>{children}</Provider>;
+    return (
+      <Provider store={store}>
+        <SocketProvider>{children}</SocketProvider>
+      </Provider>
+    );
   }
   return { store, ...render(ui, { wrapper: Wrapper, ...renderOptions }) };
 }
diff --git a/frontend/tests/fixtures/project.zip b/frontend/tests/fixtures/project.zip
new file mode 100644
index 000000000000..e69de29bb2d1
diff --git a/frontend/tests/redirect.spec.ts b/frontend/tests/redirect.spec.ts
new file mode 100644
index 000000000000..7c6090509455
--- /dev/null
+++ b/frontend/tests/redirect.spec.ts
@@ -0,0 +1,61 @@
+import { expect, Page, test } from "@playwright/test";
+import path from "path";
+import { fileURLToPath } from "url";
+
+const filename = fileURLToPath(import.meta.url);
+const dirname = path.dirname(filename);
+
+const confirmSettings = async (page: Page) => {
+  const confirmPreferenceButton = page.getByRole("button", {
+    name: /confirm preferences/i,
+  });
+  await confirmPreferenceButton.click();
+
+  const configSaveButton = page.getByRole("button", {
+    name: /save/i,
+  });
+  await configSaveButton.click();
+
+  const confirmChanges = page.getByRole("button", {
+    name: /yes, close settings/i,
+  });
+  await confirmChanges.click();
+};
+
+test("should redirect to /app after uploading a project zip", async ({
+  page,
+}) => {
+  await page.goto("/");
+
+  const fileInput = page.getByLabel("Upload a .zip");
+  const filePath = path.join(dirname, "fixtures/project.zip");
+  await fileInput.setInputFiles(filePath);
+
+  await page.waitForURL("/app");
+});
+
+test("should redirect to /app after selecting a repo", async ({ page }) => {
+  await page.goto("/");
+  await confirmSettings(page);
+
+  // enter a github token to view the repositories
+  const connectToGitHubButton = page.getByRole("button", {
+    name: /connect to github/i,
+  });
+  await connectToGitHubButton.click();
+  const tokenInput = page.getByLabel(/github token\*/i);
+  await tokenInput.fill("fake-token");
+
+  const submitButton = page.getByTestId("connect-to-github");
+  await submitButton.click();
+
+  // select a repository
+  const repoDropdown = page.getByLabel(/github repository/i);
+  await repoDropdown.click();
+
+  const repoItem = page.getByTestId("github-repo-item").first();
+  await repoItem.click();
+
+  await page.waitForURL("/app");
+  expect(page.url()).toBe("http://127.0.0.1:3000/app");
+});
diff --git a/frontend/tsconfig.json b/frontend/tsconfig.json
index be5fe7a7cd82..dcd184186d6c 100644
--- a/frontend/tsconfig.json
+++ b/frontend/tsconfig.json
@@ -39,4 +39,4 @@
     // Vite takes care of building everything, not tsc.
     "noEmit": true
   }
-}
\ No newline at end of file
+}
diff --git a/frontend/vite.config.ts b/frontend/vite.config.ts
index d9ed9134f8e3..4c403aca8496 100644
--- a/frontend/vite.config.ts
+++ b/frontend/vite.config.ts
@@ -5,6 +5,7 @@ import { defineConfig, loadEnv } from "vite";
 import viteTsconfigPaths from "vite-tsconfig-paths";
 import svgr from "vite-plugin-svgr";
 import { vitePlugin as remix } from "@remix-run/dev";
+import { configDefaults } from "vitest/config";
 
 export default defineConfig(({ mode }) => {
   const {
@@ -90,6 +91,7 @@ export default defineConfig(({ mode }) => {
     test: {
       environment: "jsdom",
       setupFiles: ["vitest.setup.ts"],
+      exclude: [...configDefaults.exclude, "tests"],
       coverage: {
         reporter: ["text", "json", "html", "lcov", "text-summary"],
         reportsDirectory: "coverage",
diff --git a/openhands/__init__.py b/openhands/__init__.py
index dfd64a744083..4b918466a4a0 100644
--- a/openhands/__init__.py
+++ b/openhands/__init__.py
@@ -7,21 +7,15 @@ def get_version():
     try:
         from importlib.metadata import PackageNotFoundError, version
 
-        try:
-            return version(__package_name__)
-        except PackageNotFoundError:
-            pass
-    except ImportError:
+        return version(__package_name__)
+    except (ImportError, PackageNotFoundError):
         pass
 
     try:
         from pkg_resources import DistributionNotFound, get_distribution
 
-        try:
-            return get_distribution(__package_name__).version
-        except DistributionNotFound:
-            pass
-    except ImportError:
+        return get_distribution(__package_name__).version
+    except (ImportError, DistributionNotFound):
         pass
 
     # Try getting the version from pyproject.toml
diff --git a/openhands/agenthub/__init__.py b/openhands/agenthub/__init__.py
index 1ec266ce7501..489ecc7aaead 100644
--- a/openhands/agenthub/__init__.py
+++ b/openhands/agenthub/__init__.py
@@ -14,9 +14,7 @@
     delegator_agent,
     dummy_agent,
     planner_agent,
-    searcher_agent,
     supervisor_agent,
-    tester_agent,
 )
 
 __all__ = [
@@ -26,9 +24,7 @@
     'delegator_agent',
     'dummy_agent',
     'browsing_agent',
-    'searcher_agent',
     'supervisor_agent',
-    'tester_agent',
 ]
 
 for agent in all_microagents.values():
diff --git a/openhands/agenthub/browsing_agent/browsing_agent.py b/openhands/agenthub/browsing_agent/browsing_agent.py
index 0460506d04f3..822677bab526 100644
--- a/openhands/agenthub/browsing_agent/browsing_agent.py
+++ b/openhands/agenthub/browsing_agent/browsing_agent.py
@@ -150,13 +150,13 @@ def step(self, state: State) -> Action:
         last_obs = None
         last_action = None
 
-        if EVAL_MODE and len(state.history.get_events_as_list()) == 1:
+        if EVAL_MODE and len(state.history) == 1:
             # for webarena and miniwob++ eval, we need to retrieve the initial observation already in browser env
             # initialize and retrieve the first observation by issuing an noop OP
             # For non-benchmark browsing, the browser env starts with a blank page, and the agent is expected to first navigate to desired websites
             return BrowseInteractiveAction(browser_actions='noop()')
 
-        for event in state.history.get_events():
+        for event in state.history:
             if isinstance(event, BrowseInteractiveAction):
                 prev_actions.append(event.browser_actions)
                 last_action = event
diff --git a/openhands/agenthub/codeact_agent/codeact_agent.py b/openhands/agenthub/codeact_agent/codeact_agent.py
index 4dbb1503dcc0..91d04a75ef6a 100644
--- a/openhands/agenthub/codeact_agent/codeact_agent.py
+++ b/openhands/agenthub/codeact_agent/codeact_agent.py
@@ -17,6 +17,7 @@
     Action,
     AgentDelegateAction,
     AgentFinishAction,
+    BrowseInteractiveAction,
     CmdRunAction,
     FileEditAction,
     IPythonRunCellAction,
@@ -24,6 +25,7 @@
 )
 from openhands.events.observation import (
     AgentDelegateObservation,
+    BrowserOutputObservation,
     CmdOutputObservation,
     FileEditObservation,
     IPythonRunCellObservation,
@@ -43,7 +45,7 @@
 
 
 class CodeActAgent(Agent):
-    VERSION = '2.1'
+    VERSION = '2.2'
     """
     The Code Act Agent is a minimalist agent.
     The agent works by passing the model a list of action-observation pairs and prompting the model to take the next step.
@@ -116,11 +118,11 @@ def __init__(
         if self.function_calling_active:
             # Function calling mode
             self.tools = codeact_function_calling.get_tools(
-                codeact_enable_browsing_delegate=self.config.codeact_enable_browsing_delegate,
+                codeact_enable_browsing=self.config.codeact_enable_browsing,
                 codeact_enable_jupyter=self.config.codeact_enable_jupyter,
                 codeact_enable_llm_editor=self.config.codeact_enable_llm_editor,
             )
-            logger.info(
+            logger.debug(
                 f'TOOLS loaded for CodeActAgent: {json.dumps(self.tools, indent=2)}'
             )
             self.system_prompt = codeact_function_calling.SYSTEM_PROMPT
@@ -153,10 +155,10 @@ def get_action_message(
 
         Args:
             action (Action): The action to convert. Can be one of:
-                - AgentDelegateAction: For delegating tasks to other agents
                 - CmdRunAction: For executing bash commands
                 - IPythonRunCellAction: For running IPython code
                 - FileEditAction: For editing files
+                - BrowseInteractiveAction: For browsing the web
                 - AgentFinishAction: For ending the interaction
                 - MessageAction: For sending messages
             pending_tool_call_action_messages (dict[str, Message]): Dictionary mapping response IDs
@@ -180,6 +182,7 @@ def get_action_message(
                 CmdRunAction,
                 IPythonRunCellAction,
                 FileEditAction,
+                BrowseInteractiveAction,
             ),
         ) or (isinstance(action, AgentFinishAction) and action.source == 'agent'):
             if self.function_calling_active:
@@ -196,13 +199,17 @@ def get_action_message(
                 pending_tool_call_action_messages[llm_response.id] = Message(
                     role=assistant_msg.role,
                     # tool call content SHOULD BE a string
-                    content=[TextContent(text=assistant_msg.content)]
+                    content=[TextContent(text=assistant_msg.content or '')]
                     if assistant_msg.content is not None
                     else [],
                     tool_calls=assistant_msg.tool_calls,
                 )
                 return []
             else:
+                assert not isinstance(action, BrowseInteractiveAction), (
+                    'BrowseInteractiveAction is not supported in non-function calling mode. Action: '
+                    + str(action)
+                )
                 content = [TextContent(text=self.action_parser.action_to_str(action))]
                 return [
                     Message(
@@ -212,7 +219,7 @@ def get_action_message(
                 ]
         elif isinstance(action, MessageAction):
             role = 'user' if action.source == 'user' else 'assistant'
-            content = [TextContent(text=action.content)]
+            content = [TextContent(text=action.content or '')]
             if self.llm.vision_is_active() and action.images_urls:
                 content.append(ImageContent(image_urls=action.images_urls))
             return [
@@ -277,6 +284,12 @@ def get_observation_message(
         elif isinstance(obs, FileEditObservation):
             text = obs_prefix + truncate_content(str(obs), max_message_chars)
             message = Message(role='user', content=[TextContent(text=text)])
+        elif isinstance(obs, BrowserOutputObservation):
+            text = obs.get_agent_obs_text()
+            message = Message(
+                role='user',
+                content=[TextContent(text=obs_prefix + text)],
+            )
         elif isinstance(obs, AgentDelegateObservation):
             text = obs_prefix + truncate_content(
                 obs.outputs['content'] if 'content' in obs.outputs else '',
@@ -335,8 +348,8 @@ def step(self, state: State) -> Action:
             return self.pending_actions.popleft()
 
         # if we're done, go back
-        latest_user_message = state.history.get_last_user_message()
-        if latest_user_message and latest_user_message.strip() == '/exit':
+        last_user_message = state.get_last_user_message()
+        if last_user_message and last_user_message.strip() == '/exit':
             return AgentFinishAction()
 
         # prepare what we want to send to the LLM
@@ -346,6 +359,7 @@ def step(self, state: State) -> Action:
         }
         if self.function_calling_active:
             params['tools'] = self.tools
+            params['parallel_tool_calls'] = False
         else:
             params['stop'] = [
                 '</execute_ipython>',
@@ -416,7 +430,7 @@ def _get_messages(self, state: State) -> list[Message]:
 
         pending_tool_call_action_messages: dict[str, Message] = {}
         tool_call_id_to_message: dict[str, Message] = {}
-        events = list(state.history.get_events())
+        events = list(state.history)
         for event in events:
             # create a regular message from an event
             if isinstance(event, Action):
diff --git a/openhands/agenthub/codeact_agent/function_calling.py b/openhands/agenthub/codeact_agent/function_calling.py
index f5519124ac81..770d2c679def 100644
--- a/openhands/agenthub/codeact_agent/function_calling.py
+++ b/openhands/agenthub/codeact_agent/function_calling.py
@@ -5,6 +5,7 @@
 
 import json
 
+from browsergym.core.action.highlevel import HighLevelActionSet
 from litellm import (
     ChatCompletionToolParam,
     ChatCompletionToolParamFunctionChunk,
@@ -16,6 +17,7 @@
     Action,
     AgentDelegateAction,
     AgentFinishAction,
+    BrowseInteractiveAction,
     CmdRunAction,
     FileEditAction,
     IPythonRunCellAction,
@@ -23,9 +25,10 @@
 )
 from openhands.events.tool import ToolCallMetadata
 
-SYSTEM_PROMPT = """You are a helpful assistant that can interact with a computer to solve tasks.
+SYSTEM_PROMPT = """You are OpenHands agent, a helpful AI assistant that can interact with a computer to solve tasks.
 <IMPORTANT>
 * If user provides a path, you should NOT assume it's relative to the current working directory. Instead, you should explore the file system to find the file before working on it.
+* When configuring git credentials, use "openhands" as the user.name and "openhands@all-hands.dev" as the user.email by default, unless explicitly instructed otherwise.
 </IMPORTANT>
 """
 
@@ -272,24 +275,146 @@ def __init__(self):
     ),
 )
 
-_BROWSER_DELEGATION = """Delegate the task to another browsing agent.
-The assistant should delegate the task if it needs to browse the Internet.
+# from browsergym/core/action/highlevel.py
+_browser_action_space = HighLevelActionSet(
+    subsets=['bid', 'nav'],
+    strict=False,  # less strict on the parsing of the actions
+    multiaction=True,  # enable to agent to take multiple actions at once
+)
+
+
+_BROWSER_DESCRIPTION = """Interact with the browser using Python code.
+The following 15 functions are available. Nothing else is supported.
+
+goto(url: str)
+    Description: Navigate to a url.
+    Examples:
+        goto('http://www.example.com')
+
+go_back()
+    Description: Navigate to the previous page in history.
+    Examples:
+        go_back()
+
+go_forward()
+    Description: Navigate to the next page in history.
+    Examples:
+        go_forward()
+
+noop(wait_ms: float = 1000)
+    Description: Do nothing, and optionally wait for the given time (in milliseconds).
+    You can use this to get the current page content and/or wait for the page to load.
+    Examples:
+        noop()
+
+        noop(500)
+
+scroll(delta_x: float, delta_y: float)
+    Description: Scroll horizontally and vertically. Amounts in pixels, positive for right or down scrolling, negative for left or up scrolling. Dispatches a wheel event.
+    Examples:
+        scroll(0, 200)
+
+        scroll(-50.2, -100.5)
+
+fill(bid: str, value: str)
+    Description: Fill out a form field. It focuses the element and triggers an input event with the entered text. It works for <input>, <textarea> and [contenteditable] elements.
+    Examples:
+        fill('237', 'example value')
+
+        fill('45', 'multi-line\nexample')
+
+        fill('a12', 'example with "quotes"')
+
+select_option(bid: str, options: str | list[str])
+    Description: Select one or multiple options in a <select> element. You can specify option value or label to select. Multiple options can be selected.
+    Examples:
+        select_option('a48', 'blue')
+
+        select_option('c48', ['red', 'green', 'blue'])
+
+click(bid: str, button: Literal['left', 'middle', 'right'] = 'left', modifiers: list[typing.Literal['Alt', 'Control', 'ControlOrMeta', 'Meta', 'Shift']] = [])
+    Description: Click an element.
+    Examples:
+        click('a51')
+
+        click('b22', button='right')
+
+        click('48', button='middle', modifiers=['Shift'])
+
+dblclick(bid: str, button: Literal['left', 'middle', 'right'] = 'left', modifiers: list[typing.Literal['Alt', 'Control', 'ControlOrMeta', 'Meta', 'Shift']] = [])
+    Description: Double click an element.
+    Examples:
+        dblclick('12')
+
+        dblclick('ca42', button='right')
+
+        dblclick('178', button='middle', modifiers=['Shift'])
+
+hover(bid: str)
+    Description: Hover over an element.
+    Examples:
+        hover('b8')
+
+press(bid: str, key_comb: str)
+    Description: Focus the matching element and press a combination of keys. It accepts the logical key names that are emitted in the keyboardEvent.key property of the keyboard events: Backquote, Minus, Equal, Backslash, Backspace, Tab, Delete, Escape, ArrowDown, End, Enter, Home, Insert, PageDown, PageUp, ArrowRight, ArrowUp, F1 - F12, Digit0 - Digit9, KeyA - KeyZ, etc. You can alternatively specify a single character you'd like to produce such as "a" or "#". Following modification shortcuts are also supported: Shift, Control, Alt, Meta, ShiftLeft, ControlOrMeta. ControlOrMeta resolves to Control on Windows and Linux and to Meta on macOS.
+    Examples:
+        press('88', 'Backspace')
+
+        press('a26', 'ControlOrMeta+a')
+
+        press('a61', 'Meta+Shift+t')
+
+focus(bid: str)
+    Description: Focus the matching element.
+    Examples:
+        focus('b455')
+
+clear(bid: str)
+    Description: Clear the input field.
+    Examples:
+        clear('996')
+
+drag_and_drop(from_bid: str, to_bid: str)
+    Description: Perform a drag & drop. Hover the element that will be dragged. Press left mouse button. Move mouse to the element that will receive the drop. Release left mouse button.
+    Examples:
+        drag_and_drop('56', '498')
+
+upload_file(bid: str, file: str | list[str])
+    Description: Click an element and wait for a "filechooser" event, then select one or multiple input files for upload. Relative file paths are resolved relative to the current working directory. An empty list clears the selected files.
+    Examples:
+        upload_file('572', '/home/user/my_receipt.pdf')
+
+        upload_file('63', ['/home/bob/Documents/image.jpg', '/home/bob/Documents/file.zip'])
+
+Multiple actions can be provided at once, but will be executed sequentially without any feedback from the page.
+More than 2-3 actions usually leads to failure or unexpected behavior. Example:
+fill('a12', 'example with "quotes"')
+click('a51')
+click('48', button='middle', modifiers=['Shift'])
 """
 
-BrowserDelegationTool = ChatCompletionToolParam(
+for _, action in _browser_action_space.action_set.items():
+    assert (
+        action.signature in _BROWSER_DESCRIPTION
+    ), f'Browser description mismatch. Please double check if the BrowserGym updated their action space.\n\nAction: {action.signature}'
+    assert (
+        action.description in _BROWSER_DESCRIPTION
+    ), f'Browser description mismatch. Please double check if the BrowserGym updated their action space.\n\nAction: {action.description}'
+
+BrowserTool = ChatCompletionToolParam(
     type='function',
     function=ChatCompletionToolParamFunctionChunk(
-        name='delegate_to_browsing_agent',
-        description=_BROWSER_DELEGATION,
+        name='browser',
+        description=_BROWSER_DESCRIPTION,
         parameters={
             'type': 'object',
             'properties': {
-                'task': {
+                'code': {
                     'type': 'string',
-                    'description': 'The task for the browsing agent to execute. It should include all the necessary context and specify what information the browsing agent should return.',
-                },
+                    'description': 'The Python code that interacts with the browser.',
+                }
             },
-            'required': ['task'],
+            'required': ['code'],
         },
     ),
 )
@@ -357,6 +482,8 @@ def response_to_actions(response: ModelResponse) -> list[Action]:
                     f'TOOL CALL: str_replace_editor -> file_editor with code: {code}'
                 )
                 action = IPythonRunCellAction(code=code, include_extra=False)
+            elif tool_call.function.name == 'browser':
+                action = BrowseInteractiveAction(browser_actions=arguments['code'])
             else:
                 raise RuntimeError(f'Unknown tool call: {tool_call.function.name}')
 
@@ -381,13 +508,13 @@ def response_to_actions(response: ModelResponse) -> list[Action]:
 
 
 def get_tools(
-    codeact_enable_browsing_delegate: bool = False,
+    codeact_enable_browsing: bool = False,
     codeact_enable_llm_editor: bool = False,
     codeact_enable_jupyter: bool = False,
 ) -> list[ChatCompletionToolParam]:
     tools = [CmdRunTool, FinishTool]
-    if codeact_enable_browsing_delegate:
-        tools.append(BrowserDelegationTool)
+    if codeact_enable_browsing:
+        tools.append(BrowserTool)
     if codeact_enable_jupyter:
         tools.append(IPythonTool)
     if codeact_enable_llm_editor:
diff --git a/openhands/agenthub/codeact_swe_agent/codeact_swe_agent.py b/openhands/agenthub/codeact_swe_agent/codeact_swe_agent.py
index 6fc679aec449..7c5b039e8c47 100644
--- a/openhands/agenthub/codeact_swe_agent/codeact_swe_agent.py
+++ b/openhands/agenthub/codeact_swe_agent/codeact_swe_agent.py
@@ -154,8 +154,8 @@ def step(self, state: State) -> Action:
         - AgentFinishAction() - end the interaction
         """
         # if we're done, go back
-        latest_user_message = state.history.get_last_user_message()
-        if latest_user_message and latest_user_message.strip() == '/exit':
+        last_user_message = state.get_last_user_message()
+        if last_user_message and last_user_message.strip() == '/exit':
             return AgentFinishAction()
 
         # prepare what we want to send to the LLM
@@ -176,7 +176,7 @@ def _get_messages(self, state: State) -> list[Message]:
             Message(role='user', content=[TextContent(text=self.in_context_example)]),
         ]
 
-        for event in state.history.get_events():
+        for event in state.history:
             # create a regular message from an event
             if isinstance(event, Action):
                 message = self.get_action_message(event)
diff --git a/openhands/agenthub/delegator_agent/agent.py b/openhands/agenthub/delegator_agent/agent.py
index 29e0030423c7..7cb987c8c3f7 100644
--- a/openhands/agenthub/delegator_agent/agent.py
+++ b/openhands/agenthub/delegator_agent/agent.py
@@ -2,7 +2,7 @@
 from openhands.controller.state.state import State
 from openhands.core.config import AgentConfig
 from openhands.events.action import Action, AgentDelegateAction, AgentFinishAction
-from openhands.events.observation import AgentDelegateObservation
+from openhands.events.observation import AgentDelegateObservation, Observation
 from openhands.llm.llm import LLM
 
 
@@ -41,7 +41,11 @@ def step(self, state: State) -> Action:
             )
 
         # last observation in history should be from the delegate
-        last_observation = state.history.get_last_observation()
+        last_observation = None
+        for event in reversed(state.history):
+            if isinstance(event, Observation):
+                last_observation = event
+                break
 
         if not isinstance(last_observation, AgentDelegateObservation):
             raise Exception('Last observation is not an AgentDelegateObservation')
diff --git a/openhands/agenthub/dummy_agent/agent.py b/openhands/agenthub/dummy_agent/agent.py
index dbe4c60cfafa..272e6c935f2e 100644
--- a/openhands/agenthub/dummy_agent/agent.py
+++ b/openhands/agenthub/dummy_agent/agent.py
@@ -164,7 +164,7 @@ def step(self, state: State) -> Action:
 
             if 'observations' in prev_step and prev_step['observations']:
                 expected_observations = prev_step['observations']
-                hist_events = state.history.get_last_events(len(expected_observations))
+                hist_events = state.history[-len(expected_observations) :]
 
                 if len(hist_events) < len(expected_observations):
                     print(
diff --git a/openhands/agenthub/micro/agent.py b/openhands/agenthub/micro/agent.py
index 83225a3245cd..a9b0825afd9d 100644
--- a/openhands/agenthub/micro/agent.py
+++ b/openhands/agenthub/micro/agent.py
@@ -8,10 +8,10 @@
 from openhands.core.message import ImageContent, Message, TextContent
 from openhands.core.utils import json
 from openhands.events.action import Action
+from openhands.events.event import Event
 from openhands.events.serialization.action import action_from_dict
 from openhands.events.serialization.event import event_to_memory
 from openhands.llm.llm import LLM
-from openhands.memory.history import ShortTermHistory
 
 
 def parse_response(orig_response: str) -> Action:
@@ -32,16 +32,14 @@ class MicroAgent(Agent):
     prompt = ''
     agent_definition: dict = {}
 
-    def history_to_json(
-        self, history: ShortTermHistory, max_events: int = 20, **kwargs
-    ):
+    def history_to_json(self, history: list[Event], max_events: int = 20, **kwargs):
         """
         Serialize and simplify history to str format
         """
         processed_history = []
         event_count = 0
 
-        for event in history.get_events(reverse=True):
+        for event in reversed(history):
             if event_count >= max_events:
                 break
             processed_history.append(
diff --git a/openhands/agenthub/planner_agent/prompt.py b/openhands/agenthub/planner_agent/prompt.py
index 017c25bbef05..7b73f4353131 100644
--- a/openhands/agenthub/planner_agent/prompt.py
+++ b/openhands/agenthub/planner_agent/prompt.py
@@ -117,7 +117,7 @@ def get_hint(latest_action_id: str) -> str:
 
 def get_prompt_and_images(
     state: State, max_message_chars: int
-) -> tuple[str, list[str]]:
+) -> tuple[str, list[str] | None]:
     """Gets the prompt for the planner agent.
 
     Formatted with the most recent action-observation pairs, current task, and hint based on last action
@@ -136,7 +136,7 @@ def get_prompt_and_images(
     latest_action: Action = NullAction()
 
     # retrieve the latest HISTORY_SIZE events
-    for event_count, event in enumerate(state.history.get_events(reverse=True)):
+    for event_count, event in enumerate(reversed(state.history)):
         if event_count >= HISTORY_SIZE:
             break
         if latest_action == NullAction() and isinstance(event, Action):
diff --git a/openhands/agenthub/searcher_agent/__init__.py b/openhands/agenthub/searcher_agent/__init__.py
deleted file mode 100644
index 1f4b7d50c642..000000000000
--- a/openhands/agenthub/searcher_agent/__init__.py
+++ /dev/null
@@ -1,4 +0,0 @@
-from openhands.agenthub.searcher_agent.agent import SearcherAgent
-from openhands.controller.agent import Agent
-
-Agent.register('SearcherAgent', SearcherAgent)
diff --git a/openhands/agenthub/searcher_agent/action_parser.py b/openhands/agenthub/searcher_agent/action_parser.py
deleted file mode 100644
index 46846641b8df..000000000000
--- a/openhands/agenthub/searcher_agent/action_parser.py
+++ /dev/null
@@ -1,153 +0,0 @@
-import re
-
-from openhands.controller.action_parser import (
-    ActionParser,
-    ResponseParser,
-)
-from openhands.events.action import (
-    Action,
-    AgentFinishAction,
-    CmdRunAction,
-    IPythonRunCellAction,
-    MessageAction,
-)
-
-
-class SearcherAgentResponseParser(ResponseParser):
-    """Parser action:
-    - CmdRunAction(command) - bash command to run
-    - IPythonRunCellAction(code) - IPython code to run
-    - MessageAction(content) - Message action to run (e.g. ask for clarification)
-    - AgentFinishAction() - end the interaction
-    """
-
-    def __init__(self):
-        # Need pay attention to the item order in self.action_parsers
-        super().__init__()
-        self.action_parsers = [
-            SearcherAgentActionParserFinish(),
-            SearcherAgentActionParserCmdRun(),
-            SearcherAgentActionParserIPythonRunCell(),
-        ]
-        self.default_parser = SearcherAgentActionParserMessage()
-
-    def parse(self, response) -> Action:
-        action_str = self.parse_response(response)
-        return self.parse_action(action_str)
-
-    def parse_response(self, response) -> str:
-        action = response.choices[0].message.content
-        if action is None:
-            return ''
-        for lang in ['bash', 'ipython']:
-            if f'</execute_{lang}' in action and f'</execute_{lang}>' not in action:
-                action = action.replace(f'</execute_{lang}', f'</execute_{lang}>')
-
-            if f'<execute_{lang}>' in action and f'</execute_{lang}>' not in action:
-                action += f'</execute_{lang}>'
-        return action
-
-    def parse_action(self, action_str: str) -> Action:
-        for action_parser in self.action_parsers:
-            if action_parser.check_condition(action_str):
-                return action_parser.parse(action_str)
-        return self.default_parser.parse(action_str)
-
-
-class SearcherAgentActionParserFinish(ActionParser):
-    """Parser action:
-    - AgentFinishAction() - end the interaction
-    """
-
-    def __init__(
-        self,
-    ):
-        self.finish_command = None
-
-    def check_condition(self, action_str: str) -> bool:
-        self.finish_command = re.search(
-            r'<finish>(.*?)</finish>', action_str, re.DOTALL
-        )
-        return self.finish_command is not None
-
-    def parse(self, action_str: str) -> Action:
-        assert (
-            self.finish_command is not None
-        ), 'self.finish_command should not be None when parse is called'
-        output = self.finish_command.group(1).strip()
-        outputs = {'output': output}
-        return AgentFinishAction(outputs=outputs)
-
-
-class SearcherAgentActionParserCmdRun(ActionParser):
-    """Parser action:
-    - CmdRunAction(command) - bash command to run
-    - AgentFinishAction() - end the interaction
-    """
-
-    def __init__(
-        self,
-    ):
-        self.bash_command = None
-
-    def check_condition(self, action_str: str) -> bool:
-        self.bash_command = re.search(
-            r'<execute_bash>(.*?)</execute_bash>', action_str, re.DOTALL
-        )
-        return self.bash_command is not None
-
-    def parse(self, action_str: str) -> Action:
-        assert (
-            self.bash_command is not None
-        ), 'self.bash_command should not be None when parse is called'
-        thought = action_str.replace(self.bash_command.group(0), '').strip()
-        # a command was found
-        command_group = self.bash_command.group(1).strip()
-        if command_group.strip() == 'exit':
-            return AgentFinishAction(thought=thought)
-        return CmdRunAction(command=command_group, thought=thought)
-
-
-class SearcherAgentActionParserIPythonRunCell(ActionParser):
-    """Parser action:
-    - IPythonRunCellAction(code) - IPython code to run
-    """
-
-    def __init__(self):
-        self.python_code = None
-        self.jupyter_kernel_init_code: str = 'from agentskills import *'
-
-    def check_condition(self, action_str: str) -> bool:
-        self.python_code = re.search(
-            r'<execute_ipython>(.*?)</execute_ipython>', action_str, re.DOTALL
-        )
-        return self.python_code is not None
-
-    def parse(self, action_str: str) -> Action:
-        assert self.python_code is not None
-        code_group = self.python_code.group(1).strip()
-        thought = action_str.replace(self.python_code.group(0), '').strip()
-        return IPythonRunCellAction(
-            code=code_group,
-            thought=thought,
-            kernel_init_code=self.jupyter_kernel_init_code,
-        )
-
-
-class SearcherAgentActionParserMessage(ActionParser):
-    """Parser action:
-    - MessageAction(content) - Message action to run (e.g. ask for clarification)
-    """
-
-    def __init__(
-        self,
-    ):
-        pass
-
-    def check_condition(self, action_str: str) -> bool:
-        # We assume the LLM is GOOD enough that when it returns pure natural language
-        # it wants to talk to the user
-        return True
-
-    def parse(self, action_str: str) -> Action:
-        return MessageAction(content=action_str, wait_for_response=True)
diff --git a/openhands/agenthub/searcher_agent/agent.py b/openhands/agenthub/searcher_agent/agent.py
deleted file mode 100644
index 195f3823eed7..000000000000
--- a/openhands/agenthub/searcher_agent/agent.py
+++ /dev/null
@@ -1,203 +0,0 @@
-import logging
-
-from openhands.agenthub.searcher_agent.action_parser import SearcherAgentResponseParser
-from openhands.agenthub.searcher_agent.prompt import get_prompt
-from openhands.controller.agent import Agent
-from openhands.controller.state.state import State
-from openhands.core.config import AgentConfig
-from openhands.core.config.llm_config import LLMConfig
-from openhands.core.message import Message, TextContent
-from openhands.events.action import Action, AgentFinishAction, IPythonRunCellAction
-from openhands.events.action.commands import CmdRunAction
-from openhands.events.action.message import MessageAction
-from openhands.events.observation import IPythonRunCellObservation
-from openhands.events.observation.commands import CmdOutputObservation
-from openhands.events.observation.error import ErrorObservation
-from openhands.events.observation.observation import Observation
-from openhands.events.observation.reject import UserRejectObservation
-from openhands.llm.llm import LLM
-from openhands.runtime.plugins.agent_skills import AgentSkillsRequirement
-from openhands.runtime.plugins.jupyter import JupyterRequirement
-from openhands.runtime.plugins.requirement import PluginRequirement
-
-
-# WIP: Make this agent be able to detect when to stop and automatically stop (or make the supervisor able to stop the agent).
-class SearcherAgent(Agent):
-    VERSION = '1.0'
-    """
-    The Searcher Agent is an agent that searches the codebase for relevant information.
-    """
-
-    sandbox_plugins: list[PluginRequirement] = [
-        # NOTE: AgentSkillsRequirement need to go before JupyterRequirement, since
-        # AgentSkillsRequirement provides a lot of Python functions,
-        # and it needs to be initialized before Jupyter for Jupyter to use those functions.
-        AgentSkillsRequirement(),
-        JupyterRequirement(),
-    ]
-
-    action_parser = SearcherAgentResponseParser()
-
-    def __init__(self, llm: LLM, config: AgentConfig):
-        """Initialize the Searcher Agent with an LLM
-
-        Parameters:
-        - llm (LLM): The llm to be used by this agent
-        - config (AgentConfig): The configuration for this agent
-        """
-        # TODO: Remove this once we have a real LLM config
-        llm_config = LLMConfig(
-            model='deepseek/deepseek-chat', api_key='REDACTED', temperature=0.0
-        )
-        llm = LLM(llm_config)
-        # TODO: Remove this once we have a real AgentConfig
-        config = AgentConfig(llm_config='deepseek')
-        super().__init__(llm, config)
-        # Set up logger
-        self.logger = logging.getLogger(__name__)
-        logging.basicConfig(level=logging.DEBUG)  # Set the logging level
-
-    def get_action_message(self, action: Action) -> Message | None:
-        """Convert an Action to a Message for the LLM conversation.
-
-        Parameters:
-        - action (Action): The action to convert
-
-        Returns:
-        - Message | None: The converted message, or None if action type is not supported
-        """
-        if isinstance(action, CmdRunAction):
-            return Message(
-                role='assistant',
-                content=[
-                    TextContent(
-                        text=f'{action.thought}\n<execute_bash>\n{action.command}\n</execute_bash>'
-                    )
-                ],
-            )
-        elif isinstance(action, IPythonRunCellAction):
-            return Message(
-                role='assistant',
-                content=[
-                    TextContent(
-                        text=f'{action.thought}\n<execute_ipython>\n{action.code}\n</execute_ipython>'
-                    )
-                ],
-            )
-        elif isinstance(action, MessageAction):
-            return Message(
-                role='user' if action.source == 'user' else 'assistant',
-                content=[TextContent(text=action.content)],
-            )
-        elif isinstance(action, AgentFinishAction) and action.source == 'agent':
-            return Message(role='assistant', content=[TextContent(text=action.thought)])
-        return None
-
-    def get_observation_message(self, obs: Observation) -> Message | None:
-        """Convert an Observation to a Message for the LLM conversation.
-
-        Parameters:
-        - obs (Observation): The observation to convert
-
-        Returns:
-        - Message | None: The converted message, or None if observation type is not supported
-        """
-        obs_prefix = 'OBSERVATION:\n'
-        if isinstance(obs, CmdOutputObservation):
-            text = obs_prefix + obs.content
-            text += (
-                f'\n[Command {obs.command_id} finished with exit code {obs.exit_code}]'
-            )
-            return Message(role='user', content=[TextContent(text=text)])
-        elif isinstance(obs, IPythonRunCellObservation):
-            text = obs_prefix + obs.content
-            splitted = text.split('\n')
-            for i, line in enumerate(splitted):
-                if '![image](data:image/png;base64,' in line:
-                    splitted[i] = (
-                        '![image](data:image/png;base64, ...) already displayed to user'
-                    )
-            text = '\n'.join(splitted)
-            return Message(role='user', content=[TextContent(text=text)])
-        elif isinstance(obs, ErrorObservation):
-            text = obs_prefix + obs.content
-            text += '\n[Error occurred in processing last action]'
-            return Message(role='user', content=[TextContent(text=text)])
-        elif isinstance(obs, UserRejectObservation):
-            text = obs_prefix + obs.content
-            text += '\n[Last action has been rejected by the user]'
-            return Message(role='user', content=[TextContent(text=text)])
-        else:
-            raise ValueError(f'Unknown observation type: {type(obs)}')
-
-    def step(self, state: State) -> Action:
-        """Performs one step using the SearcherAgent.
-        This includes gathering info on previous steps and prompting the model to make a command to execute.
-
-        Parameters:
-        - state (State): used to get updated info
-
-        Returns:
-        - CmdRunAction(command) - bash command to run
-        - IPythonRunCellAction(code) - IPython code to run
-        - MessageAction(content) - Message action to run (e.g. ask for clarification)
-        - AgentFinishAction() - end the interaction
-        """
-
-        # prepare what we want to send to the LLM
-        messages = self._get_messages(state)
-        params = {
-            'messages': self.llm.format_messages_for_llm(messages),
-            'stop': [
-                '</execute_bash>',
-                '</execute_ipython>',
-            ],
-        }
-
-        response = self.llm.completion(**params)
-
-        return self.action_parser.parse(response)
-
-    def _get_messages(self, state: State) -> list[Message]:
-        # Get task and suggested approach from state inputs
-        task = state.inputs.get('task', '')
-        suggested_approach = state.inputs.get('suggested_approach', '')
-
-        messages: list[Message] = [
-            Message(
-                role='system',
-                content=[
-                    TextContent(
-                        text=get_prompt(task, suggested_approach),
-                        cache_prompt=self.llm.is_caching_prompt_active(),
-                    )
-                ],
-            ),
-        ]
-
-        for event in state.history.get_events():
-            # create message from event
-            if isinstance(event, Action):
-                message = self.get_action_message(event)
-            elif isinstance(event, Observation):
-                message = self.get_observation_message(event)
-            else:
-                raise ValueError(f'Unknown event type: {type(event)}')
-
-            # add regular message
-            if message:
-                # handle error if the message is the SAME role as the previous message
-                if messages and messages[-1].role == message.role:
-                    messages[-1].content.extend(message.content)
-                else:
-                    messages.append(message)
-
-        # Add caching to the last 2 user messages
-        if self.llm.is_caching_prompt_active():
-            user_turns_processed = 0
-            for message in reversed(messages):
-                if message.role == 'user' and user_turns_processed < 2:
-                    message.content[-1].cache_prompt = True
-                    user_turns_processed += 1
-
-        return messages
diff --git a/openhands/agenthub/searcher_agent/prompt.py b/openhands/agenthub/searcher_agent/prompt.py
deleted file mode 100644
index 6479cda7eab6..000000000000
--- a/openhands/agenthub/searcher_agent/prompt.py
+++ /dev/null
@@ -1,146 +0,0 @@
-# General Description, the goal is to devise a manager that is able to iterate if the solution has not been found yet.
-# In order to successfully fix an issue there are two phases:
-# 1. Exploring the codebase, finding the root cause of the issue.
-# 2. Implementing the solution.
-# Then the manager needs to check if the issue has been fixed, if not, it needs to iterate.
-general_description = """
-The assistant is a detail-oriented AI, an expert in searching through files and code.
-The assistant is also an expert in summarising code and its purpose.
-As a detail-oriented AI, you MUST always read more and more code until you are sure you have found
-all the information you need.
-
-The assistant's goal is to gather information about the codebase to help the programmer fix the issue.
-Here is the task you are trying to complete:
-%(task)s
-
-IMPORTANT: THE ASSISTANT SHOULD NEVER TRY TO IMPLEMENT A SOLUTION. THE ASSISTANTR ONLY GOAL IS TO GATHER INFORMATION.
-As an expert in searching through files and code, you have been equipped with a set of tools
-that will help you gather information about the codebase:
-- The assistant can execute bash commands wrapped with <execute_bash>, e.g. <execute_bash> ls </execute_bash>.
-- If a bash command returns exit code `-1`, this means the process is not yet finished.
-- The assistant must then send a second <execute_bash>. The second <execute_bash> can be empty
-  (which will retrieve any additional logs), or it can contain text to be sent to STDIN of the running process,
-  or it can contain the text `ctrl+c` to interrupt the process.
-- For commands that may run indefinitely, the output should be redirected to a file and the command run
-  in the background, e.g. <execute_bash> python3 app.py > server.log 2>&1 & </execute_bash>
-- If a command execution result says "Command timed out. Sending SIGINT to the process",
-  you should retry running the command in the background.
-
-The assistant should ONLY `run` commands that have no side-effects, like `ls` and `grep`.
-
-The assistant can use a Python environment with <execute_ipython>, e.g.:
-<execute_ipython>
-print("Hello World!")
-</execute_ipython>
-
-The assistant can install Python packages using the %%pip magic command in an IPython environment by using the following syntax: <execute_ipython> %%pip install [package needed] </execute_ipython> and should always import packages and define variables before starting to use them.
-
-Apart from the standard Python library, the assistant can also use the following functions (already imported) in <execute_ipython> environment:
-open_file(path: str, line_number: int | None = 1, context_lines: int | None = 100) -> None:
-    Opens the file at the given path in the editor. IF the file is to be edited, first use `scroll_down` repeatedly to read the full file!
-    If line_number is provided, the window will be moved to include that line.
-    It only shows the first 100 lines by default! `context_lines` is the max number of lines to be displayed, up to 100. Use `scroll_up` and `scroll_down` to view more content up or down.
-    Args:
-    path: str: The path to the file to open, preferred absolute path.
-    line_number: int | None = 1: The line number to move to. Defaults to 1.
-    context_lines: int | None = 100: Only shows this number of lines in the context window (usually from line 1), with line_number as the center (if possible). Defaults to 100.
-
-goto_line(line_number: int) -> None:
-    Moves the window to show the specified line number.
-    Args:
-    line_number: int: The line number to move to.
-
-scroll_down() -> None:
-    Moves the window down by 100 lines.
-    Args:
-    None
-
-scroll_up() -> None:
-    Moves the window up by 100 lines.
-    Args:
-    None
-
-search_dir(search_term: str, dir_path: str = './') -> None:
-    Searches for search_term in all files in dir. If dir is not provided, searches in the current directory.
-    Args:
-    search_term: str: The term to search for.
-    dir_path: str: The path to the directory to search.
-
-search_file(search_term: str, file_path: str | None = None) -> None:
-    Searches for search_term in file. If file is not provided, searches in the current open file.
-    Args:
-    search_term: str: The term to search for.
-    file_path: str | None: The path to the file to search.
-
-find_file(file_name: str, dir_path: str = './') -> None:
-    Finds all files with the given name in the specified directory.
-    Args:
-    file_name: str: The name of the file to find.
-    dir_path: str: The path to the directory to search.
-
-parse_pdf(file_path: str) -> None:
-    Parses the content of a PDF file and prints it.
-    Args:
-    file_path: str: The path to the file to open.
-
-parse_docx(file_path: str) -> None:
-    Parses the content of a DOCX file and prints it.
-    Args:
-    file_path: str: The path to the file to open.
-
-parse_latex(file_path: str) -> None:
-    Parses the content of a LaTex file and prints it.
-    Args:
-    file_path: str: The path to the file to open.
-
-parse_pptx(file_path: str) -> None:
-    Parses the content of a pptx file and prints it.
-    Args:
-    file_path: str: The path to the file to open.
-
-
-IMPORTANT:
-- `open_file` only returns the first 100 lines of the file by default! The assistant MUST use `scroll_down` repeatedly to read the full file BEFORE making edits!
-- Indentation is important and code that is not indented correctly will fail and require fixing before it can be run.
-- Any code issued should be less than 50 lines to avoid context being cut off!
-
-The assistant's manager gave you a suggested approach that you should follow:
-%(suggested_approach)s
-
-Follow the suggested approach to gather information about the codebase.
-When you think you have gathered enough information, generate a JSON with the following format:
-<finish>
-[
-  {
-    "summary": "<a detailed summary of a relevant file>",
-    "location_of_the_file": "<path to the file>",
-    "functions_of_interest": [
-      {
-        "name": "<name of the function>",
-        "summary": "<a detailed summary of the function>",
-        "calls_to_this_function": ["<list of functions that call this function>"],
-        "is_called_by_these_functions": ["<list of functions that are called by this function>"]
-      },
-    ]
-  }
-]
-</finish>
-
-IMPORTANT: Every entry in the JSON MUST be relevant to the task.
-IMPORTANT: The JSON MUST be contained inside <finish> and </finish> tags.
-IMPORTANT: The assistant MUST have at least one file in the response.
-IMPORTANT: THE ASSISTANT MUST NOT modify the codebase or NOT ADD any new files.
-"""
-
-
-def get_prompt(task: str, suggested_approach: str) -> str:
-    # Escape any % characters in the input strings
-    formatted_prompt = general_description % {
-        'task': task,
-        'suggested_approach': suggested_approach,
-    }
-
-    # Add instruction to not include json formatting
-    formatted_prompt += '\n\nIMPORTANT: Do not include ```json at the start or ``` at the end of your response. Just return the raw JSON list.'
-
-    return formatted_prompt
diff --git a/openhands/agenthub/supervisor_agent/agent.py b/openhands/agenthub/supervisor_agent/agent.py
index 0c0ba4e83b3d..722d7365cb3a 100644
--- a/openhands/agenthub/supervisor_agent/agent.py
+++ b/openhands/agenthub/supervisor_agent/agent.py
@@ -60,8 +60,7 @@ def step(self, state: State) -> Action:
             self.suggested_approaches = self.get_suggested_approaches(state)
         self.suggested_approach_index += 1
 
-        last_observation = state.history.get_last_observation()
-        # At first the history is empty, so we proceed to the SearchAgent
+        last_observation = state.history[-1] if state.history else None
         if isinstance(last_observation, AgentDelegateObservation):
             self.results[self.phase].append(last_observation.outputs.get('output', ''))
 
@@ -132,7 +131,10 @@ def step(self, state: State) -> Action:
 
     def get_suggested_approaches(self, state: State):
         self.logger.debug('No suggested approaches found, breaking down task.')
-        self.task, _ = state.get_current_user_intent()
+        task, _ = state.get_current_user_intent()
+        if not task:
+            return []
+        self.task = task
         suggested_approaches = self.ask_llm(self.task, 'search')
         self.logger.debug('Suggested approaches: %s', self.suggested_approaches)
         if not suggested_approaches:
diff --git a/openhands/agenthub/tester_agent/__init__.py b/openhands/agenthub/tester_agent/__init__.py
deleted file mode 100644
index 54be665abd42..000000000000
--- a/openhands/agenthub/tester_agent/__init__.py
+++ /dev/null
@@ -1,4 +0,0 @@
-from openhands.agenthub.tester_agent.agent import TesterAgent
-from openhands.controller.agent import Agent
-
-Agent.register('TesterAgent', TesterAgent)
diff --git a/openhands/agenthub/tester_agent/action_parser.py b/openhands/agenthub/tester_agent/action_parser.py
deleted file mode 100644
index 8abc7c353916..000000000000
--- a/openhands/agenthub/tester_agent/action_parser.py
+++ /dev/null
@@ -1,158 +0,0 @@
-import re
-
-from openhands.controller.action_parser import (
-    ActionParser,
-    ResponseParser,
-)
-from openhands.events.action import (
-    Action,
-    AgentFinishAction,
-    CmdRunAction,
-    IPythonRunCellAction,
-    MessageAction,
-)
-
-
-class TesterAgentResponseParser(ResponseParser):
-    """Parser action:
-    - CmdRunAction(command) - bash command to run
-    - IPythonRunCellAction(code) - IPython code to run
-    - MessageAction(content) - Message action to run (e.g. ask for clarification)
-    - AgentFinishAction() - end the interaction
-    """
-
-    def __init__(self):
-        # Need pay attention to the item order in self.action_parsers
-        super().__init__()
-        self.action_parsers = [
-            TesterAgentActionParserFinish(),
-            TesterAgentActionParserCmdRun(),
-            TesterAgentActionParserIPythonRunCell(),
-        ]
-        self.default_parser = TesterAgentActionParserMessage()
-
-    def parse(self, response) -> Action:
-        action_str = self.parse_response(response)
-        return self.parse_action(action_str)
-
-    def parse_response(self, response) -> str:
-        action = response.choices[0].message.content
-        if action is None:
-            return ''
-        for lang in ['bash', 'ipython', 'browse']:
-            # special handling for DeepSeek: it has stop-word bug and returns </execute_ipython instead of </execute_ipython>
-            if f'</execute_{lang}' in action and f'</execute_{lang}>' not in action:
-                action = action.replace(f'</execute_{lang}', f'</execute_{lang}>')
-
-            if f'<execute_{lang}>' in action and f'</execute_{lang}>' not in action:
-                action += f'</execute_{lang}>'
-        if '<file_edit' in action and '</file_edit>' not in action:
-            action += '</file_edit>'
-        return action
-
-    def parse_action(self, action_str: str) -> Action:
-        for action_parser in self.action_parsers:
-            if action_parser.check_condition(action_str):
-                return action_parser.parse(action_str)
-        return self.default_parser.parse(action_str)
-
-
-class TesterAgentActionParserFinish(ActionParser):
-    """Parser action:
-    - AgentFinishAction() - end the interaction
-    """
-
-    def __init__(
-        self,
-    ):
-        self.finish_command = None
-
-    def check_condition(self, action_str: str) -> bool:
-        self.finish_command = re.search(r'<finish>.*</finish>', action_str, re.DOTALL)
-        return self.finish_command is not None
-
-    def parse(self, action_str: str) -> Action:
-        assert (
-            self.finish_command is not None
-        ), 'self.finish_command should not be None when parse is called'
-        output = self.finish_command.group(1).strip()
-        outputs = {'output': output}
-        return AgentFinishAction(outputs=outputs)
-
-
-class TesterAgentActionParserCmdRun(ActionParser):
-    """Parser action:
-    - CmdRunAction(command) - bash command to run
-    - AgentFinishAction() - end the interaction
-    """
-
-    def __init__(
-        self,
-    ):
-        self.bash_command = None
-
-    def check_condition(self, action_str: str) -> bool:
-        self.bash_command = re.search(
-            r'<execute_bash>(.*?)</execute_bash>', action_str, re.DOTALL
-        )
-        return self.bash_command is not None
-
-    def parse(self, action_str: str) -> Action:
-        assert (
-            self.bash_command is not None
-        ), 'self.bash_command should not be None when parse is called'
-        thought = action_str.replace(self.bash_command.group(0), '').strip()
-        # a command was found
-        command_group = self.bash_command.group(1).strip()
-        if command_group.strip() == 'exit':
-            return AgentFinishAction(thought=thought)
-        return CmdRunAction(command=command_group, thought=thought)
-
-
-class TesterAgentActionParserIPythonRunCell(ActionParser):
-    """Parser action:
-    - IPythonRunCellAction(code) - IPython code to run
-    """
-
-    def __init__(
-        self,
-    ):
-        self.python_code = None
-        self.jupyter_kernel_init_code: str = 'from agentskills import *'
-
-    def check_condition(self, action_str: str) -> bool:
-        self.python_code = re.search(
-            r'<execute_ipython>(.*?)</execute_ipython>', action_str, re.DOTALL
-        )
-        return self.python_code is not None
-
-    def parse(self, action_str: str) -> Action:
-        assert (
-            self.python_code is not None
-        ), 'self.python_code should not be None when parse is called'
-        code_group = self.python_code.group(1).strip()
-        thought = action_str.replace(self.python_code.group(0), '').strip()
-        return IPythonRunCellAction(
-            code=code_group,
-            thought=thought,
-            kernel_init_code=self.jupyter_kernel_init_code,
-        )
-
-
-class TesterAgentActionParserMessage(ActionParser):
-    """Parser action:
-    - MessageAction(content) - Message action to run (e.g. ask for clarification)
-    """
-
-    def __init__(
-        self,
-    ):
-        pass
-
-    def check_condition(self, action_str: str) -> bool:
-        # We assume the LLM is GOOD enough that when it returns pure natural language
-        # it wants to talk to the user
-        return True
-
-    def parse(self, action_str: str) -> Action:
-        return MessageAction(content=action_str, wait_for_response=True)
diff --git a/openhands/agenthub/tester_agent/agent.py b/openhands/agenthub/tester_agent/agent.py
deleted file mode 100644
index 3dc449f43430..000000000000
--- a/openhands/agenthub/tester_agent/agent.py
+++ /dev/null
@@ -1,201 +0,0 @@
-import logging
-
-from openhands.agenthub.tester_agent.action_parser import TesterAgentResponseParser
-from openhands.agenthub.tester_agent.prompt import get_prompt
-from openhands.controller.agent import Agent
-from openhands.controller.state.state import State
-from openhands.core.config import AgentConfig
-from openhands.core.config.llm_config import LLMConfig
-from openhands.core.message import Message, TextContent
-from openhands.events.action import Action, AgentFinishAction
-from openhands.events.action.commands import CmdRunAction, IPythonRunCellAction
-from openhands.events.action.message import MessageAction
-from openhands.events.observation.commands import (
-    CmdOutputObservation,
-    IPythonRunCellObservation,
-)
-from openhands.events.observation.error import ErrorObservation
-from openhands.events.observation.observation import Observation
-from openhands.events.observation.reject import UserRejectObservation
-from openhands.llm.llm import LLM
-
-
-class TesterAgent(Agent):
-    VERSION = '1.0'
-    """
-    The Tester Agent is an agent that tries to replicate the issue.
-    """
-
-    action_parser = TesterAgentResponseParser()
-
-    def __init__(self, llm: LLM, config: AgentConfig):
-        """Initialize the Tester Agent with an LLM
-
-        Parameters:
-        - llm (LLM): The llm to be used by this agent
-        - config (AgentConfig): The configuration for this agent
-        """
-        # TODO: Remove this once we have a real LLM config
-        llm_config = LLMConfig(
-            model='deepseek/deepseek-chat', api_key='REDACTED', temperature=0.0
-        )
-        llm = LLM(llm_config)
-        # TODO: Remove this once we have a real AgentConfig
-        config = AgentConfig(llm_config='deepseek')
-        super().__init__(llm, config)
-        # Set up logger
-        self.logger = logging.getLogger(__name__)
-        logging.basicConfig(level=logging.DEBUG)  # Set the logging level
-
-    def get_action_message(self, action: Action) -> Message | None:
-        """Convert an Action to a Message for the LLM conversation.
-
-        Parameters:
-        - action (Action): The action to convert
-
-        Returns:
-        - Message | None: The converted message, or None if action type is not supported
-        """
-        if isinstance(action, CmdRunAction):
-            return Message(
-                role='assistant',
-                content=[
-                    TextContent(
-                        text=f'{action.thought}\n<execute_bash>\n{action.command}\n</execute_bash>'
-                    )
-                ],
-            )
-        elif isinstance(action, IPythonRunCellAction):
-            return Message(
-                role='assistant',
-                content=[
-                    TextContent(
-                        text=f'{action.thought}\n<execute_ipython>\n{action.code}\n</execute_ipython>'
-                    )
-                ],
-            )
-        elif isinstance(action, MessageAction):
-            return Message(
-                role='user' if action.source == 'user' else 'assistant',
-                content=[TextContent(text=action.content)],
-            )
-        elif isinstance(action, AgentFinishAction) and action.source == 'agent':
-            return Message(role='assistant', content=[TextContent(text=action.thought)])
-        return None
-
-    def get_observation_message(self, obs: Observation) -> Message | None:
-        """Convert an Observation to a Message for the LLM conversation.
-
-        Parameters:
-        - obs (Observation): The observation to convert
-
-        Returns:
-        - Message | None: The converted message, or None if observation type is not supported
-        """
-        obs_prefix = 'OBSERVATION:\n'
-        if isinstance(obs, CmdOutputObservation):
-            text = obs_prefix + obs.content
-            text += (
-                f'\n[Command {obs.command_id} finished with exit code {obs.exit_code}]'
-            )
-            return Message(role='user', content=[TextContent(text=text)])
-        elif isinstance(obs, IPythonRunCellObservation):
-            text = obs_prefix + obs.content
-            return Message(role='user', content=[TextContent(text=text)])
-        elif isinstance(obs, ErrorObservation):
-            text = obs_prefix + obs.content
-            text += '\n[Error occurred in processing last action]'
-            return Message(role='user', content=[TextContent(text=text)])
-        elif isinstance(obs, UserRejectObservation):
-            text = obs_prefix + obs.content
-            text += '\n[Last action has been rejected by the user]'
-            return Message(role='user', content=[TextContent(text=text)])
-        else:
-            raise ValueError(f'Unknown observation type: {type(obs)}')
-
-    def step(self, state: State) -> Action:
-        """Performs one step using the Tester Agent.
-        This includes gathering info on previous steps and prompting the model to make a command to execute.
-
-        Parameters:
-        - state (State): used to get updated info
-
-        Returns:
-        - CmdRunAction(command) - bash command to run
-        - IPythonRunCellAction(code) - IPython code to run
-        - MessageAction(content) - Message action to run (e.g. ask for clarification)
-        - AgentFinishAction() - end the interaction
-        """
-        # if we're done, go back
-        latest_user_message = state.history.get_last_user_message()
-        if latest_user_message and latest_user_message.strip() == '/exit':
-            return AgentFinishAction()
-
-        # prepare what we want to send to the LLM
-        messages = self._get_messages(state)
-        params = {
-            'messages': self.llm.format_messages_for_llm(messages),
-            'stop': [
-                '</execute_ipython>',
-                '</execute_bash>',
-            ],
-        }
-
-        response = self.llm.completion(**params)
-
-        return self.action_parser.parse(response)
-
-    def _get_messages(self, state: State) -> list[Message]:
-        task = state.inputs.get('task', '')
-        summary = state.inputs.get('summary', '')
-
-        messages: list[Message] = [
-            Message(
-                role='system',
-                content=[
-                    TextContent(
-                        text=get_prompt(task, summary),
-                        cache_prompt=self.llm.is_caching_prompt_active(),  # Cache system prompt
-                    )
-                ],
-            ),
-        ]
-
-        for event in state.history.get_events():
-            # create a regular message from an event
-            if isinstance(event, Action):
-                message = self.get_action_message(event)
-            elif isinstance(event, Observation):
-                message = self.get_observation_message(event)
-            else:
-                raise ValueError(f'Unknown event type: {type(event)}')
-
-            # add regular message
-            if message:
-                # handle error if the message is the SAME role as the previous message
-                if messages and messages[-1].role == message.role:
-                    messages[-1].content.extend(message.content)
-                else:
-                    messages.append(message)
-
-        # Add caching to the last 2 user messages
-        if self.llm.is_caching_prompt_active():
-            user_turns_processed = 0
-            for message in reversed(messages):
-                if message.role == 'user' and user_turns_processed < 2:
-                    message.content[
-                        -1
-                    ].cache_prompt = True  # Last item inside the message content
-                    user_turns_processed += 1
-
-        # Add environment reminder to the latest user message
-        latest_user_message = next(
-            (m for m in reversed(messages) if m.role == 'user'),
-            None,
-        )
-
-        if latest_user_message:
-            reminder_text = f'\n\nENVIRONMENT REMINDER: You have {state.max_iterations - state.iteration} turns left to complete the task. When finished reply with <finish></finish>.'
-            latest_user_message.content.append(TextContent(text=reminder_text))
-
-        return messages
diff --git a/openhands/agenthub/tester_agent/prompt.py b/openhands/agenthub/tester_agent/prompt.py
deleted file mode 100644
index 7523bc7d16f3..000000000000
--- a/openhands/agenthub/tester_agent/prompt.py
+++ /dev/null
@@ -1,329 +0,0 @@
-# General Description, the goal is to devise a manager that is able to iterate if the solution has not been found yet.
-# In order to successfully fix an issue there are two phases:
-# 1. Exploring the codebase, finding the root cause of the issue.
-# 2. Implementing the solution.
-# Then the manager needs to check if the issue has been fixed, if not, it needs to iterate.
-general_description = """
-You are a QA Engineer, an expert in testing software.
-You are given an issue and your goal is to understand how to replicate the issue.
-
-Here is the issue you are trying to replicate:
-%(task)s
-
-Some other agents have already gathered information about the codebase.
-You can use this information to understand the codebase and replicate the issue.
-%(summary)s
-
-IMPORTANT: YOU SHOULD NEVER TRY TO IMPLEMENT A SOLUTION. YOUR ONLY GOAL IS TO REPLICATE THE ISSUE.
-As an expert in testing software, you have been equipped with a set of tools
-that will help you replicate the issue:
-- You can execute bash commands wrapped with <execute_bash>, e.g. <execute_bash> ls </execute_bash>.
-- If a bash command returns exit code `-1`, this means the process is not yet finished.
-- You must then send a second <execute_bash>. The second <execute_bash> can be empty
-  (which will retrieve any additional logs), or it can contain text to be sent to STDIN of the running process,
-  or it can contain the text `ctrl+c` to interrupt the process.
-- For commands that may run indefinitely, the output should be redirected to a file and the command run
-  in the background, e.g. <execute_bash> python3 app.py > server.log 2>&1 & </execute_bash>
-- If a command execution result says "Command timed out. Sending SIGINT to the process",
-  you should retry running the command in the background.
-
-You have access to a python interpreter wrapped with <execute_ipython>.
-e.g.:
-<execute_ipython>
-print("Hello World!")
-</execute_ipython>
-
-You can install Python packages using the %pip magic command in an IPython environment by using the following syntax: <execute_ipython> %pip install [package needed] </execute_ipython> and should always import packages and define variables before starting to use them.
-
-Apart from the standard Python library, you can also use the following functions (already imported) in <execute_ipython> environment:
-open_file(path: str, line_number: int | None = 1, context_lines: int | None = 100) -> None:
-    Opens the file at the given path in the editor. IF the file is to be edited, first use `scroll_down` repeatedly to read the full file!
-    If line_number is provided, the window will be moved to include that line.
-    It only shows the first 100 lines by default! `context_lines` is the max number of lines to be displayed, up to 100. Use `scroll_up` and `scroll_down` to view more content up or down.
-    Args:
-    path: str: The path to the file to open, preferred absolute path.
-    line_number: int | None = 1: The line number to move to. Defaults to 1.
-    context_lines: int | None = 100: Only shows this number of lines in the context window (usually from line 1), with line_number as the center (if possible). Defaults to 100.
-
-goto_line(line_number: int) -> None:
-    Moves the window to show the specified line number.
-    Args:
-    line_number: int: The line number to move to.
-
-scroll_down() -> None:
-    Moves the window down by 100 lines.
-    Args:
-    None
-
-scroll_up() -> None:
-    Moves the window up by 100 lines.
-    Args:
-    None
-
-search_dir(search_term: str, dir_path: str = './') -> None:
-    Searches for search_term in all files in dir. If dir is not provided, searches in the current directory.
-    Args:
-    search_term: str: The term to search for.
-    dir_path: str: The path to the directory to search.
-
-search_file(search_term: str, file_path: str | None = None) -> None:
-    Searches for search_term in file. If file is not provided, searches in the current open file.
-    Args:
-    search_term: str: The term to search for.
-    file_path: str | None: The path to the file to search.
-
-find_file(file_name: str, dir_path: str = './') -> None:
-    Finds all files with the given name in the specified directory.
-    Args:
-    file_name: str: The name of the file to find.
-    dir_path: str: The path to the directory to search.
-
-parse_pdf(file_path: str) -> None:
-    Parses the content of a PDF file and prints it.
-    Args:
-    file_path: str: The path to the file to open.
-
-parse_docx(file_path: str) -> None:
-    Parses the content of a DOCX file and prints it.
-    Args:
-    file_path: str: The path to the file to open.
-
-parse_latex(file_path: str) -> None:
-    Parses the content of a LaTex file and prints it.
-    Args:
-    file_path: str: The path to the file to open.
-
-parse_pptx(file_path: str) -> None:
-    Parses the content of a pptx file and prints it.
-    Args:
-    file_path: str: The path to the file to open.
-
-
-IMPORTANT:
-- `open_file` only returns the first 100 lines of the file by default!
-- Indentation is important and code that is not indented correctly will fail and require fixing before it can be run.
-- Any code issued should be less than 50 lines to avoid context being cut off!
-
-
-Create a test that when executed, it will replicate the issue.
-Responses should be concise.
-You should attempt fewer things at a time instead of putting too many commands OR too much code in one "execute" block.
-Include ONLY ONE <execute_ipython>, <execute_bash>, or <execute_browse> per response, unless you is finished with the task or needs more input or action from the user in order to proceed.
-If you is finished with the task you MUST include <finish></finish> in your response.
-IMPORTANT: Execute code using <execute_ipython>, <execute_bash>, or <execute_browse> whenever possible.
-IMPORTANT: You MUST NOT edit or remove any files, but you are allowed to add new files. Code executed in <execute_ipython> WON'T BE SAVED.
-You should utilize full file paths and the `pwd` command to prevent path-related errors.
-You MUST NOT apologize to the user or thank the user after running commands or editing files. It should only address the user in response to an explicit message from the user, or to ask for more information.
-You MUST NOT push any changes to GitHub unless explicitly requested to do so.
-
-When you have successfully managed to replicate the issue in a test,
-you MUST generate a NEW test file that when executed will replicate the issue.
-Then, you MUST run the test and verify that the issue is reproducible.
-If it is not reproducible, remove the new test file and think of a different approach to replicate the issue.
-If it is reproducible, you MUST output the command to run the test file wrapped with <finish></finish> tags.
-
---- START OF EXAMPLE ---
-
-USER: I need to test if there's an issue with the ModelChain when processing multiple inputs in parallel.
-
-YOU: I'll create a test to verify the ModelChain's behavior with parallel processing.
-
-<execute_ipython>
-import torch
-from openhands.models import ModelChain
-import concurrent.futures
-
-def test_model_chain_parallel():
-    # Create a simple model chain
-    model = ModelChain(
-        backbone="resnet18",
-        num_classes=10
-    )
-
-    # Generate dummy input data
-    batch_size = 4
-    input_data = torch.randn(batch_size, 3, 224, 224)
-
-    # Process inputs in parallel
-    with concurrent.futures.ThreadPoolExecutor(max_workers=2) as executor:
-        futures = []
-        for i in range(batch_size):
-            futures.append(
-                executor.submit(model.forward, input_data[i:i+1])
-            )
-
-        # Get results
-        results = [future.result() for future in futures]
-
-    return results
-
-# Run the test
-results = test_model_chain_parallel()
-print(f"Processed {len(results)} inputs in parallel")
-</execute_ipython>
-
-USER:
-OBSERVATION:
-RuntimeError: Cannot re-initialize CUDA in forked subprocess. To use CUDA with multiprocessing, you must use 'spawn' start method
-
-YOU: Ah, I see the issue is reproducible. Let me create a test file that demonstrates this CUDA initialization problem.
-
-<execute_ipython>
-test_code = '''import torch
-import torch.multiprocessing as mp
-import pytest
-from openhands.models import ModelChain
-
-def process_input(model, input_tensor):
-    return model(input_tensor)
-
-def test_model_chain_multiprocess():
-    # Force CUDA if available
-    device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')
-
-    # Create model and move to device
-    model = ModelChain(
-        backbone="resnet18",
-        num_classes=10
-    ).to(device)
-
-    # Create dummy input
-    input_tensor = torch.randn(1, 3, 224, 224).to(device)
-
-    # Try parallel processing
-    mp.set_start_method('fork')  # This will cause the CUDA error
-    processes = []
-
-    for _ in range(2):
-        p = mp.Process(target=process_input, args=(model, input_tensor))
-        p.start()
-        processes.append(p)
-
-    for p in processes:
-        p.join()'''
-
-with open('test_model_chain_cuda.py', 'w') as f:
-    f.write(test_code)
-</execute_ipython>
-
-YOU: I've created a test file that reproduces the CUDA initialization issue. Let me run it to confirm it works.
-
-<execute_bash>
-python3 -m pytest test_model_chain_cuda.py -v
-</execute_bash>
-
-USER:
-OBSERVATION:
-RuntimeError: Cannot re-initialize CUDA in forked subprocess. To use CUDA with multiprocessing, you must use 'spawn' start method
-
-YOU: Perfect, I've successfully replicated the issue and created a test file that demonstrates the problem.
-
-<finish>
-python3 -m pytest test_model_chain_cuda.py -v
-</finish>
-
---- END OF EXAMPLE ---
-
---- START OF EXAMPLE 2 ---
-
-USER: I need to test if the CLI's quiet mode is working properly. The issue is that there's no way to reduce verbosity below the default level.
-
-YOU: I'll create a test to verify the CLI's verbosity behavior.
-
-<execute_ipython>
-import subprocess
-import tempfile
-
-def test_cli_verbosity():
-    # Create a temporary file with some content to test
-    with tempfile.NamedTemporaryFile(mode='w', suffix='.sql') as temp_file:
-        temp_file.write("SELECT  *  FROM table")  # Extra spaces to trigger formatting
-        temp_file.flush()
-
-        # Run command with default verbosity
-        default_result = subprocess.run(
-            ['sqlfluff', 'fix', temp_file.name],
-            capture_output=True,
-            text=True
-        )
-
-        # Run command with --quiet flag (which doesn't exist yet, this should fail)
-        quiet_result = subprocess.run(
-            ['sqlfluff', 'fix', '--quiet', temp_file.name],
-            capture_output=True,
-            text=True
-        )
-
-        print("Default verbosity output length:", len(default_result.stdout.split('\n')))
-        print("Default output:", default_result.stdout)
-        print("\nQuiet mode output length:", len(quiet_result.stdout.split('\n')))
-        print("Quiet output:", quiet_result.stdout)
-
-test_cli_verbosity()
-</execute_ipython>
-
-USER:
-OBSERVATION:
-FileNotFoundError: [Errno 2] No such file or directory: 'sqlfluff'
-
-YOU: Let me create a test file that demonstrates this verbosity issue with our CLI tool.
-
-<execute_ipython>
-test_code = '''import subprocess
-import tempfile
-import pytest
-
-def test_cli_verbosity_levels():
-    # Create a temporary file with content that needs formatting
-    with tempfile.NamedTemporaryFile(mode='w', suffix='.sql') as temp_file:
-        temp_file.write("SELECT  *  FROM table")  # Extra spaces to trigger formatting
-        temp_file.flush()
-
-        # Test default verbosity
-        default_cmd = ['sqlfluff', 'fix', temp_file.name]
-        default_output = subprocess.run(default_cmd, capture_output=True, text=True)
-
-        # Test with quiet flag (should fail as feature doesn't exist)
-        quiet_cmd = ['sqlfluff', 'fix', '--quiet', temp_file.name]
-        quiet_output = subprocess.run(quiet_cmd, capture_output=True, text=True)
-
-        # Verify that default output is more verbose
-        assert len(default_output.stdout.split('\\n')) > 1, "Default output should be verbose"
-
-        # This assertion should fail since --quiet flag doesn't exist
-        with pytest.raises(subprocess.CalledProcessError):
-            subprocess.run(quiet_cmd, check=True, capture_output=True, text=True)'''
-
-with open('test_cli_verbosity.py', 'w') as f:
-    f.write(test_code)
-</execute_ipython>
-
-YOU: Let me run the test to verify it demonstrates the missing quiet mode functionality.
-
-<execute_bash>
-python3 -m pytest test_cli_verbosity.py -v
-</execute_bash>
-
-USER:
-OBSERVATION:
-FileNotFoundError: [Errno 2] No such file or directory: 'sqlfluff'
-
-YOU: Perfect, I've created a test file that demonstrates the lack of quiet mode functionality. The test fails because sqlfluff is not installed, but the test case correctly shows that the CLI tool lacks the ability to reduce verbosity below the default level.
-
-<finish>
-python3 -m pytest test_cli_verbosity.py -v
-</finish>
-
---- END OF EXAMPLE 2 ---
-
-IMPORTANT: YOU MUST ONLY GENERATE A <finish></finish> TAG IN YOUR RESPONSE with only ONE command to run the test file.
-
-"""
-
-
-def get_prompt(task: str, summary: str) -> str:
-    formatted_prompt = (general_description) % {
-        'task': task,
-        'summary': summary,
-    }
-    return formatted_prompt
diff --git a/openhands/controller/agent_controller.py b/openhands/controller/agent_controller.py
index b2978f5bfb50..f95ead873833 100644
--- a/openhands/controller/agent_controller.py
+++ b/openhands/controller/agent_controller.py
@@ -2,7 +2,7 @@
 import copy
 import logging
 import traceback
-from typing import Type
+from typing import Callable, ClassVar, Type
 
 import litellm
 
@@ -36,9 +36,8 @@
 from openhands.events.observation import (
     AgentDelegateObservation,
     AgentStateChangedObservation,
-    CmdOutputObservation,
     ErrorObservation,
-    FatalErrorObservation,
+    NullObservation,
     Observation,
 )
 from openhands.events.serialization.event import truncate_content
@@ -64,7 +63,12 @@ class AgentController:
     parent: 'AgentController | None' = None
     delegate: 'AgentController | None' = None
     _pending_action: Action | None = None
-    logger: logging.Logger
+    filter_out: ClassVar[tuple[type[Event], ...]] = (
+        NullAction,
+        NullObservation,
+        ChangeAgentStateAction,
+        AgentStateChangedObservation,
+    )
 
     def __init__(
         self,
@@ -79,6 +83,7 @@ def __init__(
         initial_state: State | None = None,
         is_delegate: bool = False,
         headless_mode: bool = True,
+        status_callback: Callable | None = None,
     ):
         """Initializes a new instance of the AgentController class.
 
@@ -105,7 +110,7 @@ def __init__(
         # subscribe to the event stream
         self.event_stream = event_stream
         self.event_stream.subscribe(
-            EventStreamSubscriber.AGENT_CONTROLLER, self.on_event, append=is_delegate
+            EventStreamSubscriber.AGENT_CONTROLLER, self.on_event, self.id
         )
 
         # state from the previous session, state from a parent agent, or a fresh state
@@ -122,11 +127,38 @@ def __init__(
 
         # stuck helper
         self._stuck_detector = StuckDetector(self.state)
+        self.status_callback = status_callback
 
     async def close(self):
-        """Closes the agent controller, canceling any ongoing tasks and unsubscribing from the event stream."""
+        """Closes the agent controller, canceling any ongoing tasks and unsubscribing from the event stream.
+
+        Note that it's fairly important that this closes properly, otherwise the state is incomplete."""
         await self.set_agent_state_to(AgentState.STOPPED)
-        self.event_stream.unsubscribe(EventStreamSubscriber.AGENT_CONTROLLER)
+
+        # we made history, now is the time to rewrite it!
+        # the final state.history will be used by external scripts like evals, tests, etc.
+        # history will need to be complete WITH delegates events
+        # like the regular agent history, it does not include:
+        # - 'hidden' events, events with hidden=True
+        # - backend events (the default 'filtered out' types, types in self.filter_out)
+        start_id = self.state.start_id if self.state.start_id >= 0 else 0
+        end_id = (
+            self.state.end_id
+            if self.state.end_id >= 0
+            else self.event_stream.get_latest_event_id()
+        )
+        self.state.history = list(
+            self.event_stream.get_events(
+                start_id=start_id,
+                end_id=end_id,
+                reverse=False,
+                filter_out_type=self.filter_out,
+                filter_hidden=True,
+            )
+        )
+
+        # unsubscribe from the event stream
+        self.event_stream.unsubscribe(EventStreamSubscriber.AGENT_CONTROLLER, self.id)
 
     def log(self, level: str, message: str, extra: dict | None = None):
         """Logs a message to the agent controller's logger.
@@ -135,7 +167,7 @@ def log(self, level: str, message: str, extra: dict | None = None):
             message (str): The message to log.
         """
         message = f'[Agent Controller {self.id}] {message}'
-        getattr(logger, level)(message, extra=extra)
+        getattr(logger, level)(message, extra=extra, stacklevel=2)
 
     def update_state_before_step(self):
         self.state.iteration += 1
@@ -145,22 +177,16 @@ async def update_state_after_step(self):
         # update metrics especially for cost. Use deepcopy to avoid it being modified by agent.reset()
         self.state.local_metrics = copy.deepcopy(self.agent.llm.metrics)
 
-    async def report_error(self, message: str, exception: Exception | None = None):
-        """Reports an error to the user and sends the exception to the LLM next step, in the hope it can self-correct.
-
-        This method should be called for a particular type of errors, which have:
-        - a user-friendly message, which will be shown in the chat box. This should not be a raw exception message.
-        - an ErrorObservation that can be sent to the LLM by the user role, with the exception message, so it can self-correct next time.
-        """
-        self.state.last_error = message
-        if exception:
-            self.state.last_error += f': {exception}'
-        detail = str(exception) if exception is not None else ''
-        if exception is not None and isinstance(exception, litellm.AuthenticationError):
-            detail = 'Please check your credentials. Is your API key correct?'
-        self.event_stream.add_event(
-            ErrorObservation(f'{message}:{detail}'), EventSource.USER
-        )
+    async def _react_to_exception(
+        self,
+        e: Exception,
+    ):
+        await self.set_agent_state_to(AgentState.ERROR)
+        if self.status_callback is not None:
+            err_id = ''
+            if isinstance(e, litellm.AuthenticationError):
+                err_id = 'STATUS$ERROR_LLM_AUTHENTICATION'
+            self.status_callback('error', err_id, str(e))
 
     async def start_step_loop(self):
         """The main loop for the agent's step-by-step execution."""
@@ -175,12 +201,7 @@ async def start_step_loop(self):
             except Exception as e:
                 traceback.print_exc()
                 self.log('error', f'Error while running the agent: {e}')
-                self.log('error', traceback.format_exc())
-                await self.report_error(
-                    'There was an unexpected error while running the agent', exception=e
-                )
-                await self.set_agent_state_to(AgentState.ERROR)
-                break
+                await self._react_to_exception(e)
 
             await asyncio.sleep(0.1)
 
@@ -192,6 +213,11 @@ async def on_event(self, event: Event):
         """
         if hasattr(event, 'hidden') and event.hidden:
             return
+
+        # if the event is not filtered out, add it to the history
+        if not any(isinstance(event, filter_type) for filter_type in self.filter_out):
+            self.state.history.append(event)
+
         if isinstance(event, Action):
             await self._handle_action(event)
         elif isinstance(event, Observation):
@@ -230,15 +256,6 @@ async def _handle_observation(self, observation: Observation):
         Args:
             observation (observation): The observation to handle.
         """
-        if (
-            self._pending_action
-            and hasattr(self._pending_action, 'confirmation_state')
-            and self._pending_action.confirmation_state
-            == ActionConfirmationStatus.AWAITING_CONFIRMATION
-        ):
-            return
-
-        # Make sure we print the observation in the same way as the LLM sees it
         observation_to_print = copy.deepcopy(observation)
         if len(observation_to_print.content) > self.agent.llm.config.max_message_chars:
             observation_to_print.content = truncate_content(
@@ -246,7 +263,6 @@ async def _handle_observation(self, observation: Observation):
             )
         self.log('debug', str(observation_to_print), extra={'msg_type': 'OBSERVATION'})
 
-        # Merge with the metrics from the LLM - it will to synced to the controller's local metrics in update_state_after_step()
         if observation.llm_metrics is not None:
             self.agent.llm.metrics.merge(observation.llm_metrics)
 
@@ -257,20 +273,9 @@ async def _handle_observation(self, observation: Observation):
             if self.state.agent_state == AgentState.USER_REJECTED:
                 await self.set_agent_state_to(AgentState.AWAITING_USER_INPUT)
             return
-
-        if isinstance(observation, CmdOutputObservation):
-            return
-        elif isinstance(observation, AgentDelegateObservation):
-            self.state.history.on_event(observation)
         elif isinstance(observation, ErrorObservation):
             if self.state.agent_state == AgentState.ERROR:
                 self.state.metrics.merge(self.state.local_metrics)
-        elif isinstance(observation, FatalErrorObservation):
-            self.state.last_error = (
-                f'There was a fatal error during agent execution: {str(observation)}'
-            )
-            self.state.metrics.merge(self.state.local_metrics)
-            await self.set_agent_state_to(AgentState.ERROR)
 
     async def _handle_message_action(self, action: MessageAction):
         """Handles message actions from the event stream.
@@ -309,7 +314,7 @@ async def set_agent_state_to(self, new_state: AgentState):
         if new_state == self.state.agent_state:
             return
 
-        if new_state == AgentState.STOPPED or new_state == AgentState.ERROR:
+        if new_state in (AgentState.STOPPED, AgentState.ERROR):
             self.reset_task()
         elif (
             new_state == AgentState.RUNNING
@@ -335,8 +340,7 @@ async def set_agent_state_to(self, new_state: AgentState):
                 if self.state.metrics.accumulated_cost >= self.max_budget_per_task:
                     self.max_budget_per_task += self._initial_max_budget_per_task
         elif self._pending_action is not None and (
-            new_state == AgentState.USER_CONFIRMED
-            or new_state == AgentState.USER_REJECTED
+            new_state in (AgentState.USER_CONFIRMED, AgentState.USER_REJECTED)
         ):
             if hasattr(self._pending_action, 'thought'):
                 self._pending_action.thought = ''  # type: ignore[union-attr]
@@ -349,7 +353,8 @@ async def set_agent_state_to(self, new_state: AgentState):
 
         self.state.agent_state = new_state
         self.event_stream.add_event(
-            AgentStateChangedObservation('', self.state.agent_state), EventSource.AGENT
+            AgentStateChangedObservation('', self.state.agent_state),
+            EventSource.ENVIRONMENT,
         )
 
         if new_state == AgentState.INIT and self.state.resume_state:
@@ -393,11 +398,15 @@ async def start_delegate(self, action: AgentDelegateAction):
             delegate_level=self.state.delegate_level + 1,
             # global metrics should be shared between parent and child
             metrics=self.state.metrics,
+            # start on top of the stream
+            start_id=self.event_stream.get_latest_event_id() + 1,
         )
         self.log(
             'debug',
             f'start delegate, creating agent {delegate_agent.name} using LLM {llm}',
         )
+
+        self.event_stream.unsubscribe(EventStreamSubscriber.AGENT_CONTROLLER, self.id)
         self.delegate = AgentController(
             sid=self.id + '-delegate',
             agent=delegate_agent,
@@ -422,12 +431,8 @@ async def _step(self) -> None:
             await asyncio.sleep(1)
             return
 
-        # check if agent got stuck before taking any action
         if self._is_stuck():
-            # This need to go BEFORE report_error to sync metrics
-            self.event_stream.add_event(
-                FatalErrorObservation('Agent got stuck in a loop'), EventSource.USER
-            )
+            await self._react_to_exception(RuntimeError('Agent got stuck in a loop'))
             return
 
         if self.delegate is not None:
@@ -466,15 +471,12 @@ async def _step(self) -> None:
             if action is None:
                 raise LLMNoActionError('No action was returned')
         except (LLMMalformedActionError, LLMNoActionError, LLMResponseError) as e:
-            # report to the user
-            # and send the underlying exception to the LLM for self-correction
-            await self.report_error(str(e))
-            return
-        # FIXME: more graceful handling of litellm.exceptions.ContextWindowExceededError
-        # e.g. try to condense the memory and try again
-        except litellm.exceptions.ContextWindowExceededError as e:
-            self.state.last_error = str(e)
-            await self.set_agent_state_to(AgentState.ERROR)
+            self.event_stream.add_event(
+                ErrorObservation(
+                    content=str(e),
+                ),
+                EventSource.AGENT,
+            )
             return
 
         if action.runnable:
@@ -496,13 +498,12 @@ async def _step(self) -> None:
             self.event_stream.add_event(action, EventSource.AGENT)
 
         await self.update_state_after_step()
+
         self.log('debug', str(action), extra={'msg_type': 'ACTION'})
 
     async def _delegate_step(self):
         """Executes a single step of the delegate agent."""
-        self.log('debug', 'Delegate not none, awaiting...')
         await self.delegate._step()  # type: ignore[union-attr]
-        self.log('debug', 'Delegate step done')
         assert self.delegate is not None
         delegate_state = self.delegate.get_agent_state()
         self.log('debug', f'Delegate state: {delegate_state}')
@@ -510,12 +511,26 @@ async def _delegate_step(self):
             # update iteration that shall be shared across agents
             self.state.iteration = self.delegate.state.iteration
 
+            # emit AgentDelegateObservation to mark delegate termination due to error
+            delegate_outputs = (
+                self.delegate.state.outputs if self.delegate.state else {}
+            )
+            content = (
+                f'{self.delegate.agent.name} encountered an error during execution.'
+            )
+            obs = AgentDelegateObservation(outputs=delegate_outputs, content=content)
+            self.event_stream.add_event(obs, EventSource.AGENT)
+
             # close the delegate upon error
             await self.delegate.close()
+
+            # resubscribe parent when delegate is finished
+            self.event_stream.subscribe(
+                EventStreamSubscriber.AGENT_CONTROLLER, self.on_event, self.id
+            )
             self.delegate = None
             self.delegateAction = None
 
-            await self.report_error('Delegator agent encountered an error')
         elif delegate_state in (AgentState.FINISHED, AgentState.REJECTED):
             self.log('debug', 'Delegate agent has finished execution')
             # retrieve delegate result
@@ -527,6 +542,11 @@ async def _delegate_step(self):
             # close delegate controller: we must close the delegate controller before adding new events
             await self.delegate.close()
 
+            # resubscribe parent when delegate is finished
+            self.event_stream.subscribe(
+                EventStreamSubscriber.AGENT_CONTROLLER, self.on_event, self.id
+            )
+
             # update delegate result observation
             # TODO: replace this with AI-generated summary (#2395)
             formatted_output = ', '.join(
@@ -535,9 +555,7 @@ async def _delegate_step(self):
             content = (
                 f'{self.delegate.agent.name} finishes task with {formatted_output}'
             )
-            obs: Observation = AgentDelegateObservation(
-                outputs=outputs, content=content
-            )
+            obs = AgentDelegateObservation(outputs=outputs, content=content)
 
             # clean up delegate status
             self.delegate = None
@@ -564,21 +582,18 @@ async def _handle_traffic_control(
         else:
             self.state.traffic_control_state = TrafficControlState.THROTTLING
             if self.headless_mode:
-                # This need to go BEFORE report_error to sync metrics
-                await self.set_agent_state_to(AgentState.ERROR)
-                # set to ERROR state if running in headless mode
-                # since user cannot resume on the web interface
-                await self.report_error(
-                    f'Agent reached maximum {limit_type} in headless mode, task stopped. '
+                e = RuntimeError(
+                    f'Agent reached maximum {limit_type} in headless mode. '
                     f'Current {limit_type}: {current_value:.2f}, max {limit_type}: {max_value:.2f}'
                 )
+                await self._react_to_exception(e)
             else:
-                await self.set_agent_state_to(AgentState.PAUSED)
-                await self.report_error(
-                    f'Agent reached maximum {limit_type}, task paused. '
+                e = RuntimeError(
+                    f'Agent reached maximum {limit_type}. '
                     f'Current {limit_type}: {current_value:.2f}, max {limit_type}: {max_value:.2f}. '
-                    f'{TRAFFIC_CONTROL_REMINDER}'
                 )
+                # FIXME: this isn't really an exception--we should have a different path
+                await self._react_to_exception(e)
             stop_step = True
         return stop_step
 
@@ -603,8 +618,10 @@ def set_initial_state(
             max_iterations: The maximum number of iterations allowed for the task.
             confirmation_mode: Whether to enable confirmation mode.
         """
-        # state from the previous session, state from a parent agent, or a new state
-        # note that this is called twice when restoring a previous session, first with state=None
+        # state can come from:
+        # - the previous session, in which case it has history
+        # - from a parent agent, in which case it has no history
+        # - None / a new state
         if state is None:
             self.state = State(
                 inputs={},
@@ -614,27 +631,109 @@ def set_initial_state(
         else:
             self.state = state
 
-        # when restored from a previous session, the State object will have history, start_id, and end_id
-        # connect it to the event stream
-        self.state.history.set_event_stream(self.event_stream)
+            if self.state.start_id <= -1:
+                self.state.start_id = 0
 
-        # if start_id was not set in State, we're starting fresh, at the top of the stream
-        start_id = self.state.start_id
-        if start_id == -1:
-            start_id = self.event_stream.get_latest_event_id() + 1
-        else:
             self.log(
-                'debug', f'AgentController {self.id} restoring from event {start_id}'
+                'debug',
+                f'AgentController {self.id} initializing history from event {self.state.start_id}',
             )
 
+            self._init_history()
+
+    def _init_history(self):
+        """Initializes the agent's history from the event stream.
+
+        The history is a list of events that:
+        - Excludes events of types listed in self.filter_out
+        - Excludes events with hidden=True attribute
+        - For delegate events (between AgentDelegateAction and AgentDelegateObservation):
+            - Excludes all events between the action and observation
+            - Includes the delegate action and observation themselves
+        """
+
+        # define range of events to fetch
+        # delegates start with a start_id and initially won't find any events
+        # otherwise we're restoring a previous session
+        start_id = self.state.start_id if self.state.start_id >= 0 else 0
+        end_id = (
+            self.state.end_id
+            if self.state.end_id >= 0
+            else self.event_stream.get_latest_event_id()
+        )
+
+        # sanity check
+        if start_id > end_id + 1:
+            self.log(
+                'debug',
+                f'start_id {start_id} is greater than end_id + 1 ({end_id + 1}). History will be empty.',
+            )
+            self.state.history = []
+            return
+
+        # Get all events, filtering out backend events and hidden events
+        events = list(
+            self.event_stream.get_events(
+                start_id=start_id,
+                end_id=end_id,
+                reverse=False,
+                filter_out_type=self.filter_out,
+                filter_hidden=True,
+            )
+        )
+
+        # Find all delegate action/observation pairs
+        delegate_ranges: list[tuple[int, int]] = []
+        delegate_action_ids: list[int] = []  # stack of unmatched delegate action IDs
+
+        for event in events:
+            if isinstance(event, AgentDelegateAction):
+                delegate_action_ids.append(event.id)
+                # Note: we can get agent=event.agent and task=event.inputs.get('task','')
+                # if we need to track these in the future
+
+            elif isinstance(event, AgentDelegateObservation):
+                # Match with most recent unmatched delegate action
+                if not delegate_action_ids:
+                    self.log(
+                        'error',
+                        f'Found AgentDelegateObservation without matching action at id={event.id}',
+                    )
+                    continue
+
+                action_id = delegate_action_ids.pop()
+                delegate_ranges.append((action_id, event.id))
+
+        # Filter out events between delegate action/observation pairs
+        if delegate_ranges:
+            filtered_events: list[Event] = []
+            current_idx = 0
+
+            for start_id, end_id in sorted(delegate_ranges):
+                # Add events before delegate range
+                filtered_events.extend(
+                    event for event in events[current_idx:] if event.id < start_id
+                )
+
+                # Add delegate action and observation
+                filtered_events.extend(
+                    event for event in events if event.id in (start_id, end_id)
+                )
+
+                # Update index to after delegate range
+                current_idx = next(
+                    (i for i, e in enumerate(events) if e.id > end_id), len(events)
+                )
+
+            # Add any remaining events after last delegate range
+            filtered_events.extend(events[current_idx:])
+
+            self.state.history = filtered_events
+        else:
+            self.state.history = events
+
         # make sure history is in sync
         self.state.start_id = start_id
-        self.state.history.start_id = start_id
-
-        # if there was an end_id saved in State, set it in history
-        # currently not used, later useful for delegates
-        if self.state.end_id > -1:
-            self.state.history.end_id = self.state.end_id
 
     def _is_stuck(self):
         """Checks if the agent or its delegate is stuck in a loop.
diff --git a/openhands/controller/state/state.py b/openhands/controller/state/state.py
index 52a21f0499aa..96c0ab7e8322 100644
--- a/openhands/controller/state/state.py
+++ b/openhands/controller/state/state.py
@@ -11,8 +11,8 @@
     MessageAction,
 )
 from openhands.events.action.agent import AgentFinishAction
+from openhands.events.event import Event, EventSource
 from openhands.llm.metrics import Metrics
-from openhands.memory.history import ShortTermHistory
 from openhands.storage.files import FileStore
 
 
@@ -77,10 +77,9 @@ class State:
     # max number of iterations for the current task
     max_iterations: int = 100
     confirmation_mode: bool = False
-    history: ShortTermHistory = field(default_factory=ShortTermHistory)
+    history: list[Event] = field(default_factory=list)
     inputs: dict = field(default_factory=dict)
     outputs: dict = field(default_factory=dict)
-    last_error: str | None = None
     agent_state: AgentState = AgentState.LOADING
     resume_state: AgentState | None = None
     traffic_control_state: TrafficControlState = TrafficControlState.NORMAL
@@ -94,9 +93,11 @@ class State:
     start_id: int = -1
     end_id: int = -1
     almost_stuck: int = 0
+    delegates: dict[tuple[int, int], tuple[str, str]] = field(default_factory=dict)
     # NOTE: This will never be used by the controller, but it can be used by different
     # evaluation tasks to store extra data needed to track the progress/state of the task.
     extra_data: dict[str, Any] = field(default_factory=dict)
+    last_error: str = ''
 
     def save_to_session(self, sid: str, file_store: FileStore):
         pickled = pickle.dumps(self)
@@ -115,7 +116,7 @@ def restore_from_session(sid: str, file_store: FileStore) -> 'State':
             pickled = base64.b64decode(encoded)
             state = pickle.loads(pickled)
         except Exception as e:
-            logger.warning(f'Failed to restore state from session: {e}')
+            logger.warning(f'Could not restore state from session: {e}')
             raise e
 
         # update state
@@ -124,49 +125,45 @@ def restore_from_session(sid: str, file_store: FileStore) -> 'State':
         else:
             state.resume_state = None
 
-        # don't carry last_error anymore after restore
-        state.last_error = None
-
         # first state after restore
         state.agent_state = AgentState.LOADING
         return state
 
     def __getstate__(self):
+        # don't pickle history, it will be restored from the event stream
         state = self.__dict__.copy()
-
-        # save the relevant data from recent history
-        # so that we can restore it when the state is restored
-        if 'history' in state:
-            state['start_id'] = state['history'].start_id
-            state['end_id'] = state['history'].end_id
-
-        # don't save history object itself
-        state.pop('history', None)
+        state['history'] = []
         return state
 
     def __setstate__(self, state):
         self.__dict__.update(state)
 
-        # recreate the history object
+        # make sure we always have the attribute history
         if not hasattr(self, 'history'):
-            self.history = ShortTermHistory()
+            self.history = []
 
-        # restore the relevant data in history from the state
-        self.history.start_id = self.start_id
-        self.history.end_id = self.end_id
-
-        # remove the restored data from the state if any
-
-    def get_current_user_intent(self):
+    def get_current_user_intent(self) -> tuple[str | None, list[str] | None]:
         """Returns the latest user message and image(if provided) that appears after a FinishAction, or the first (the task) if nothing was finished yet."""
         last_user_message = None
         last_user_message_image_urls: list[str] | None = []
-        for event in self.history.get_events(reverse=True):
+        for event in reversed(self.history):
             if isinstance(event, MessageAction) and event.source == 'user':
                 last_user_message = event.content
                 last_user_message_image_urls = event.images_urls
             elif isinstance(event, AgentFinishAction):
                 if last_user_message is not None:
-                    return last_user_message
+                    return last_user_message, None
 
         return last_user_message, last_user_message_image_urls
+
+    def get_last_agent_message(self) -> str | None:
+        for event in reversed(self.history):
+            if isinstance(event, MessageAction) and event.source == EventSource.AGENT:
+                return event.content
+        return None
+
+    def get_last_user_message(self) -> str | None:
+        for event in reversed(self.history):
+            if isinstance(event, MessageAction) and event.source == EventSource.USER:
+                return event.content
+        return None
diff --git a/openhands/controller/stuck.py b/openhands/controller/stuck.py
index 230d5f2e81ac..0eb0f4c893ca 100644
--- a/openhands/controller/stuck.py
+++ b/openhands/controller/stuck.py
@@ -28,7 +28,7 @@ def is_stuck(self):
         # filter out MessageAction with source='user' from history
         filtered_history = [
             event
-            for event in self.state.history.get_events()
+            for event in self.state.history
             if not (
                 (isinstance(event, MessageAction) and event.source == EventSource.USER)
                 or
diff --git a/openhands/core/cli.py b/openhands/core/cli.py
index 6a2620790f6e..5a4f30da7fdc 100644
--- a/openhands/core/cli.py
+++ b/openhands/core/cli.py
@@ -1,6 +1,8 @@
 import asyncio
 import logging
+import sys
 from typing import Type
+from uuid import uuid4
 
 from termcolor import colored
 
@@ -13,6 +15,7 @@
     load_app_config,
 )
 from openhands.core.logger import openhands_logger as logger
+from openhands.core.loop import run_agent_until_done
 from openhands.core.schema import AgentState
 from openhands.events import EventSource, EventStream, EventStreamSubscriber
 from openhands.events.action import (
@@ -61,7 +64,7 @@ def display_event(event: Event):
         if hasattr(event, 'thought'):
             display_message(event.thought)
     if isinstance(event, MessageAction):
-        if event.source != EventSource.USER:
+        if event.source == EventSource.AGENT:
             display_message(event.content)
     if isinstance(event, CmdRunAction):
         display_command(event.command)
@@ -114,7 +117,6 @@ async def main():
         sid=sid,
         plugins=agent_cls.sandbox_plugins,
     )
-    await runtime.connect()
 
     controller = AgentController(
         agent=agent,
@@ -124,14 +126,17 @@ async def main():
         event_stream=event_stream,
     )
 
-    if controller is not None:
-        controller.agent_task = asyncio.create_task(controller.start_step_loop())
-
     async def prompt_for_next_task():
-        next_message = input('How can I help? >> ')
+        # Run input() in a thread pool to avoid blocking the event loop
+        loop = asyncio.get_event_loop()
+        next_message = await loop.run_in_executor(
+            None, lambda: input('How can I help? >> ')
+        )
+        if not next_message.strip():
+            await prompt_for_next_task()
         if next_message == 'exit':
             event_stream.add_event(
-                ChangeAgentStateAction(AgentState.STOPPED), EventSource.USER
+                ChangeAgentStateAction(AgentState.STOPPED), EventSource.ENVIRONMENT
             )
             return
         action = MessageAction(content=next_message)
@@ -140,31 +145,45 @@ async def prompt_for_next_task():
     async def on_event(event: Event):
         display_event(event)
         if isinstance(event, AgentStateChangedObservation):
-            if event.agent_state == AgentState.ERROR:
-                print('An error occurred. Please try again.')
             if event.agent_state in [
                 AgentState.AWAITING_USER_INPUT,
                 AgentState.FINISHED,
-                AgentState.ERROR,
             ]:
                 await prompt_for_next_task()
 
-    event_stream.subscribe(EventStreamSubscriber.MAIN, on_event)
+    event_stream.subscribe(EventStreamSubscriber.MAIN, on_event, str(uuid4()))
 
-    await prompt_for_next_task()
+    await runtime.connect()
 
-    while controller.state.agent_state not in [
-        AgentState.STOPPED,
-    ]:
-        await asyncio.sleep(1)  # Give back control for a tick, so the agent can run
+    asyncio.create_task(prompt_for_next_task())
 
-    print('Exiting...')
-    await controller.close()
+    await run_agent_until_done(
+        controller, runtime, [AgentState.STOPPED, AgentState.ERROR]
+    )
 
 
 if __name__ == '__main__':
-    loop = asyncio.get_event_loop()
+    loop = asyncio.new_event_loop()
+    asyncio.set_event_loop(loop)
     try:
         loop.run_until_complete(main())
+    except KeyboardInterrupt:
+        print('Received keyboard interrupt, shutting down...')
+    except ConnectionRefusedError as e:
+        print(f'Connection refused: {e}')
+        sys.exit(1)
+    except Exception as e:
+        print(f'An error occurred: {e}')
+        sys.exit(1)
     finally:
-        pass
+        try:
+            # Cancel all running tasks
+            pending = asyncio.all_tasks(loop)
+            for task in pending:
+                task.cancel()
+            # Wait for all tasks to complete with a timeout
+            loop.run_until_complete(asyncio.gather(*pending, return_exceptions=True))
+            loop.close()
+        except Exception as e:
+            print(f'Error during cleanup: {e}')
+            sys.exit(1)
diff --git a/openhands/core/config/agent_config.py b/openhands/core/config/agent_config.py
index 5061bfd4755c..bfaf978e5703 100644
--- a/openhands/core/config/agent_config.py
+++ b/openhands/core/config/agent_config.py
@@ -9,7 +9,7 @@ class AgentConfig:
 
     Attributes:
         function_calling: Whether function calling is enabled. Default is True.
-        codeact_enable_browsing_delegate: Whether browsing delegate is enabled in the action space. Default is False. Only works with function calling.
+        codeact_enable_browsing: Whether browsing delegate is enabled in the action space. Default is False. Only works with function calling.
         codeact_enable_llm_editor: Whether LLM editor is enabled in the action space. Default is False. Only works with function calling.
         codeact_enable_jupyter: Whether Jupyter is enabled in the action space. Default is False.
         micro_agent_name: The name of the micro agent to use for this agent.
@@ -19,7 +19,7 @@ class AgentConfig:
     """
 
     function_calling: bool = True
-    codeact_enable_browsing_delegate: bool = True
+    codeact_enable_browsing: bool = True
     codeact_enable_llm_editor: bool = False
     codeact_enable_jupyter: bool = True
     micro_agent_name: str | None = None
diff --git a/openhands/core/config/app_config.py b/openhands/core/config/app_config.py
index a60c5070286f..6511f634983a 100644
--- a/openhands/core/config/app_config.py
+++ b/openhands/core/config/app_config.py
@@ -38,7 +38,6 @@ class AppConfig:
         e2b_api_key: The E2B API key.
         disable_color: Whether to disable color. For terminals that don't support color.
         debug: Whether to enable debugging.
-        enable_cli_session: Whether to enable saving and restoring the session when run from CLI.
         file_uploads_max_file_size_mb: Maximum file size for uploads in megabytes. 0 means no limit.
         file_uploads_restrict_file_types: Whether to restrict file types for file uploads. Defaults to False.
         file_uploads_allowed_extensions: List of allowed file extensions for uploads. ['.*'] means all extensions are allowed.
@@ -67,7 +66,6 @@ class AppConfig:
     disable_color: bool = False
     jwt_secret: str = uuid.uuid4().hex
     debug: bool = False
-    enable_cli_session: bool = False
     file_uploads_max_file_size_mb: int = 0
     file_uploads_restrict_file_types: bool = False
     file_uploads_allowed_extensions: list[str] = field(default_factory=lambda: ['.*'])
diff --git a/openhands/core/loop.py b/openhands/core/loop.py
new file mode 100644
index 000000000000..2a2808dd0980
--- /dev/null
+++ b/openhands/core/loop.py
@@ -0,0 +1,50 @@
+import asyncio
+
+from openhands.controller import AgentController
+from openhands.core.logger import openhands_logger as logger
+from openhands.core.schema import AgentState
+from openhands.runtime.base import Runtime
+
+
+async def run_agent_until_done(
+    controller: AgentController,
+    runtime: Runtime,
+    end_states: list[AgentState],
+):
+    """
+    run_agent_until_done takes a controller and a runtime, and will run
+    the agent until it reaches a terminal state.
+    Note that runtime must be connected before being passed in here.
+    """
+    controller.agent_task = asyncio.create_task(controller.start_step_loop())
+
+    def status_callback(msg_type, msg_id, msg):
+        if msg_type == 'error':
+            logger.error(msg)
+            if controller:
+                controller.state.last_error = msg
+                asyncio.create_task(controller.set_agent_state_to(AgentState.ERROR))
+        else:
+            logger.info(msg)
+
+    if hasattr(runtime, 'status_callback') and runtime.status_callback:
+        raise ValueError(
+            'Runtime status_callback was set, but run_agent_until_done will override it'
+        )
+    if hasattr(controller, 'status_callback') and controller.status_callback:
+        raise ValueError(
+            'Controller status_callback was set, but run_agent_until_done will override it'
+        )
+
+    runtime.status_callback = status_callback
+    controller.status_callback = status_callback
+
+    while controller.state.agent_state not in end_states:
+        await asyncio.sleep(1)
+
+    if not controller.agent_task.done():
+        controller.agent_task.cancel()
+        try:
+            await controller.agent_task
+        except asyncio.CancelledError:
+            pass
diff --git a/openhands/core/main.py b/openhands/core/main.py
index 0d653ea8b08f..5f6c27066c85 100644
--- a/openhands/core/main.py
+++ b/openhands/core/main.py
@@ -17,6 +17,7 @@
     parse_arguments,
 )
 from openhands.core.logger import openhands_logger as logger
+from openhands.core.loop import run_agent_until_done
 from openhands.core.schema import AgentState
 from openhands.events import EventSource, EventStream, EventStreamSubscriber
 from openhands.events.action import MessageAction
@@ -122,19 +123,20 @@ async def run_controller(
 
     if runtime is None:
         runtime = create_runtime(config, sid=sid)
-        await runtime.connect()
 
     event_stream = runtime.event_stream
-    # restore cli session if enabled
+
+    # restore cli session if available
     initial_state = None
-    if config.enable_cli_session:
-        try:
-            logger.debug(f'Restoring agent state from cli session {event_stream.sid}')
-            initial_state = State.restore_from_session(
-                event_stream.sid, event_stream.file_store
-            )
-        except Exception as e:
-            logger.debug(f'Error restoring state: {e}')
+    try:
+        logger.debug(
+            f'Trying to restore agent state from cli session {event_stream.sid} if available'
+        )
+        initial_state = State.restore_from_session(
+            event_stream.sid, event_stream.file_store
+        )
+    except Exception as e:
+        logger.debug(f'Cannot restore agent state: {e}')
 
     # init controller with this initial state
     controller = AgentController(
@@ -147,9 +149,6 @@ async def run_controller(
         headless_mode=headless_mode,
     )
 
-    if controller is not None:
-        controller.agent_task = asyncio.create_task(controller.start_step_loop())
-
     assert isinstance(
         initial_user_action, Action
     ), f'initial user actions must be an Action, got {type(initial_user_action)}'
@@ -160,7 +159,7 @@ async def run_controller(
     )
 
     # start event is a MessageAction with the task, either resumed or new
-    if config.enable_cli_session and initial_state is not None:
+    if initial_state is not None:
         # we're resuming the previous session
         event_stream.add_event(
             MessageAction(
@@ -171,7 +170,7 @@ async def run_controller(
             ),
             EventSource.USER,
         )
-    elif initial_state is None:
+    else:
         # init with the provided actions
         event_stream.add_event(initial_user_action, EventSource.USER)
 
@@ -187,33 +186,36 @@ async def on_event(event: Event):
                 action = MessageAction(content=message)
                 event_stream.add_event(action, EventSource.USER)
 
-    event_stream.subscribe(EventStreamSubscriber.MAIN, on_event)
-    while controller.state.agent_state not in [
+    event_stream.subscribe(EventStreamSubscriber.MAIN, on_event, sid)
+
+    await runtime.connect()
+
+    end_states = [
         AgentState.FINISHED,
         AgentState.REJECTED,
         AgentState.ERROR,
         AgentState.PAUSED,
         AgentState.STOPPED,
-    ]:
-        await asyncio.sleep(1)  # Give back control for a tick, so the agent can run
+    ]
+
+    try:
+        await run_agent_until_done(controller, runtime, end_states)
+    except Exception as e:
+        logger.error(f'Exception in main loop: {e}')
 
     # save session when we're about to close
-    if config.enable_cli_session:
+    if config.file_store is not None and config.file_store != 'memory':
         end_state = controller.get_state()
+        # NOTE: the saved state does not include delegates events
         end_state.save_to_session(event_stream.sid, event_stream.file_store)
 
-    # close when done
-    await controller.close()
     state = controller.get_state()
 
     # save trajectories if applicable
     if config.trajectories_path is not None:
         file_path = os.path.join(config.trajectories_path, sid + '.json')
         os.makedirs(os.path.dirname(file_path), exist_ok=True)
-        histories = [
-            event_to_trajectory(event)
-            for event in state.history.get_events(include_delegates=True)
-        ]
+        histories = [event_to_trajectory(event) for event in state.history]
         with open(file_path, 'w') as f:
             json.dump(histories, f)
 
diff --git a/openhands/core/message.py b/openhands/core/message.py
index 38568b504b4a..e538bec44bbe 100644
--- a/openhands/core/message.py
+++ b/openhands/core/message.py
@@ -49,6 +49,8 @@ def serialize_model(self):
 
 
 class Message(BaseModel):
+    # NOTE: this is not the same as EventSource
+    # These are the roles in the LLM's APIs
     role: Literal['user', 'system', 'assistant', 'tool']
     content: list[TextContent | ImageContent] = Field(default_factory=list)
     cache_enabled: bool = False
diff --git a/openhands/events/action/browse.py b/openhands/events/action/browse.py
index 441c86314056..41816216d6d5 100644
--- a/openhands/events/action/browse.py
+++ b/openhands/events/action/browse.py
@@ -36,7 +36,9 @@ class BrowseInteractiveAction(Action):
 
     @property
     def message(self) -> str:
-        return f'Executing browser actions: {self.browser_actions}'
+        return (
+            f'I am interacting with the browser:\n' f'```\n{self.browser_actions}\n```'
+        )
 
     def __str__(self) -> str:
         ret = '**BrowseInteractiveAction**\n'
diff --git a/openhands/events/action/message.py b/openhands/events/action/message.py
index 55fb21f359d3..0e3bb26a1cc2 100644
--- a/openhands/events/action/message.py
+++ b/openhands/events/action/message.py
@@ -7,7 +7,7 @@
 @dataclass
 class MessageAction(Action):
     content: str
-    images_urls: list | None = None
+    images_urls: list[str] | None = None
     wait_for_response: bool = False
     action: str = ActionType.MESSAGE
     security_risk: ActionSecurityRisk | None = None
diff --git a/openhands/events/event.py b/openhands/events/event.py
index 6ec68acc5586..126172bac793 100644
--- a/openhands/events/event.py
+++ b/openhands/events/event.py
@@ -9,6 +9,7 @@
 class EventSource(str, Enum):
     AGENT = 'agent'
     USER = 'user'
+    ENVIRONMENT = 'environment'
 
 
 @dataclass
diff --git a/openhands/events/observation/__init__.py b/openhands/events/observation/__init__.py
index a0fad86dfb49..28525b09aabb 100644
--- a/openhands/events/observation/__init__.py
+++ b/openhands/events/observation/__init__.py
@@ -6,7 +6,7 @@
 )
 from openhands.events.observation.delegate import AgentDelegateObservation
 from openhands.events.observation.empty import NullObservation
-from openhands.events.observation.error import ErrorObservation, FatalErrorObservation
+from openhands.events.observation.error import ErrorObservation
 from openhands.events.observation.files import (
     FileEditObservation,
     FileReadObservation,
@@ -26,7 +26,6 @@
     'FileWriteObservation',
     'FileEditObservation',
     'ErrorObservation',
-    'FatalErrorObservation',
     'AgentStateChangedObservation',
     'AgentDelegateObservation',
     'SuccessObservation',
diff --git a/openhands/events/observation/browse.py b/openhands/events/observation/browse.py
index 29daaefc4a30..9632fac57d54 100644
--- a/openhands/events/observation/browse.py
+++ b/openhands/events/observation/browse.py
@@ -1,5 +1,7 @@
 from dataclasses import dataclass, field
 
+from browsergym.utils.obs import flatten_axtree_to_str
+
 from openhands.core.schema import ObservationType
 from openhands.events.observation.observation import Observation
 
@@ -29,7 +31,7 @@ def message(self) -> str:
         return 'Visited ' + self.url
 
     def __str__(self) -> str:
-        return (
+        ret = (
             '**BrowserOutputObservation**\n'
             f'URL: {self.url}\n'
             f'Error: {self.error}\n'
@@ -38,5 +40,47 @@ def __str__(self) -> str:
             f'Last browser action: {self.last_browser_action}\n'
             f'Last browser action error: {self.last_browser_action_error}\n'
             f'Focused element bid: {self.focused_element_bid}\n'
-            f'CONTENT: {self.content}\n'
+            f'Content: {self.content}\n'
+        )
+        ret += '--- Agent Observation ---\n'
+        ret += self.get_agent_obs_text()
+        return ret
+
+    def get_agent_obs_text(self) -> str:
+        """Get a concise text that will be shown to the agent."""
+        text = f'[Current URL: {self.url}]\n'
+        text += f'[Focused element bid: {self.focused_element_bid}]\n\n'
+        if self.error:
+            text += (
+                '================ BEGIN error message ===============\n'
+                'The following error occurred when executing the last action:\n'
+                f'{self.last_browser_action_error}\n'
+                '================ END error message ===============\n'
+            )
+        else:
+            text += '[Action executed successfully.]\n'
+
+        try:
+            # We do not filter visible only here because we want to show the full content
+            # of the web page to the agent for simplicity.
+            # FIXME: handle the case when the web page is too large
+            cur_axtree_txt = self.get_axtree_str(filter_visible_only=False)
+            text += (
+                f'============== BEGIN accessibility tree ==============\n'
+                f'{cur_axtree_txt}\n'
+                f'============== END accessibility tree ==============\n'
+            )
+        except Exception as e:
+            text += f'\n[Error encountered when processing the accessibility tree: {e}]'
+        return text
+
+    def get_axtree_str(self, filter_visible_only: bool = False) -> str:
+        cur_axtree_txt = flatten_axtree_to_str(
+            self.axtree_object,
+            extra_properties=self.extra_element_properties,
+            with_clickable=True,
+            skip_generic=False,
+            filter_visible_only=filter_visible_only,
         )
+        self._axtree_str = cur_axtree_txt
+        return cur_axtree_txt
diff --git a/openhands/events/observation/error.py b/openhands/events/observation/error.py
index cfbb291eb0fa..4ed05b89ac78 100644
--- a/openhands/events/observation/error.py
+++ b/openhands/events/observation/error.py
@@ -13,6 +13,7 @@ class ErrorObservation(Observation):
     """
 
     observation: str = ObservationType.ERROR
+    error_id: str = ''
 
     @property
     def message(self) -> str:
@@ -20,17 +21,3 @@ def message(self) -> str:
 
     def __str__(self) -> str:
         return f'**ErrorObservation**\n{self.content}'
-
-
-@dataclass
-class FatalErrorObservation(Observation):
-    """This data class represents a fatal error encountered by the agent.
-
-    This is the type of error that LLM CANNOT recover from, and the agent controller should stop the execution and report the error to the user.
-    E.g., Remote runtime action execution failure: 503 Server Error: Service Unavailable for url OR 404 Not Found.
-    """
-
-    observation: str = ObservationType.ERROR
-
-    def __str__(self) -> str:
-        return f'**FatalErrorObservation**\n{self.content}'
diff --git a/openhands/events/stream.py b/openhands/events/stream.py
index aafbcc2fc874..625abd7a5221 100644
--- a/openhands/events/stream.py
+++ b/openhands/events/stream.py
@@ -11,32 +11,49 @@
 from openhands.events.serialization.event import event_from_dict, event_to_dict
 from openhands.runtime.utils.shutdown_listener import should_continue
 from openhands.storage import FileStore
+from openhands.utils.async_utils import call_sync_from_async
 
 
 class EventStreamSubscriber(str, Enum):
     AGENT_CONTROLLER = 'agent_controller'
     SECURITY_ANALYZER = 'security_analyzer'
+    RESOLVER = 'openhands_resolver'
     SERVER = 'server'
     RUNTIME = 'runtime'
     MAIN = 'main'
     TEST = 'test'
 
 
-def session_exists(sid: str, file_store: FileStore) -> bool:
+async def session_exists(sid: str, file_store: FileStore) -> bool:
     try:
-        file_store.list(f'sessions/{sid}')
+        await call_sync_from_async(file_store.list, f'sessions/{sid}')
         return True
     except FileNotFoundError:
         return False
 
 
+class AsyncEventStreamWrapper:
+    def __init__(self, event_stream, *args, **kwargs):
+        self.event_stream = event_stream
+        self.args = args
+        self.kwargs = kwargs
+
+    async def __aiter__(self):
+        loop = asyncio.get_running_loop()
+
+        # Create an async generator that yields events
+        for event in self.event_stream.get_events(*self.args, **self.kwargs):
+            # Run the blocking get_events() in a thread pool
+            yield await loop.run_in_executor(None, lambda e=event: e)  # type: ignore
+
+
 @dataclass
 class EventStream:
     sid: str
     file_store: FileStore
-    # For each subscriber ID, there is a stack of callback functions - useful
-    # when there are agent delegates
-    _subscribers: dict[str, list[Callable]] = field(default_factory=dict)
+    # For each subscriber ID, there is a map of callback functions - useful
+    # when there are multiple listeners
+    _subscribers: dict[str, dict[str, Callable]] = field(default_factory=dict)
     _cur_id: int = 0
     _lock: threading.Lock = field(default_factory=threading.Lock)
 
@@ -67,12 +84,27 @@ def _get_id_from_filename(filename: str) -> int:
 
     def get_events(
         self,
-        start_id=0,
-        end_id=None,
-        reverse=False,
+        start_id: int = 0,
+        end_id: int | None = None,
+        reverse: bool = False,
         filter_out_type: tuple[type[Event], ...] | None = None,
         filter_hidden=False,
     ) -> Iterable[Event]:
+        """
+        Retrieve events from the event stream, optionally filtering out events of a given type
+        and events marked as hidden.
+
+        Args:
+            start_id: The ID of the first event to retrieve. Defaults to 0.
+            end_id: The ID of the last event to retrieve. Defaults to the last event in the stream.
+            reverse: Whether to retrieve events in reverse order. Defaults to False.
+            filter_out_type: A tuple of event types to filter out. Typically used to filter out backend events from the agent.
+            filter_hidden: If True, filters out events with the 'hidden' attribute set to True.
+
+        Yields:
+            Events from the stream that match the criteria.
+        """
+
         def should_filter(event: Event):
             if filter_hidden and hasattr(event, 'hidden') and event.hidden:
                 return True
@@ -117,31 +149,42 @@ def get_latest_event(self) -> Event:
     def get_latest_event_id(self) -> int:
         return self._cur_id - 1
 
-    def subscribe(self, id: EventStreamSubscriber, callback: Callable, append=False):
-        if id in self._subscribers:
-            if append:
-                self._subscribers[id].append(callback)
-            else:
-                raise ValueError('Subscriber already exists: ' + id)
-        else:
-            self._subscribers[id] = [callback]
+    def subscribe(
+        self, subscriber_id: EventStreamSubscriber, callback: Callable, callback_id: str
+    ):
+        if subscriber_id not in self._subscribers:
+            self._subscribers[subscriber_id] = {}
 
-    def unsubscribe(self, id: EventStreamSubscriber):
-        if id not in self._subscribers:
-            logger.warning('Subscriber not found during unsubscribe: ' + id)
-        else:
-            self._subscribers[id].pop()
-            if len(self._subscribers[id]) == 0:
-                del self._subscribers[id]
+        if callback_id in self._subscribers[subscriber_id]:
+            raise ValueError(
+                f'Callback ID on subscriber {subscriber_id} already exists: {callback_id}'
+            )
+
+        self._subscribers[subscriber_id][callback_id] = callback
+
+    def unsubscribe(self, subscriber_id: EventStreamSubscriber, callback_id: str):
+        if subscriber_id not in self._subscribers:
+            logger.warning(f'Subscriber not found during unsubscribe: {subscriber_id}')
+            return
+
+        if callback_id not in self._subscribers[subscriber_id]:
+            logger.warning(f'Callback not found during unsubscribe: {callback_id}')
+            return
+
+        del self._subscribers[subscriber_id][callback_id]
 
     def add_event(self, event: Event, source: EventSource):
         try:
-            asyncio.get_running_loop().create_task(self.async_add_event(event, source))
+            asyncio.get_running_loop().create_task(self._async_add_event(event, source))
         except RuntimeError:
             # No event loop running...
-            asyncio.run(self.async_add_event(event, source))
+            asyncio.run(self._async_add_event(event, source))
 
-    async def async_add_event(self, event: Event, source: EventSource):
+    async def _async_add_event(self, event: Event, source: EventSource):
+        if hasattr(event, '_id') and event.id is not None:
+            raise ValueError(
+                'Event already has an ID. It was probably added back to the EventStream from inside a handler, trigging a loop.'
+            )
         with self._lock:
             event._id = self._cur_id  # type: ignore [attr-defined]
             self._cur_id += 1
@@ -153,9 +196,10 @@ async def async_add_event(self, event: Event, source: EventSource):
             self.file_store.write(self._get_filename_for_id(event.id), json.dumps(data))
         tasks = []
         for key in sorted(self._subscribers.keys()):
-            stack = self._subscribers[key]
-            callback = stack[-1]
-            tasks.append(asyncio.create_task(callback(event)))
+            callbacks = self._subscribers[key]
+            for callback_id in callbacks:
+                callback = callbacks[callback_id]
+                tasks.append(asyncio.create_task(callback(event)))
         if tasks:
             await asyncio.wait(tasks)
 
diff --git a/openhands/llm/llm.py b/openhands/llm/llm.py
index 5fbd8da9c8d7..8327569dff61 100644
--- a/openhands/llm/llm.py
+++ b/openhands/llm/llm.py
@@ -49,6 +49,7 @@
 CACHE_PROMPT_SUPPORTED_MODELS = [
     'claude-3-5-sonnet-20241022',
     'claude-3-5-sonnet-20240620',
+    'claude-3-5-haiku-20241022',
     'claude-3-haiku-20240307',
     'claude-3-opus-20240229',
 ]
@@ -57,6 +58,7 @@
 FUNCTION_CALLING_SUPPORTED_MODELS = [
     'claude-3-5-sonnet-20240620',
     'claude-3-5-sonnet-20241022',
+    'claude-3-5-haiku-20241022',
     'gpt-4o',
     'gpt-4o-mini',
 ]
@@ -82,6 +84,7 @@ def __init__(
             config: The LLM configuration.
             metrics: The metrics to use.
         """
+        self._tried_model_info = False
         self.metrics: Metrics = (
             metrics if metrics is not None else Metrics(model_name=config.model)
         )
@@ -91,56 +94,6 @@ def __init__(
         # litellm actually uses base Exception here for unknown model
         self.model_info: ModelInfo | None = None
 
-        try:
-            if self.config.model.startswith('openrouter'):
-                self.model_info = litellm.get_model_info(self.config.model)
-        except Exception as e:
-            logger.debug(f'Error getting model info: {e}')
-
-        if self.config.model.startswith('litellm_proxy/'):
-            # IF we are using LiteLLM proxy, get model info from LiteLLM proxy
-            # GET {base_url}/v1/model/info with litellm_model_id as path param
-            response = requests.get(
-                f'{self.config.base_url}/v1/model/info',
-                headers={'Authorization': f'Bearer {self.config.api_key}'},
-            )
-            resp_json = response.json()
-            if 'data' not in resp_json:
-                logger.error(
-                    f'Error getting model info from LiteLLM proxy: {resp_json}'
-                )
-            all_model_info = resp_json.get('data', [])
-            current_model_info = next(
-                (
-                    info
-                    for info in all_model_info
-                    if info['model_name']
-                    == self.config.model.removeprefix('litellm_proxy/')
-                ),
-                None,
-            )
-            if current_model_info:
-                self.model_info = current_model_info['model_info']
-
-        # Last two attempts to get model info from NAME
-        if not self.model_info:
-            try:
-                self.model_info = litellm.get_model_info(
-                    self.config.model.split(':')[0]
-                )
-            # noinspection PyBroadException
-            except Exception:
-                pass
-        if not self.model_info:
-            try:
-                self.model_info = litellm.get_model_info(
-                    self.config.model.split('/')[-1]
-                )
-            # noinspection PyBroadException
-            except Exception:
-                pass
-        logger.debug(f'Model info: {self.model_info}')
-
         if self.config.log_completions:
             if self.config.log_completions_folder is None:
                 raise RuntimeError(
@@ -148,32 +101,26 @@ def __init__(
                 )
             os.makedirs(self.config.log_completions_folder, exist_ok=True)
 
-        # Set the max tokens in an LM-specific way if not set
-        if self.config.max_input_tokens is None:
-            if (
-                self.model_info is not None
-                and 'max_input_tokens' in self.model_info
-                and isinstance(self.model_info['max_input_tokens'], int)
-            ):
-                self.config.max_input_tokens = self.model_info['max_input_tokens']
-            else:
-                # Safe fallback for any potentially viable model
-                self.config.max_input_tokens = 4096
+        self._completion = partial(
+            litellm_completion,
+            model=self.config.model,
+            api_key=self.config.api_key,
+            base_url=self.config.base_url,
+            api_version=self.config.api_version,
+            custom_llm_provider=self.config.custom_llm_provider,
+            max_tokens=self.config.max_output_tokens,
+            timeout=self.config.timeout,
+            temperature=self.config.temperature,
+            top_p=self.config.top_p,
+            drop_params=self.config.drop_params,
+        )
 
-        if self.config.max_output_tokens is None:
-            # Safe default for any potentially viable model
-            self.config.max_output_tokens = 4096
-            if self.model_info is not None:
-                # max_output_tokens has precedence over max_tokens, if either exists.
-                # litellm has models with both, one or none of these 2 parameters!
-                if 'max_output_tokens' in self.model_info and isinstance(
-                    self.model_info['max_output_tokens'], int
-                ):
-                    self.config.max_output_tokens = self.model_info['max_output_tokens']
-                elif 'max_tokens' in self.model_info and isinstance(
-                    self.model_info['max_tokens'], int
-                ):
-                    self.config.max_output_tokens = self.model_info['max_tokens']
+        if self.vision_is_active():
+            logger.debug('LLM: model has vision enabled')
+        if self.is_caching_prompt_active():
+            logger.debug('LLM: caching prompt enabled')
+        if self.is_function_calling_active():
+            logger.debug('LLM: model supports function calling')
 
         self._completion = partial(
             litellm_completion,
@@ -207,6 +154,7 @@ def __init__(
         )
         def wrapper(*args, **kwargs):
             """Wrapper for the litellm completion function. Logs the input and output of the completion function."""
+            self.init_model_info()
             messages: list[dict[str, Any]] | dict[str, Any] = []
 
             # some callers might send the model and messages directly
@@ -300,6 +248,87 @@ def completion(self):
         """
         return self._completion
 
+    def init_model_info(self):
+        if self._tried_model_info:
+            return
+        self._tried_model_info = True
+        try:
+            if self.config.model.startswith('openrouter'):
+                self.model_info = litellm.get_model_info(self.config.model)
+        except Exception as e:
+            logger.debug(f'Error getting model info: {e}')
+
+        if self.config.model.startswith('litellm_proxy/'):
+            # IF we are using LiteLLM proxy, get model info from LiteLLM proxy
+            # GET {base_url}/v1/model/info with litellm_model_id as path param
+            response = requests.get(
+                f'{self.config.base_url}/v1/model/info',
+                headers={'Authorization': f'Bearer {self.config.api_key}'},
+            )
+            resp_json = response.json()
+            if 'data' not in resp_json:
+                logger.error(
+                    f'Error getting model info from LiteLLM proxy: {resp_json}'
+                )
+            all_model_info = resp_json.get('data', [])
+            current_model_info = next(
+                (
+                    info
+                    for info in all_model_info
+                    if info['model_name']
+                    == self.config.model.removeprefix('litellm_proxy/')
+                ),
+                None,
+            )
+            if current_model_info:
+                self.model_info = current_model_info['model_info']
+
+        # Last two attempts to get model info from NAME
+        if not self.model_info:
+            try:
+                self.model_info = litellm.get_model_info(
+                    self.config.model.split(':')[0]
+                )
+            # noinspection PyBroadException
+            except Exception:
+                pass
+        if not self.model_info:
+            try:
+                self.model_info = litellm.get_model_info(
+                    self.config.model.split('/')[-1]
+                )
+            # noinspection PyBroadException
+            except Exception:
+                pass
+        logger.debug(f'Model info: {self.model_info}')
+
+        # Set the max tokens in an LM-specific way if not set
+        if self.config.max_input_tokens is None:
+            if (
+                self.model_info is not None
+                and 'max_input_tokens' in self.model_info
+                and isinstance(self.model_info['max_input_tokens'], int)
+            ):
+                self.config.max_input_tokens = self.model_info['max_input_tokens']
+            else:
+                # Safe fallback for any potentially viable model
+                self.config.max_input_tokens = 4096
+
+        if self.config.max_output_tokens is None:
+            # Safe default for any potentially viable model
+            self.config.max_output_tokens = 4096
+            if self.model_info is not None:
+                # max_output_tokens has precedence over max_tokens, if either exists.
+                # litellm has models with both, one or none of these 2 parameters!
+                if 'max_output_tokens' in self.model_info and isinstance(
+                    self.model_info['max_output_tokens'], int
+                ):
+                    self.config.max_output_tokens = self.model_info['max_output_tokens']
+                elif 'max_tokens' in self.model_info and isinstance(
+                    self.model_info['max_tokens'], int
+                ):
+                    self.config.max_output_tokens = self.model_info['max_tokens']
+
     def vision_is_active(self):
         return not self.config.disable_vision and self._supports_vision()
 
@@ -324,14 +353,15 @@ def is_caching_prompt_active(self) -> bool:
         Returns:
             boolean: True if prompt caching is supported and enabled for the given model.
         """
-        return (
-            self.config.caching_prompt is True
-            and self.model_info is not None
-            and self.model_info.get('supports_prompt_caching', False)
-            and (
+        return self.config.caching_prompt is True and (
+            (
                 self.config.model in CACHE_PROMPT_SUPPORTED_MODELS
                 or self.config.model.split('/')[-1] in CACHE_PROMPT_SUPPORTED_MODELS
             )
+            or (
+                self.model_info is not None
+                and self.model_info.get('supports_prompt_caching', False)
+            )
         )
 
     def is_function_calling_active(self) -> bool:
@@ -341,7 +371,7 @@ def is_function_calling_active(self) -> bool:
             or self.config.model.split('/')[-1] in FUNCTION_CALLING_SUPPORTED_MODELS
             or any(m in self.config.model for m in FUNCTION_CALLING_SUPPORTED_MODELS)
         )
-        return model_name_supported and (
+        return model_name_supported or (
             self.model_info is not None
             and self.model_info.get('supports_function_calling', False)
         )
diff --git a/openhands/memory/__init__.py b/openhands/memory/__init__.py
index 0ce208cef581..12c499c768be 100644
--- a/openhands/memory/__init__.py
+++ b/openhands/memory/__init__.py
@@ -1,5 +1,4 @@
 from openhands.memory.condenser import MemoryCondenser
-from openhands.memory.history import ShortTermHistory
 from openhands.memory.memory import LongTermMemory
 
-__all__ = ['LongTermMemory', 'ShortTermHistory', 'MemoryCondenser']
+__all__ = ['LongTermMemory', 'MemoryCondenser']
diff --git a/openhands/memory/history.py b/openhands/memory/history.py
deleted file mode 100644
index 1e4cfb8b5f05..000000000000
--- a/openhands/memory/history.py
+++ /dev/null
@@ -1,224 +0,0 @@
-from typing import ClassVar, Iterable
-
-from openhands.core.logger import openhands_logger as logger
-from openhands.events.action.action import Action
-from openhands.events.action.agent import (
-    AgentDelegateAction,
-    ChangeAgentStateAction,
-)
-from openhands.events.action.empty import NullAction
-from openhands.events.action.message import MessageAction
-from openhands.events.event import Event, EventSource
-from openhands.events.observation.agent import AgentStateChangedObservation
-from openhands.events.observation.delegate import AgentDelegateObservation
-from openhands.events.observation.empty import NullObservation
-from openhands.events.observation.observation import Observation
-from openhands.events.serialization.event import event_to_dict
-from openhands.events.stream import EventStream
-from openhands.events.utils import get_pairs_from_events
-
-
-class ShortTermHistory(list[Event]):
-    """A list of events that represents the short-term memory of the agent.
-
-    This class provides methods to retrieve and filter the events in the history of the running agent from the event stream.
-    """
-
-    start_id: int
-    end_id: int
-    _event_stream: EventStream
-    delegates: dict[tuple[int, int], tuple[str, str]]
-    filter_out: ClassVar[tuple[type[Event], ...]] = (
-        NullAction,
-        NullObservation,
-        ChangeAgentStateAction,
-        AgentStateChangedObservation,
-    )
-
-    def __init__(self):
-        super().__init__()
-        self.start_id = -1
-        self.end_id = -1
-        self.delegates = {}
-
-    def set_event_stream(self, event_stream: EventStream):
-        self._event_stream = event_stream
-
-    def get_events_as_list(self, include_delegates: bool = False) -> list[Event]:
-        """Return the history as a list of Event objects."""
-        return list(self.get_events(include_delegates=include_delegates))
-
-    def get_events(
-        self,
-        reverse: bool = False,
-        include_delegates: bool = False,
-        include_hidden=False,
-    ) -> Iterable[Event]:
-        """Return the events as a stream of Event objects."""
-        # TODO handle AgentRejectAction, if it's not part of a chunk ending with an AgentDelegateObservation
-        # or even if it is, because currently we don't add it to the summary
-
-        # iterate from start_id to end_id, or reverse
-        start_id = self.start_id if self.start_id != -1 else 0
-        end_id = (
-            self.end_id
-            if self.end_id != -1
-            else self._event_stream.get_latest_event_id()
-        )
-
-        for event in self._event_stream.get_events(
-            start_id=start_id,
-            end_id=end_id,
-            reverse=reverse,
-            filter_out_type=self.filter_out,
-        ):
-            if not include_hidden and hasattr(event, 'hidden') and event.hidden:
-                continue
-            # TODO add summaries
-            # and filter out events that were included in a summary
-
-            # filter out the events from a delegate of the current agent
-            if not include_delegates and not any(
-                # except for the delegate action and observation themselves, currently
-                # AgentDelegateAction has id = delegate_start
-                # AgentDelegateObservation has id = delegate_end
-                delegate_start < event.id < delegate_end
-                for delegate_start, delegate_end in self.delegates.keys()
-            ):
-                yield event
-            elif include_delegates:
-                yield event
-
-    def get_last_action(self, end_id: int = -1) -> Action | None:
-        """Return the last action from the event stream, filtered to exclude unwanted events."""
-        # from end_id in reverse, find the first action
-        end_id = self._event_stream.get_latest_event_id() if end_id == -1 else end_id
-
-        last_action = next(
-            (
-                event
-                for event in self._event_stream.get_events(
-                    end_id=end_id, reverse=True, filter_out_type=self.filter_out
-                )
-                if isinstance(event, Action)
-            ),
-            None,
-        )
-
-        return last_action
-
-    def get_last_observation(self, end_id: int = -1) -> Observation | None:
-        """Return the last observation from the event stream, filtered to exclude unwanted events."""
-        # from end_id in reverse, find the first observation
-        end_id = self._event_stream.get_latest_event_id() if end_id == -1 else end_id
-
-        last_observation = next(
-            (
-                event
-                for event in self._event_stream.get_events(
-                    end_id=end_id, reverse=True, filter_out_type=self.filter_out
-                )
-                if isinstance(event, Observation)
-            ),
-            None,
-        )
-
-        return last_observation
-
-    def get_last_user_message(self) -> str:
-        """Return the content of the last user message from the event stream."""
-        last_user_message = next(
-            (
-                event.content
-                for event in self._event_stream.get_events(reverse=True)
-                if isinstance(event, MessageAction) and event.source == EventSource.USER
-            ),
-            None,
-        )
-
-        return last_user_message if last_user_message is not None else ''
-
-    def get_last_agent_message(self) -> str:
-        """Return the content of the last agent message from the event stream."""
-        last_agent_message = next(
-            (
-                event.content
-                for event in self._event_stream.get_events(reverse=True)
-                if isinstance(event, MessageAction)
-                and event.source == EventSource.AGENT
-            ),
-            None,
-        )
-
-        return last_agent_message if last_agent_message is not None else ''
-
-    def get_last_events(self, n: int) -> list[Event]:
-        """Return the last n events from the event stream."""
-        # dummy agent is using this
-        # it should work, but it's not great to store temporary lists now just for a test
-        end_id = self._event_stream.get_latest_event_id()
-        start_id = max(0, end_id - n + 1)
-
-        return list(
-            event
-            for event in self._event_stream.get_events(
-                start_id=start_id,
-                end_id=end_id,
-                filter_out_type=self.filter_out,
-            )
-        )
-
-    def has_delegation(self) -> bool:
-        for event in self._event_stream.get_events():
-            if isinstance(event, AgentDelegateObservation):
-                return True
-        return False
-
-    def on_event(self, event: Event):
-        if not isinstance(event, AgentDelegateObservation):
-            return
-
-        logger.debug('AgentDelegateObservation received')
-
-        # figure out what this delegate's actions were
-        # from the last AgentDelegateAction to this AgentDelegateObservation
-        # and save their ids as start and end ids
-        # in order to use later to exclude them from parent stream
-        # or summarize them
-        delegate_end = event.id
-        delegate_start = -1
-        delegate_agent: str = ''
-        delegate_task: str = ''
-        for prev_event in self._event_stream.get_events(
-            end_id=event.id - 1, reverse=True
-        ):
-            if isinstance(prev_event, AgentDelegateAction):
-                delegate_start = prev_event.id
-                delegate_agent = prev_event.agent
-                delegate_task = prev_event.inputs.get('task', '')
-                break
-
-        if delegate_start == -1:
-            logger.error(
-                f'No AgentDelegateAction found for AgentDelegateObservation with id={delegate_end}'
-            )
-            return
-
-        self.delegates[(delegate_start, delegate_end)] = (delegate_agent, delegate_task)
-        logger.debug(
-            f'Delegate {delegate_agent} with task {delegate_task} ran from id={delegate_start} to id={delegate_end}'
-        )
-
-    # TODO remove me when unnecessary
-    # history is now available as a filtered stream of events, rather than list of pairs of (Action, Observation)
-    # we rebuild the pairs here
-    # for compatibility with the existing output format in evaluations
-    def compatibility_for_eval_history_pairs(self) -> list[tuple[dict, dict]]:
-        history_pairs = []
-
-        for action, observation in get_pairs_from_events(
-            self.get_events_as_list(include_delegates=True)
-        ):
-            history_pairs.append((event_to_dict(action), event_to_dict(observation)))
-
-        return history_pairs
diff --git a/openhands/runtime/action_execution_server.py b/openhands/runtime/action_execution_server.py
index 8b263dfbe30e..883ea8e95401 100644
--- a/openhands/runtime/action_execution_server.py
+++ b/openhands/runtime/action_execution_server.py
@@ -37,7 +37,6 @@
 from openhands.events.observation import (
     CmdOutputObservation,
     ErrorObservation,
-    FatalErrorObservation,
     FileReadObservation,
     FileWriteObservation,
     IPythonRunCellObservation,
@@ -168,7 +167,7 @@ async def run_action(self, action) -> Observation:
 
     async def run(
         self, action: CmdRunAction
-    ) -> CmdOutputObservation | FatalErrorObservation:
+    ) -> CmdOutputObservation | ErrorObservation:
         return self.bash_session.run(action)
 
     async def run_ipython(self, action: IPythonRunCellAction) -> Observation:
diff --git a/openhands/runtime/base.py b/openhands/runtime/base.py
index 474ba741a476..94dfeb3f5b5d 100644
--- a/openhands/runtime/base.py
+++ b/openhands/runtime/base.py
@@ -5,6 +5,8 @@
 from abc import abstractmethod
 from typing import Callable
 
+from requests.exceptions import ConnectionError
+
 from openhands.core.config import AppConfig, SandboxConfig
 from openhands.core.logger import openhands_logger as logger
 from openhands.events import EventSource, EventStream, EventStreamSubscriber
@@ -31,6 +33,22 @@
 from openhands.runtime.utils.edit import FileEditRuntimeMixin
 from openhands.utils.async_utils import call_sync_from_async
 
+STATUS_MESSAGES = {
+    'STATUS$STARTING_RUNTIME': 'Starting runtime...',
+    'STATUS$STARTING_CONTAINER': 'Starting container...',
+    'STATUS$PREPARING_CONTAINER': 'Preparing container...',
+    'STATUS$CONTAINER_STARTED': 'Container started.',
+    'STATUS$WAITING_FOR_CLIENT': 'Waiting for client...',
+}
+
+
+class RuntimeNotReadyError(Exception):
+    pass
+
+
+class RuntimeDisconnectedError(Exception):
+    pass
+
 
 def _default_env_vars(sandbox_config: SandboxConfig) -> dict[str, str]:
     ret = {}
@@ -54,6 +72,7 @@ class Runtime(FileEditRuntimeMixin):
     config: AppConfig
     initial_env_vars: dict[str, str]
     attach_to_existing: bool
+    status_callback: Callable | None
 
     def __init__(
         self,
@@ -62,14 +81,16 @@ def __init__(
         sid: str = 'default',
         plugins: list[PluginRequirement] | None = None,
         env_vars: dict[str, str] | None = None,
-        status_message_callback: Callable | None = None,
+        status_callback: Callable | None = None,
         attach_to_existing: bool = False,
     ):
         self.sid = sid
         self.event_stream = event_stream
-        self.event_stream.subscribe(EventStreamSubscriber.RUNTIME, self.on_event)
+        self.event_stream.subscribe(
+            EventStreamSubscriber.RUNTIME, self.on_event, self.sid
+        )
         self.plugins = plugins if plugins is not None and len(plugins) > 0 else []
-        self.status_message_callback = status_message_callback
+        self.status_callback = status_callback
         self.attach_to_existing = attach_to_existing
 
         self.config = copy.deepcopy(config)
@@ -95,7 +116,17 @@ def close(self) -> None:
 
     def log(self, level: str, message: str) -> None:
         message = f'[runtime {self.sid}] {message}'
-        getattr(logger, level)(message)
+        getattr(logger, level)(message, stacklevel=2)
+
+    def send_status_message(self, message_id: str):
+        """Sends a status message if the callback function was provided."""
+        if self.status_callback:
+            msg = STATUS_MESSAGES.get(message_id, '')
+            self.status_callback('info', message_id, msg)
+
+    def send_error_message(self, message_id: str, message: str):
+        if self.status_callback:
+            self.status_callback('error', message_id, message)
 
     # ====================================================================
 
@@ -131,13 +162,28 @@ async def on_event(self, event: Event) -> None:
             if event.timeout is None:
                 event.timeout = self.config.sandbox.timeout
             assert event.timeout is not None
-            observation: Observation = await call_sync_from_async(
-                self.run_action, event
-            )
+            try:
+                observation: Observation = await call_sync_from_async(
+                    self.run_action, event
+                )
+            except Exception as e:
+                err_id = ''
+                if isinstance(e, ConnectionError) or isinstance(
+                    e, RuntimeDisconnectedError
+                ):
+                    err_id = 'STATUS$ERROR_RUNTIME_DISCONNECTED'
+                self.log('error', f'Unexpected error while running action {e}')
+                self.log('error', f'Problematic action: {str(event)}')
+                self.send_error_message(err_id, str(e))
+                self.close()
+                return
+
             observation._cause = event.id  # type: ignore[attr-defined]
             observation.tool_call_metadata = event.tool_call_metadata
+
+            # this might be unnecessary, since source should be set by the event stream when we're here
             source = event.source if event.source else EventSource.AGENT
-            await self.event_stream.async_add_event(observation, source)  # type: ignore[arg-type]
+            self.event_stream.add_event(observation, source)  # type: ignore[arg-type]
 
     def run_action(self, action: Action) -> Observation:
         """Run an action and return the resulting observation.
diff --git a/openhands/runtime/browser/browser_env.py b/openhands/runtime/browser/browser_env.py
index 2ca48b8db37b..9bad97b9bb2b 100644
--- a/openhands/runtime/browser/browser_env.py
+++ b/openhands/runtime/browser/browser_env.py
@@ -81,7 +81,10 @@ def browser_process(self):
                 raise ValueError(
                     f'Unsupported browsergym eval env: {self.browsergym_eval_env}'
                 )
-            env = gym.make(self.browsergym_eval_env)
+            env = gym.make(
+                self.browsergym_eval_env,
+                tags_to_mark='all',
+            )
         else:
             env = gym.make(
                 'browsergym/openended',
@@ -89,6 +92,7 @@ def browser_process(self):
                 wait_for_user_message=False,
                 headless=True,
                 disable_env_checker=True,
+                tags_to_mark='all',
             )
 
         obs, info = env.reset()
diff --git a/openhands/runtime/builder/docker.py b/openhands/runtime/builder/docker.py
index a630f3975dcf..a3cb5af39f3d 100644
--- a/openhands/runtime/builder/docker.py
+++ b/openhands/runtime/builder/docker.py
@@ -17,7 +17,7 @@ def __init__(self, docker_client: docker.DockerClient):
 
         version_info = self.docker_client.version()
         server_version = version_info.get('Version', '').replace('-', '.')
-        if tuple(map(int, server_version.split('.'))) < (18, 9):
+        if tuple(map(int, server_version.split('.')[:2])) < (18, 9):
             raise RuntimeError('Docker server version must be >= 18.09 to use BuildKit')
 
         self.rolling_logger = RollingLogger(max_lines=10)
diff --git a/openhands/runtime/builder/remote.py b/openhands/runtime/builder/remote.py
index f96afb38eeb2..b1b14752cb89 100644
--- a/openhands/runtime/builder/remote.py
+++ b/openhands/runtime/builder/remote.py
@@ -7,7 +7,7 @@
 
 from openhands.core.logger import openhands_logger as logger
 from openhands.runtime.builder import RuntimeBuilder
-from openhands.runtime.utils.request import is_429_error, send_request_with_retry
+from openhands.runtime.utils.request import send_request
 from openhands.runtime.utils.shutdown_listener import (
     should_continue,
     sleep_if_should_continue,
@@ -45,18 +45,21 @@ def build(self, path: str, tags: list[str], platform: str | None = None) -> str:
             files.append(('tags', (None, tag)))
 
         # Send the POST request to /build (Begins the build process)
-        response = send_request_with_retry(
-            self.session,
-            'POST',
-            f'{self.api_url}/build',
-            files=files,
-            timeout=30,
-            retry_fns=[is_429_error],
-        )
-
-        if response.status_code != 202:
-            logger.error(f'Build initiation failed: {response.text}')
-            raise RuntimeError(f'Build initiation failed: {response.text}')
+        try:
+            response = send_request(
+                self.session,
+                'POST',
+                f'{self.api_url}/build',
+                files=files,
+                timeout=30,
+            )
+        except requests.exceptions.HTTPError as e:
+            if e.response.status_code == 429:
+                logger.warning('Build was rate limited. Retrying in 30 seconds.')
+                time.sleep(30)
+                return self.build(path, tags, platform)
+            else:
+                raise e
 
         build_data = response.json()
         build_id = build_data['build_id']
@@ -70,12 +73,11 @@ def build(self, path: str, tags: list[str], platform: str | None = None) -> str:
                 logger.error('Build timed out after 30 minutes')
                 raise RuntimeError('Build timed out after 30 minutes')
 
-            status_response = send_request_with_retry(
+            status_response = send_request(
                 self.session,
                 'GET',
                 f'{self.api_url}/build_status',
                 params={'build_id': build_id},
-                timeout=30,
             )
 
             if status_response.status_code != 200:
@@ -112,12 +114,11 @@ def build(self, path: str, tags: list[str], platform: str | None = None) -> str:
     def image_exists(self, image_name: str, pull_from_repo: bool = True) -> bool:
         """Checks if an image exists in the remote registry using the /image_exists endpoint."""
         params = {'image': image_name}
-        response = send_request_with_retry(
+        response = send_request(
             self.session,
             'GET',
             f'{self.api_url}/image_exists',
             params=params,
-            timeout=30,
         )
 
         if response.status_code != 200:
diff --git a/openhands/runtime/impl/e2b/e2b_runtime.py b/openhands/runtime/impl/e2b/e2b_runtime.py
index b5233574f0b2..7c9c297f424c 100644
--- a/openhands/runtime/impl/e2b/e2b_runtime.py
+++ b/openhands/runtime/impl/e2b/e2b_runtime.py
@@ -27,14 +27,14 @@ def __init__(
         sid: str = 'default',
         plugins: list[PluginRequirement] | None = None,
         sandbox: E2BSandbox | None = None,
-        status_message_callback: Optional[Callable] = None,
+        status_callback: Optional[Callable] = None,
     ):
         super().__init__(
             config,
             event_stream,
             sid,
             plugins,
-            status_message_callback=status_message_callback,
+            status_callback=status_callback,
         )
         if sandbox is None:
             self.sandbox = E2BSandbox()
diff --git a/openhands/runtime/impl/e2b/sandbox.py b/openhands/runtime/impl/e2b/sandbox.py
index 666dc43f701c..d145dac35115 100644
--- a/openhands/runtime/impl/e2b/sandbox.py
+++ b/openhands/runtime/impl/e2b/sandbox.py
@@ -4,9 +4,7 @@
 from glob import glob
 
 from e2b import Sandbox as E2BSandbox
-from e2b.sandbox.exception import (
-    TimeoutException,
-)
+from e2b.sandbox.exception import TimeoutException
 
 from openhands.core.config import SandboxConfig
 from openhands.core.logger import openhands_logger as logger
diff --git a/openhands/runtime/impl/eventstream/eventstream_runtime.py b/openhands/runtime/impl/eventstream/eventstream_runtime.py
index e76f258bacc3..e90fb7680b2e 100644
--- a/openhands/runtime/impl/eventstream/eventstream_runtime.py
+++ b/openhands/runtime/impl/eventstream/eventstream_runtime.py
@@ -25,7 +25,7 @@
 )
 from openhands.events.action.action import Action
 from openhands.events.observation import (
-    FatalErrorObservation,
+    ErrorObservation,
     NullObservation,
     Observation,
     UserRejectObservation,
@@ -36,8 +36,9 @@
 from openhands.runtime.builder import DockerRuntimeBuilder
 from openhands.runtime.plugins import PluginRequirement
 from openhands.runtime.utils import find_available_tcp_port
-from openhands.runtime.utils.request import send_request_with_retry
+from openhands.runtime.utils.request import send_request
 from openhands.runtime.utils.runtime_build import build_runtime_image
+from openhands.utils.async_utils import call_sync_from_async
 from openhands.utils.tenacity_stop import stop_if_should_exit
 
 
@@ -123,7 +124,7 @@ def init_base_runtime(
         sid: str = 'default',
         plugins: list[PluginRequirement] | None = None,
         env_vars: dict[str, str] | None = None,
-        status_message_callback: Callable | None = None,
+        status_callback: Callable | None = None,
         attach_to_existing: bool = False,
     ):
         super().__init__(
@@ -132,7 +133,7 @@ def init_base_runtime(
             sid,
             plugins,
             env_vars,
-            status_message_callback,
+            status_callback,
             attach_to_existing,
         )
 
@@ -143,7 +144,7 @@ def __init__(
         sid: str = 'default',
         plugins: list[PluginRequirement] | None = None,
         env_vars: dict[str, str] | None = None,
-        status_message_callback: Callable | None = None,
+        status_callback: Callable | None = None,
         attach_to_existing: bool = False,
     ):
         self.config = config
@@ -151,7 +152,7 @@ def __init__(
         self._container_port = 30001  # initial dummy value
         self.api_url = f'{self.config.sandbox.local_runtime_url}:{self._container_port}'
         self.session = requests.Session()
-        self.status_message_callback = status_message_callback
+        self.status_callback = status_callback
 
         self.docker_client: docker.DockerClient = self._init_docker_client()
         self.base_container_image = self.config.sandbox.base_container_image
@@ -181,7 +182,7 @@ def __init__(
             sid,
             plugins,
             env_vars,
-            status_message_callback,
+            status_callback,
             attach_to_existing,
         )
 
@@ -205,21 +206,21 @@ async def connect(self):
             self.log(
                 'info', f'Starting runtime with image: {self.runtime_container_image}'
             )
-            self._init_container()
+            await call_sync_from_async(self._init_container)
             self.log('info', f'Container started: {self.container_name}')
 
         else:
-            self._attach_to_container()
+            await call_sync_from_async(self._attach_to_container)
 
         if not self.attach_to_existing:
             self.log('info', f'Waiting for client to become ready at {self.api_url}...')
         self.send_status_message('STATUS$WAITING_FOR_CLIENT')
-        self._wait_until_alive()
+        await call_sync_from_async(self._wait_until_alive)
         if not self.attach_to_existing:
             self.log('info', 'Runtime is ready.')
 
         if not self.attach_to_existing:
-            self.setup_initial_env()
+            await call_sync_from_async(self.setup_initial_env)
 
         self.log(
             'debug',
@@ -238,82 +239,74 @@ def _init_docker_client() -> docker.DockerClient:
             )
             raise ex
 
-    @tenacity.retry(
-        stop=tenacity.stop_after_attempt(5) | stop_if_should_exit(),
-        wait=tenacity.wait_exponential(multiplier=1, min=4, max=60),
-    )
     def _init_container(self):
-        try:
-            self.log('debug', 'Preparing to start container...')
-            self.send_status_message('STATUS$PREPARING_CONTAINER')
-            plugin_arg = ''
-            if self.plugins is not None and len(self.plugins) > 0:
-                plugin_arg = (
-                    f'--plugins {" ".join([plugin.name for plugin in self.plugins])} '
-                )
-
-            self._host_port = self._find_available_port()
-            self._container_port = (
-                self._host_port
-            )  # in future this might differ from host port
-            self.api_url = (
-                f'{self.config.sandbox.local_runtime_url}:{self._container_port}'
+        self.log('debug', 'Preparing to start container...')
+        self.send_status_message('STATUS$PREPARING_CONTAINER')
+        plugin_arg = ''
+        if self.plugins is not None and len(self.plugins) > 0:
+            plugin_arg = (
+                f'--plugins {" ".join([plugin.name for plugin in self.plugins])} '
             )
 
-            use_host_network = self.config.sandbox.use_host_network
-            network_mode: str | None = 'host' if use_host_network else None
-            port_mapping: dict[str, list[dict[str, str]]] | None = (
-                None
-                if use_host_network
-                else {
-                    f'{self._container_port}/tcp': [{'HostPort': str(self._host_port)}]
-                }
-            )
+        self._host_port = self._find_available_port()
+        self._container_port = (
+            self._host_port
+        )  # in future this might differ from host port
+        self.api_url = f'{self.config.sandbox.local_runtime_url}:{self._container_port}'
 
-            if use_host_network:
-                self.log(
-                    'warn',
-                    'Using host network mode. If you are using MacOS, please make sure you have the latest version of Docker Desktop and enabled host network feature: https://docs.docker.com/network/drivers/host/#docker-desktop',
-                )
+        use_host_network = self.config.sandbox.use_host_network
+        network_mode: str | None = 'host' if use_host_network else None
+        port_mapping: dict[str, list[dict[str, str]]] | None = (
+            None
+            if use_host_network
+            else {f'{self._container_port}/tcp': [{'HostPort': str(self._host_port)}]}
+        )
 
-            # Combine environment variables
-            environment = {
-                'port': str(self._container_port),
-                'PYTHONUNBUFFERED': 1,
-            }
-            if self.config.debug or DEBUG:
-                environment['DEBUG'] = 'true'
+        if use_host_network:
+            self.log(
+                'warn',
+                'Using host network mode. If you are using MacOS, please make sure you have the latest version of Docker Desktop and enabled host network feature: https://docs.docker.com/network/drivers/host/#docker-desktop',
+            )
 
-            self.log('debug', f'Workspace Base: {self.config.workspace_base}')
-            if (
-                self.config.workspace_mount_path is not None
-                and self.config.workspace_mount_path_in_sandbox is not None
-            ):
-                # e.g. result would be: {"/home/user/openhands/workspace": {'bind': "/workspace", 'mode': 'rw'}}
-                volumes = {
-                    self.config.workspace_mount_path: {
-                        'bind': self.config.workspace_mount_path_in_sandbox,
-                        'mode': 'rw',
-                    }
+        # Combine environment variables
+        environment = {
+            'port': str(self._container_port),
+            'PYTHONUNBUFFERED': 1,
+        }
+        if self.config.debug or DEBUG:
+            environment['DEBUG'] = 'true'
+
+        self.log('debug', f'Workspace Base: {self.config.workspace_base}')
+        if (
+            self.config.workspace_mount_path is not None
+            and self.config.workspace_mount_path_in_sandbox is not None
+        ):
+            # e.g. result would be: {"/home/user/openhands/workspace": {'bind': "/workspace", 'mode': 'rw'}}
+            volumes = {
+                self.config.workspace_mount_path: {
+                    'bind': self.config.workspace_mount_path_in_sandbox,
+                    'mode': 'rw',
                 }
-                logger.debug(f'Mount dir: {self.config.workspace_mount_path}')
-            else:
-                logger.debug(
-                    'Mount dir is not set, will not mount the workspace directory to the container'
-                )
-                volumes = None
-            self.log(
-                'debug',
-                f'Sandbox workspace: {self.config.workspace_mount_path_in_sandbox}'
+            }
+            logger.debug(f'Mount dir: {self.config.workspace_mount_path}')
+        else:
+            logger.debug(
+                'Mount dir is not set, will not mount the workspace directory to the container'
             )
+            volumes = None
+        self.log(
+            'debug',
+            f'Sandbox workspace: {self.config.workspace_mount_path_in_sandbox}',
+        )
 
-            if self.config.sandbox.browsergym_eval_env is not None:
-                browsergym_arg = (
-                    f'--browsergym-eval-env {self.config.sandbox.browsergym_eval_env}'
-                )
-            else:
-                browsergym_arg = ''
+        if self.config.sandbox.browsergym_eval_env is not None:
+            browsergym_arg = (
+                f'--browsergym-eval-env {self.config.sandbox.browsergym_eval_env}'
+            )
+        else:
+            browsergym_arg = ''
 
+        try:
             self.container = self.docker_client.containers.run(
                 self.runtime_container_image,
                 command=(
@@ -337,6 +330,21 @@ def _init_container(self):
             self.log_buffer = LogBuffer(self.container, self.log)
             self.log('debug', f'Container started. Server url: {self.api_url}')
             self.send_status_message('STATUS$CONTAINER_STARTED')
+        except docker.errors.APIError as e:
+            # check 409 error
+            if '409' in str(e):
+                self.log(
+                    'warning',
+                    f'Container {self.container_name} already exists. Removing...',
+                )
+                self._close_containers(rm_all_containers=True)
+                return self._init_container()
+
+            else:
+                self.log(
+                    'error',
+                    f'Error: Instance {self.container_name} FAILED to start container!\n',
+                )
         except Exception as e:
             self.log(
                 'error',
@@ -384,27 +392,20 @@ def _refresh_logs(self):
 
     @tenacity.retry(
         stop=tenacity.stop_after_delay(120) | stop_if_should_exit(),
-        wait=tenacity.wait_exponential(multiplier=2, min=1, max=20),
         reraise=(ConnectionRefusedError,),
+        wait=tenacity.wait_fixed(2),
     )
     def _wait_until_alive(self):
         self._refresh_logs()
         if not self.log_buffer:
             raise RuntimeError('Runtime client is not ready.')
 
-        response = send_request_with_retry(
+        send_request(
             self.session,
             'GET',
             f'{self.api_url}/alive',
-            retry_exceptions=[ConnectionRefusedError],
-            timeout=300,  # 5 minutes gives the container time to be alive 🧟‍♂️
+            timeout=5,
         )
-        if response.status_code == 200:
-            return
-        else:
-            msg = f'Action execution API is not alive. Response: {response}'
-            self.log('error', msg)
-            raise RuntimeError(msg)
 
     def close(self, rm_all_containers: bool = True):
         """Closes the EventStreamRuntime and associated objects
@@ -421,7 +422,9 @@ def close(self, rm_all_containers: bool = True):
 
         if self.attach_to_existing:
             return
+        self._close_containers(rm_all_containers)
 
+    def _close_containers(self, rm_all_containers: bool = True):
         try:
             containers = self.docker_client.containers.list(all=True)
             for container in containers:
@@ -466,10 +469,11 @@ def run_action(self, action: Action) -> Observation:
                 return NullObservation('')
             action_type = action.action  # type: ignore[attr-defined]
             if action_type not in ACTION_TYPE_TO_CLASS:
-                return FatalErrorObservation(f'Action {action_type} does not exist.')
+                raise ValueError(f'Action {action_type} does not exist.')
             if not hasattr(self, action_type):
-                return FatalErrorObservation(
-                    f'Action {action_type} is not supported in the current runtime.'
+                return ErrorObservation(
+                    f'Action {action_type} is not supported in the current runtime.',
+                    error_id='AGENT_ERROR$BAD_ACTION',
                 )
             if (
                 getattr(action, 'confirmation_state', None)
@@ -484,33 +488,21 @@ def run_action(self, action: Action) -> Observation:
             assert action.timeout is not None
 
             try:
-                response = send_request_with_retry(
+                response = send_request(
                     self.session,
                     'POST',
                     f'{self.api_url}/execute_action',
                     json={'action': event_to_dict(action)},
-                    timeout=action.timeout,
+                    # wait a few more seconds to get the timeout error from client side
+                    timeout=action.timeout + 5,
                 )
-                if response.status_code == 200:
-                    output = response.json()
-                    obs = observation_from_dict(output)
-                    obs._cause = action.id  # type: ignore[attr-defined]
-                else:
-                    self.log('debug', f'action: {action}')
-                    self.log('debug', f'response: {response}')
-                    error_message = response.text
-                    self.log('error', f'Error from server: {error_message}')
-                    obs = FatalErrorObservation(
-                        f'Action execution failed: {error_message}'
-                    )
+                output = response.json()
+                obs = observation_from_dict(output)
+                obs._cause = action.id  # type: ignore[attr-defined]
             except requests.Timeout:
-                self.log('error', 'No response received within the timeout period.')
-                obs = FatalErrorObservation(
-                    f'Action execution timed out after {action.timeout} seconds.'
+                raise RuntimeError(
+                    f'Runtime failed to return execute_action before the requested timeout of {action.timeout}s'
                 )
-            except Exception as e:
-                self.log('error', f'Error during action execution: {e}')
-                obs = FatalErrorObservation(f'Action execution failed: {str(e)}')
             self._refresh_logs()
             return obs
 
@@ -567,7 +559,7 @@ def copy_to(
 
             params = {'destination': sandbox_dest, 'recursive': str(recursive).lower()}
 
-            response = send_request_with_retry(
+            send_request(
                 self.session,
                 'POST',
                 f'{self.api_url}/upload_file',
@@ -575,11 +567,6 @@ def copy_to(
                 params=params,
                 timeout=300,
             )
-            if response.status_code == 200:
-                return
-            else:
-                error_message = response.text
-                raise Exception(f'Copy operation failed: {error_message}')
 
         except requests.Timeout:
             raise TimeoutError('Copy operation timed out')
@@ -604,31 +591,25 @@ def list_files(self, path: str | None = None) -> list[str]:
             if path is not None:
                 data['path'] = path
 
-            response = send_request_with_retry(
+            response = send_request(
                 self.session,
                 'POST',
                 f'{self.api_url}/list_files',
                 json=data,
-                timeout=30,  # 30 seconds because the container should already be alive
+                timeout=10,
             )
-            if response.status_code == 200:
-                response_json = response.json()
-                assert isinstance(response_json, list)
-                return response_json
-            else:
-                error_message = response.text
-                raise Exception(f'List files operation failed: {error_message}')
+            response_json = response.json()
+            assert isinstance(response_json, list)
+            return response_json
         except requests.Timeout:
             raise TimeoutError('List files operation timed out')
-        except Exception as e:
-            raise RuntimeError(f'List files operation failed: {str(e)}')
 
     def copy_from(self, path: str) -> bytes:
         """Zip all files in the sandbox and return as a stream of bytes."""
         self._refresh_logs()
         try:
             params = {'path': path}
-            response = send_request_with_retry(
+            response = send_request(
                 self.session,
                 'GET',
                 f'{self.api_url}/download_files',
@@ -636,16 +617,10 @@ def copy_from(self, path: str) -> bytes:
                 stream=True,
                 timeout=30,
             )
-            if response.status_code == 200:
-                data = response.content
-                return data
-            else:
-                error_message = response.text
-                raise Exception(f'Copy operation failed: {error_message}')
+            data = response.content
+            return data
         except requests.Timeout:
             raise TimeoutError('Copy operation timed out')
-        except Exception as e:
-            raise RuntimeError(f'Copy operation failed: {str(e)}')
 
     def _is_port_in_use_docker(self, port):
         containers = self.docker_client.containers.list()
@@ -663,8 +638,3 @@ def _find_available_port(self, max_attempts=5):
                 return port
         # If no port is found after max_attempts, return the last tried port
         return port
-
-    def send_status_message(self, message: str):
-        """Sends a status message if the callback function was provided."""
-        if self.status_message_callback:
-            self.status_message_callback(message)
diff --git a/openhands/runtime/impl/modal/modal_runtime.py b/openhands/runtime/impl/modal/modal_runtime.py
index 3a484c43e693..0e598a437f41 100644
--- a/openhands/runtime/impl/modal/modal_runtime.py
+++ b/openhands/runtime/impl/modal/modal_runtime.py
@@ -75,7 +75,7 @@ def __init__(
         sid: str = 'default',
         plugins: list[PluginRequirement] | None = None,
         env_vars: dict[str, str] | None = None,
-        status_message_callback: Callable | None = None,
+        status_callback: Callable | None = None,
         attach_to_existing: bool = False,
     ):
         assert config.modal_api_token_id, 'Modal API token id is required'
@@ -102,7 +102,7 @@ def __init__(
         self.container_port = 3000
 
         self.session = requests.Session()
-        self.status_message_callback = status_message_callback
+        self.status_callback = status_callback
         self.base_container_image_id = self.config.sandbox.base_container_image
         self.runtime_container_image_id = self.config.sandbox.runtime_container_image
         self.action_semaphore = threading.Semaphore(1)  # Ensure one action at a time
@@ -122,7 +122,7 @@ def __init__(
             sid,
             plugins,
             env_vars,
-            status_message_callback,
+            status_callback,
             attach_to_existing,
         )
 
diff --git a/openhands/runtime/impl/remote/remote_runtime.py b/openhands/runtime/impl/remote/remote_runtime.py
index 7e533269d1ab..1e6fdecf51a9 100644
--- a/openhands/runtime/impl/remote/remote_runtime.py
+++ b/openhands/runtime/impl/remote/remote_runtime.py
@@ -1,12 +1,11 @@
 import os
 import tempfile
 import threading
-import time
 from typing import Callable, Optional
 from zipfile import ZipFile
 
 import requests
-from requests.exceptions import Timeout
+import tenacity
 
 from openhands.core.config import AppConfig
 from openhands.events import EventStream
@@ -21,22 +20,26 @@
 )
 from openhands.events.action.action import Action
 from openhands.events.observation import (
-    FatalErrorObservation,
+    ErrorObservation,
     NullObservation,
     Observation,
 )
 from openhands.events.serialization import event_to_dict, observation_from_dict
 from openhands.events.serialization.action import ACTION_TYPE_TO_CLASS
-from openhands.runtime.base import Runtime
+from openhands.runtime.base import (
+    Runtime,
+    RuntimeDisconnectedError,
+    RuntimeNotReadyError,
+)
 from openhands.runtime.builder.remote import RemoteRuntimeBuilder
 from openhands.runtime.plugins import PluginRequirement
 from openhands.runtime.utils.command import get_remote_startup_command
 from openhands.runtime.utils.request import (
-    is_404_error,
-    is_503_error,
-    send_request_with_retry,
+    send_request,
 )
 from openhands.runtime.utils.runtime_build import build_runtime_image
+from openhands.utils.async_utils import call_sync_from_async
+from openhands.utils.tenacity_stop import stop_if_should_exit
 
 
 class RemoteRuntime(Runtime):
@@ -51,31 +54,32 @@ def __init__(
         sid: str = 'default',
         plugins: list[PluginRequirement] | None = None,
         env_vars: dict[str, str] | None = None,
-        status_message_callback: Optional[Callable] = None,
+        status_callback: Optional[Callable] = None,
         attach_to_existing: bool = False,
     ):
+        # We need to set session and action_semaphore before the __init__ below, or we get odd errors
+        self.session = requests.Session()
+        self.action_semaphore = threading.Semaphore(1)
+
         super().__init__(
             config,
             event_stream,
             sid,
             plugins,
             env_vars,
-            status_message_callback,
+            status_callback,
             attach_to_existing,
         )
-
         if self.config.sandbox.api_key is None:
             raise ValueError(
                 'API key is required to use the remote runtime. '
                 'Please set the API key in the config (config.toml) or as an environment variable (SANDBOX_API_KEY).'
             )
-        self.session = requests.Session()
         self.session.headers.update({'X-API-Key': self.config.sandbox.api_key})
-        self.action_semaphore = threading.Semaphore(1)
 
         if self.config.workspace_base is not None:
             self.log(
-                'warning',
+                'debug',
                 'Setting workspace_base is not supported in the remote runtime.',
             )
 
@@ -86,9 +90,13 @@ def __init__(
         self.runtime_url: str | None = None
 
     async def connect(self):
-        self._start_or_attach_to_runtime()
-        self._wait_until_alive()
-        self.setup_initial_env()
+        await call_sync_from_async(self._start_or_attach_to_runtime)
+        try:
+            await call_sync_from_async(self._wait_until_alive)
+        except RuntimeNotReadyError:
+            self.log('error', 'Runtime failed to start, timed out before ready')
+            raise
+        await call_sync_from_async(self.setup_initial_env)
 
     def _start_or_attach_to_runtime(self):
         existing_runtime = self._check_existing_runtime()
@@ -127,44 +135,40 @@ def _start_or_attach_to_runtime(self):
 
     def _check_existing_runtime(self) -> bool:
         try:
-            response = send_request_with_retry(
-                self.session,
+            response = self._send_request(
                 'GET',
                 f'{self.config.sandbox.remote_runtime_api_url}/runtime/{self.sid}',
                 timeout=5,
             )
-        except Exception as e:
+        except requests.HTTPError as e:
+            if e.response.status_code == 404:
+                return False
             self.log('debug', f'Error while looking for remote runtime: {e}')
+            raise
+
+        data = response.json()
+        status = data.get('status')
+        if status == 'running':
+            self._parse_runtime_response(response)
+            return True
+        elif status == 'stopped':
+            self.log('debug', 'Found existing remote runtime, but it is stopped')
             return False
-
-        if response.status_code == 200:
-            data = response.json()
-            status = data.get('status')
-            if status == 'running':
-                self._parse_runtime_response(response)
-                return True
-            elif status == 'stopped':
-                self.log('debug', 'Found existing remote runtime, but it is stopped')
-                return False
-            elif status == 'paused':
-                self.log('debug', 'Found existing remote runtime, but it is paused')
-                self._parse_runtime_response(response)
-                self._resume_runtime()
-                return True
-            else:
-                self.log('error', f'Invalid response from runtime API: {data}')
-                return False
+        elif status == 'paused':
+            self.log('debug', 'Found existing remote runtime, but it is paused')
+            self._parse_runtime_response(response)
+            self._resume_runtime()
+            return True
         else:
-            self.log('debug', 'Could not find existing remote runtime')
+            self.log('error', f'Invalid response from runtime API: {data}')
             return False
 
     def _build_runtime(self):
         self.log('debug', f'Building RemoteRuntime config:\n{self.config}')
-        response = send_request_with_retry(
-            self.session,
+        response = self._send_request(
             'GET',
             f'{self.config.sandbox.remote_runtime_api_url}/registry_prefix',
-            timeout=30,
+            timeout=10,
         )
         response_json = response.json()
         registry_prefix = response_json['registry_prefix']
@@ -191,14 +195,13 @@ def _build_runtime(self):
             force_rebuild=self.config.sandbox.force_rebuild_runtime,
         )
 
-        response = send_request_with_retry(
-            self.session,
+        response = self._send_request(
             'GET',
             f'{self.config.sandbox.remote_runtime_api_url}/image_exists',
             params={'image': self.container_image},
-            timeout=30,
+            timeout=10,
         )
-        if response.status_code != 200 or not response.json()['exists']:
+        if not response.json()['exists']:
             raise RuntimeError(f'Container image {self.container_image} does not exist')
 
     def _start_runtime(self):
@@ -228,17 +231,11 @@ def _start_runtime(self):
         }
 
         # Start the sandbox using the /start endpoint
-        response = send_request_with_retry(
-            self.session,
+        response = self._send_request(
             'POST',
             f'{self.config.sandbox.remote_runtime_api_url}/start',
             json=start_request,
-            timeout=300,
         )
-        if response.status_code != 201:
-            raise RuntimeError(
-                f'[Runtime (ID={self.runtime_id})] Failed to start runtime: {response.text}'
-            )
         self._parse_runtime_response(response)
         self.log(
             'debug',
@@ -246,17 +243,12 @@ def _start_runtime(self):
         )
 
     def _resume_runtime(self):
-        response = send_request_with_retry(
-            self.session,
+        self._send_request(
             'POST',
             f'{self.config.sandbox.remote_runtime_api_url}/resume',
             json={'runtime_id': self.runtime_id},
             timeout=30,
         )
-        if response.status_code != 200:
-            raise RuntimeError(
-                f'[Runtime (ID={self.runtime_id})] Failed to resume runtime: {response.text}'
-            )
         self.log('debug', 'Runtime resumed.')
 
     def _parse_runtime_response(self, response: requests.Response):
@@ -268,72 +260,57 @@ def _parse_runtime_response(self, response: requests.Response):
                 {'X-Session-API-Key': start_response['session_api_key']}
             )
 
+    @tenacity.retry(
+        stop=tenacity.stop_after_delay(180) | stop_if_should_exit(),
+        reraise=True,
+        retry=tenacity.retry_if_exception_type(RuntimeNotReadyError),
+        wait=tenacity.wait_fixed(2),
+    )
     def _wait_until_alive(self):
         self.log('debug', f'Waiting for runtime to be alive at url: {self.runtime_url}')
-        # send GET request to /runtime/<id>
-        pod_running = False
-        max_not_found_count = 12  # 2 minutes
-        not_found_count = 0
-        while not pod_running:
-            runtime_info_response = send_request_with_retry(
-                self.session,
-                'GET',
-                f'{self.config.sandbox.remote_runtime_api_url}/runtime/{self.runtime_id}',
-                timeout=5,
-            )
-            if runtime_info_response.status_code != 200:
-                raise RuntimeError(
-                    f'Failed to get runtime status: {runtime_info_response.status_code}. Response: {runtime_info_response.text}'
-                )
-            runtime_data = runtime_info_response.json()
-            assert runtime_data['runtime_id'] == self.runtime_id
-            pod_status = runtime_data['pod_status']
-            self.log(
-                'debug',
-                f'Waiting for runtime pod to be active. Current status: {pod_status}',
-            )
-            if pod_status == 'Ready':
-                pod_running = True
-                break
-            elif pod_status == 'Not Found' and not_found_count < max_not_found_count:
-                not_found_count += 1
+        runtime_info_response = self._send_request(
+            'GET',
+            f'{self.config.sandbox.remote_runtime_api_url}/runtime/{self.runtime_id}',
+        )
+        runtime_data = runtime_info_response.json()
+        assert 'runtime_id' in runtime_data
+        assert runtime_data['runtime_id'] == self.runtime_id
+        assert 'pod_status' in runtime_data
+        pod_status = runtime_data['pod_status']
+        if pod_status == 'Ready':
+            try:
+                self._send_request(
+                    'GET',
+                    f'{self.runtime_url}/alive',
+                )  # will raise exception if we don't get 200 back.
+            except requests.HTTPError as e:
                 self.log(
-                    'debug',
-                    f'Runtime pod not found. Count: {not_found_count} / {max_not_found_count}',
+                    'warning', f"Runtime /alive failed, but pod says it's ready: {e}"
                 )
-            elif pod_status in ('Failed', 'Unknown', 'Not Found'):
-                # clean up the runtime
-                self.close()
-                raise RuntimeError(
-                    f'Runtime (ID={self.runtime_id}) failed to start. Current status: {pod_status}'
+                raise RuntimeNotReadyError(
+                    f'Runtime /alive failed to respond with 200: {e}'
                 )
-            # Pending otherwise - add proper sleep
-            time.sleep(10)
+            return
+        if pod_status in ('Failed', 'Unknown', 'Not Found'):
+            # clean up the runtime
+            self.close()
+            raise RuntimeError(
+                f'Runtime (ID={self.runtime_id}) failed to start. Current status: {pod_status}'
+            )
 
-        response = send_request_with_retry(
-            self.session,
-            'GET',
-            f'{self.runtime_url}/alive',
-            # Retry 404 & 503 errors for the /alive endpoint
-            # because the runtime might just be starting up
-            # and have not registered the endpoint yet
-            retry_fns=[is_404_error, is_503_error],
-            # leave enough time for the runtime to start up
-            timeout=600,
+        self.log(
+            'debug',
+            f'Waiting for runtime pod to be active. Current status: {pod_status}',
         )
-        if response.status_code != 200:
-            msg = f'Runtime (ID={self.runtime_id}) is not alive yet. Status: {response.status_code}.'
-            self.log('warning', msg)
-            raise RuntimeError(msg)
+        raise RuntimeNotReadyError()
 
     def close(self, timeout: int = 10):
         if self.config.sandbox.keep_remote_runtime_alive or self.attach_to_existing:
             self.session.close()
             return
-        if self.runtime_id:
+        if self.runtime_id and self.session:
             try:
-                response = send_request_with_retry(
-                    self.session,
+                response = self._send_request(
                     'POST',
                     f'{self.config.sandbox.remote_runtime_api_url}/stop',
                     json={'runtime_id': self.runtime_id},
@@ -361,12 +338,11 @@ def run_action(self, action: Action) -> Observation:
                 return NullObservation('')
             action_type = action.action  # type: ignore[attr-defined]
             if action_type not in ACTION_TYPE_TO_CLASS:
-                return FatalErrorObservation(
-                    f'[Runtime (ID={self.runtime_id})] Action {action_type} does not exist.'
-                )
+                raise ValueError(f'Action {action_type} does not exist.')
             if not hasattr(self, action_type):
-                return FatalErrorObservation(
-                    f'[Runtime (ID={self.runtime_id})] Action {action_type} is not supported in the current runtime.'
+                return ErrorObservation(
+                    f'[Runtime (ID={self.runtime_id})] Action {action_type} is not supported in the current runtime.',
+                    error_id='AGENT_ERROR$BAD_ACTION',
                 )
 
             assert action.timeout is not None
@@ -374,36 +350,37 @@ def run_action(self, action: Action) -> Observation:
             try:
                 request_body = {'action': event_to_dict(action)}
                 self.log('debug', f'Request body: {request_body}')
-                response = send_request_with_retry(
-                    self.session,
+                response = self._send_request(
                     'POST',
                     f'{self.runtime_url}/execute_action',
                     json=request_body,
-                    timeout=action.timeout,
-                )
-                if response.status_code == 200:
-                    output = response.json()
-                    obs = observation_from_dict(output)
-                    obs._cause = action.id  # type: ignore[attr-defined]
-                    return obs
-                else:
-                    error_message = response.text
-                    self.log('error', f'Error from server: {error_message}')
-                    obs = FatalErrorObservation(
-                        f'Action execution failed: {error_message}'
-                    )
-            except Timeout:
-                self.log('error', 'No response received within the timeout period.')
-                obs = FatalErrorObservation(
-                    f'[Runtime (ID={self.runtime_id})] Action execution timed out'
+                    # wait a few more seconds to get the timeout error from client side
+                    timeout=action.timeout + 5,
                 )
-            except Exception as e:
-                self.log('error', f'Error during action execution: {e}')
-                obs = FatalErrorObservation(
-                    f'[Runtime (ID={self.runtime_id})] Action execution failed: {str(e)}'
+                output = response.json()
+                obs = observation_from_dict(output)
+                obs._cause = action.id  # type: ignore[attr-defined]
+            except requests.Timeout:
+                raise RuntimeError(
+                    f'Runtime failed to return execute_action before the requested timeout of {action.timeout}s'
                 )
             return obs
 
+    def _send_request(self, method, url, **kwargs):
+        is_runtime_request = self.runtime_url and self.runtime_url in url
+        try:
+            return send_request(self.session, method, url, **kwargs)
+        except requests.Timeout:
+            self.log('error', 'No response received within the timeout period.')
+            raise
+        except requests.HTTPError as e:
+            if is_runtime_request and e.response.status_code == 404:
+                raise RuntimeDisconnectedError(
+                    f'404 error while connecting to {self.runtime_url}'
+                )
+            else:
+                raise e
+
     def run(self, action: CmdRunAction) -> Observation:
         return self.run_action(action)
 
@@ -450,32 +427,16 @@ def copy_to(
 
             params = {'destination': sandbox_dest, 'recursive': str(recursive).lower()}
 
-            response = send_request_with_retry(
-                self.session,
+            response = self._send_request(
                 'POST',
                 f'{self.runtime_url}/upload_file',
                 files=upload_data,
                 params=params,
                 timeout=300,
             )
-            if response.status_code == 200:
-                self.log(
-                    'debug',
-                    f'Copy completed: host:{host_src} -> runtime:{sandbox_dest}. Response: {response.text}',
-                )
-                return
-            else:
-                error_message = response.text
-                raise Exception(
-                    f'[Runtime (ID={self.runtime_id})] Copy operation failed: {error_message}'
-                )
-        except TimeoutError:
-            raise TimeoutError(
-                f'[Runtime (ID={self.runtime_id})] Copy operation timed out'
-            )
-        except Exception as e:
-            raise RuntimeError(
-                f'[Runtime (ID={self.runtime_id})] Copy operation failed: {str(e)}'
+            self.log(
+                'debug',
+                f'Copy completed: host:{host_src} -> runtime:{sandbox_dest}. Response: {response.text}',
             )
         finally:
             if recursive:
@@ -485,64 +446,27 @@ def copy_to(
             )
 
     def list_files(self, path: str | None = None) -> list[str]:
-        try:
-            data = {}
-            if path is not None:
-                data['path'] = path
+        data = {}
+        if path is not None:
+            data['path'] = path
 
-            response = send_request_with_retry(
-                self.session,
-                'POST',
-                f'{self.runtime_url}/list_files',
-                json=data,
-                timeout=30,
-            )
-            if response.status_code == 200:
-                response_json = response.json()
-                assert isinstance(response_json, list)
-                return response_json
-            else:
-                error_message = response.text
-                raise Exception(
-                    f'[Runtime (ID={self.runtime_id})] List files operation failed: {error_message}'
-                )
-        except TimeoutError:
-            raise TimeoutError(
-                f'[Runtime (ID={self.runtime_id})] List files operation timed out'
-            )
-        except Exception as e:
-            raise RuntimeError(
-                f'[Runtime (ID={self.runtime_id})] List files operation failed: {str(e)}'
-            )
+        response = self._send_request(
+            'POST',
+            f'{self.runtime_url}/list_files',
+            json=data,
+            timeout=30,
+        )
+        response_json = response.json()
+        assert isinstance(response_json, list)
+        return response_json
 
     def copy_from(self, path: str) -> bytes:
         """Zip all files in the sandbox and return as a stream of bytes."""
-        try:
-            params = {'path': path}
-            response = send_request_with_retry(
-                self.session,
-                'GET',
-                f'{self.runtime_url}/download_files',
-                params=params,
-                timeout=30,
-            )
-            if response.status_code == 200:
-                return response.content
-            else:
-                error_message = response.text
-                raise Exception(
-                    f'[Runtime (ID={self.runtime_id})] Copy operation failed: {error_message}'
-                )
-        except requests.Timeout:
-            raise TimeoutError(
-                f'[Runtime (ID={self.runtime_id})] Copy operation timed out'
-            )
-        except Exception as e:
-            raise RuntimeError(
-                f'[Runtime (ID={self.runtime_id})] Copy operation failed: {str(e)}'
-            )
-
-    def send_status_message(self, message: str):
-        """Sends a status message if the callback function was provided."""
-        if self.status_message_callback:
-            self.status_message_callback(message)
+        params = {'path': path}
+        response = self._send_request(
+            'GET',
+            f'{self.runtime_url}/download_files',
+            params=params,
+            timeout=30,
+        )
+        return response.content
diff --git a/openhands/runtime/plugins/agent_skills/file_editor/impl.py b/openhands/runtime/plugins/agent_skills/file_editor/impl.py
index e0944ab6593e..613c550e7a80 100644
--- a/openhands/runtime/plugins/agent_skills/file_editor/impl.py
+++ b/openhands/runtime/plugins/agent_skills/file_editor/impl.py
@@ -46,13 +46,13 @@ def __call__(
         if command == 'view':
             return self.view(_path, view_range)
         elif command == 'create':
-            if not file_text:
+            if file_text is None:
                 raise ToolError('Parameter `file_text` is required for command: create')
             self.write_file(_path, file_text)
             self._file_history[_path].append(file_text)
             return ToolResult(output=f'File created successfully at: {_path}')
         elif command == 'str_replace':
-            if not old_str:
+            if old_str is None:
                 raise ToolError(
                     'Parameter `old_str` is required for command: str_replace'
                 )
@@ -62,7 +62,7 @@ def __call__(
                 raise ToolError(
                     'Parameter `insert_line` is required for command: insert'
                 )
-            if not new_str:
+            if new_str is None:
                 raise ToolError('Parameter `new_str` is required for command: insert')
             return self.insert(_path, insert_line, new_str)
         elif command == 'undo_edit':
diff --git a/openhands/runtime/utils/bash.py b/openhands/runtime/utils/bash.py
index fba16787c6dc..a5019315a038 100644
--- a/openhands/runtime/utils/bash.py
+++ b/openhands/runtime/utils/bash.py
@@ -9,7 +9,7 @@
 from openhands.events.event import EventSource
 from openhands.events.observation import (
     CmdOutputObservation,
-    FatalErrorObservation,
+    ErrorObservation,
 )
 
 SOFT_TIMEOUT_SECONDS = 5
@@ -275,7 +275,7 @@ def _continue_bash(
                 output += '\r\n' + bash_prompt
         return output, exit_code
 
-    def run(self, action: CmdRunAction) -> CmdOutputObservation | FatalErrorObservation:
+    def run(self, action: CmdRunAction) -> CmdOutputObservation | ErrorObservation:
         try:
             assert (
                 action.timeout is not None
@@ -329,6 +329,6 @@ def run(self, action: CmdRunAction) -> CmdOutputObservation | FatalErrorObservat
                 interpreter_details=python_interpreter,
             )
         except UnicodeDecodeError as e:
-            return FatalErrorObservation(
-                f'Runtime bash execution failed: Command output could not be decoded as utf-8. {str(e)}'
+            return ErrorObservation(
+                f'Runtime bash execution failed: Command output could not be decoded as utf-8. {str(e)}',
             )
diff --git a/openhands/runtime/utils/edit.py b/openhands/runtime/utils/edit.py
index 9a2e1775e10c..ff5e343fae2d 100644
--- a/openhands/runtime/utils/edit.py
+++ b/openhands/runtime/utils/edit.py
@@ -13,7 +13,6 @@
 )
 from openhands.events.observation import (
     ErrorObservation,
-    FatalErrorObservation,
     FileEditObservation,
     FileReadObservation,
     FileWriteObservation,
@@ -214,8 +213,8 @@ def edit(self, action: FileEditAction) -> Observation:
             if isinstance(obs, ErrorObservation):
                 return obs
             if not isinstance(obs, FileWriteObservation):
-                return FatalErrorObservation(
-                    f'Fatal Runtime in editing: Expected FileWriteObservation, got {type(obs)}: {str(obs)}'
+                raise ValueError(
+                    f'Expected FileWriteObservation, got {type(obs)}: {str(obs)}'
                 )
             return FileEditObservation(
                 content=get_diff('', action.content, action.path),
@@ -225,8 +224,8 @@ def edit(self, action: FileEditAction) -> Observation:
                 new_content=action.content,
             )
         if not isinstance(obs, FileReadObservation):
-            return FatalErrorObservation(
-                f'Fatal Runtime in editing: Expected FileReadObservation, got {type(obs)}: {str(obs)}'
+            raise ValueError(
+                f'Expected FileReadObservation, got {type(obs)}: {str(obs)}'
             )
 
         original_file_content = obs.content
diff --git a/openhands/runtime/utils/request.py b/openhands/runtime/utils/request.py
index 6940827a3adf..655fe304e5e4 100644
--- a/openhands/runtime/utils/request.py
+++ b/openhands/runtime/utils/request.py
@@ -1,22 +1,12 @@
-from typing import Any, Callable, Type
+from typing import Any
 
 import requests
 from requests.exceptions import (
     ChunkedEncodingError,
     ConnectionError,
 )
-from tenacity import (
-    retry,
-    retry_if_exception,
-    retry_if_exception_type,
-    stop_after_delay,
-    wait_exponential,
-)
 from urllib3.exceptions import IncompleteRead
 
-from openhands.core.logger import openhands_logger as logger
-from openhands.utils.tenacity_stop import stop_if_should_exit
-
 
 def is_server_error(exception):
     return (
@@ -60,37 +50,13 @@ def is_502_error(exception):
 ]
 
 
-def send_request_with_retry(
+def send_request(
     session: requests.Session,
     method: str,
     url: str,
-    timeout: int,
-    retry_exceptions: list[Type[Exception]] | None = None,
-    retry_fns: list[Callable[[Exception], bool]] | None = None,
+    timeout: int = 10,
     **kwargs: Any,
 ) -> requests.Response:
-    exceptions_to_catch = retry_exceptions or DEFAULT_RETRY_EXCEPTIONS
-    retry_condition = retry_if_exception_type(
-        tuple(exceptions_to_catch)
-    ) | retry_if_exception(is_502_error)
-    if retry_fns is not None:
-        for fn in retry_fns:
-            retry_condition |= retry_if_exception(fn)
-    # wait a few more seconds to get the timeout error from client side
-    kwargs['timeout'] = timeout + 10
-
-    @retry(
-        stop=stop_after_delay(timeout) | stop_if_should_exit(),
-        wait=wait_exponential(multiplier=1, min=4, max=20),
-        retry=retry_condition,
-        reraise=True,
-        before_sleep=lambda retry_state: logger.debug(
-            f'Retrying {method} request to {url} due to {retry_state.outcome.exception()}. Attempt {retry_state.attempt_number}'
-        ),
-    )
-    def _send_request_with_retry():
-        response = session.request(method, url, **kwargs)
-        response.raise_for_status()
-        return response
-
-    return _send_request_with_retry()
+    response = session.request(method, url, **kwargs)
+    response.raise_for_status()
+    return response
diff --git a/openhands/runtime/utils/runtime_build.py b/openhands/runtime/utils/runtime_build.py
index 8830805e227d..eab98befe538 100644
--- a/openhands/runtime/utils/runtime_build.py
+++ b/openhands/runtime/utils/runtime_build.py
@@ -175,7 +175,9 @@ def build_runtime_image_in_folder(
 
     logger.info(f'Building image: {hash_image_name}')
     if force_rebuild:
-        logger.debug(f'Force rebuild: [{runtime_image_repo}:{source_tag}] from scratch.')
+        logger.debug(
+            f'Force rebuild: [{runtime_image_repo}:{source_tag}] from scratch.'
+        )
         prep_build_folder(
             build_folder,
             base_image,
diff --git a/openhands/runtime/utils/runtime_templates/Dockerfile.j2 b/openhands/runtime/utils/runtime_templates/Dockerfile.j2
index 3de7f42e5ff4..f9fb596d3414 100644
--- a/openhands/runtime/utils/runtime_templates/Dockerfile.j2
+++ b/openhands/runtime/utils/runtime_templates/Dockerfile.j2
@@ -87,6 +87,7 @@ COPY ./code/pyproject.toml ./code/poetry.lock /openhands/code/
 RUN if [ -d /openhands/code/openhands ]; then rm -rf /openhands/code/openhands; fi
 COPY ./code/pyproject.toml ./code/poetry.lock /openhands/code/
 COPY ./code/openhands /openhands/code/openhands
+RUN chmod a+rwx /openhands/code/openhands/__init__.py
 
 # ================================================================
 # END: Build from versioned image
diff --git a/openhands/runtime/utils/tenacity_stop.py b/openhands/runtime/utils/tenacity_stop.py
index e4b634547704..48fdead86647 100644
--- a/openhands/runtime/utils/tenacity_stop.py
+++ b/openhands/runtime/utils/tenacity_stop.py
@@ -1,12 +1,11 @@
-
-
 from tenacity import RetryCallState
 from tenacity.stop import stop_base
+
 from openhands.runtime.utils.shutdown_listener import should_exit
 
 
 class stop_if_should_exit(stop_base):
     """Stop if the should_exit flag is set."""
 
-    def __call__(self, retry_state: "RetryCallState") -> bool:
+    def __call__(self, retry_state: 'RetryCallState') -> bool:
         return should_exit()
diff --git a/openhands/security/analyzer.py b/openhands/security/analyzer.py
index fc3c39164a07..9ce1c11255a0 100644
--- a/openhands/security/analyzer.py
+++ b/openhands/security/analyzer.py
@@ -1,4 +1,5 @@
 from typing import Any
+from uuid import uuid4
 
 from fastapi import Request
 
@@ -19,7 +20,7 @@ def __init__(self, event_stream: EventStream):
         """
         self.event_stream = event_stream
         self.event_stream.subscribe(
-            EventStreamSubscriber.SECURITY_ANALYZER, self.on_event
+            EventStreamSubscriber.SECURITY_ANALYZER, self.on_event, str(uuid4())
         )
 
     async def on_event(self, event: Event) -> None:
diff --git a/openhands/security/invariant/analyzer.py b/openhands/security/invariant/analyzer.py
index ba7fb890eb94..0ba13b4ecddf 100644
--- a/openhands/security/invariant/analyzer.py
+++ b/openhands/security/invariant/analyzer.py
@@ -147,6 +147,7 @@ async def confirm(self, event: Event) -> None:
         new_event = action_from_dict(
             {'action': 'change_agent_state', 'args': {'agent_state': 'user_confirmed'}}
         )
+        # we should confirm only on agent actions
         event_source = event.source if event.source else EventSource.AGENT
         await call_sync_from_async(self.event_stream.add_event, new_event, event_source)
 
diff --git a/openhands/server/github.py b/openhands/server/github.py
new file mode 100644
index 000000000000..774b1697e7ad
--- /dev/null
+++ b/openhands/server/github.py
@@ -0,0 +1,128 @@
+import os
+
+import httpx
+
+from openhands.core.logger import openhands_logger as logger
+from openhands.server.sheets_client import GoogleSheetsClient
+
+GITHUB_CLIENT_ID = os.getenv('GITHUB_CLIENT_ID', '').strip()
+GITHUB_CLIENT_SECRET = os.getenv('GITHUB_CLIENT_SECRET', '').strip()
+
+
+class UserVerifier:
+    def __init__(self) -> None:
+        logger.info('Initializing UserVerifier')
+        self.file_users: list[str] | None = None
+        self.sheets_client: GoogleSheetsClient | None = None
+        self.spreadsheet_id: str | None = None
+
+        # Initialize from environment variables
+        self._init_file_users()
+        self._init_sheets_client()
+
+    def _init_file_users(self) -> None:
+        """Load users from text file if configured"""
+        waitlist = os.getenv('GITHUB_USER_LIST_FILE')
+        if not waitlist:
+            logger.info('GITHUB_USER_LIST_FILE not configured')
+            return
+
+        if not os.path.exists(waitlist):
+            logger.error(f'User list file not found: {waitlist}')
+            raise FileNotFoundError(f'User list file not found: {waitlist}')
+
+        try:
+            with open(waitlist, 'r') as f:
+                self.file_users = [line.strip() for line in f if line.strip()]
+            logger.info(
+                f'Successfully loaded {len(self.file_users)} users from {waitlist}'
+            )
+        except Exception as e:
+            logger.error(f'Error reading user list file {waitlist}: {str(e)}')
+
+    def _init_sheets_client(self) -> None:
+        """Initialize Google Sheets client if configured"""
+        sheet_id = os.getenv('GITHUB_USERS_SHEET_ID')
+
+        if not sheet_id:
+            logger.info('GITHUB_USERS_SHEET_ID not configured')
+            return
+
+        logger.info('Initializing Google Sheets integration')
+        self.sheets_client = GoogleSheetsClient()
+        self.spreadsheet_id = sheet_id
+
+    def is_active(self) -> bool:
+        return bool(self.file_users or (self.sheets_client and self.spreadsheet_id))
+
+    def is_user_allowed(self, username: str) -> bool:
+        """Check if user is allowed based on file and/or sheet configuration"""
+        if not self.is_active():
+            return True
+
+        logger.info(f'Checking if GitHub user {username} is allowed')
+        if self.file_users:
+            if username in self.file_users:
+                logger.info(f'User {username} found in text file allowlist')
+                return True
+            logger.debug(f'User {username} not found in text file allowlist')
+
+        if self.sheets_client and self.spreadsheet_id:
+            sheet_users = self.sheets_client.get_usernames(self.spreadsheet_id)
+            if username in sheet_users:
+                logger.info(f'User {username} found in Google Sheets allowlist')
+                return True
+            logger.debug(f'User {username} not found in Google Sheets allowlist')
+
+        logger.info(f'User {username} not found in any allowlist')
+        return False
+
+
+async def authenticate_github_user(auth_token) -> bool:
+    user_verifier = UserVerifier()
+
+    if not user_verifier.is_active():
+        logger.info('No user verification sources configured - allowing all users')
+        return True
+
+    logger.info('Checking GitHub token')
+
+    if not auth_token:
+        logger.warning('No GitHub token provided')
+        return False
+
+    login = await get_github_user(auth_token)
+
+    if not user_verifier.is_user_allowed(login):
+        logger.warning(f'GitHub user {login} not in allow list')
+        return False
+
+    logger.info(f'GitHub user {login} authenticated')
+    return True
+
+
+async def get_github_user(token: str) -> str:
+    """Get GitHub user info from token.
+
+    Args:
+        token: GitHub access token
+
+    Returns:
+        Tuple of (login, error_message)
+        If successful, error_message is None
+        If failed, login is None and error_message contains the error
+    """
+    logger.info('Fetching GitHub user info from token')
+    headers = {
+        'Accept': 'application/vnd.github+json',
+        'Authorization': f'Bearer {token}',
+        'X-GitHub-Api-Version': '2022-11-28',
+    }
+    async with httpx.AsyncClient() as client:
+        logger.debug('Making request to GitHub API')
+        response = await client.get('https://api.github.com/user', headers=headers)
+        response.raise_for_status()
+        user_data = response.json()
+        login = user_data.get('login')
+        logger.info(f'Successfully retrieved GitHub user: {login}')
+        return login
diff --git a/openhands/server/listen.py b/openhands/server/listen.py
index cc74d5ba7361..7ccc7046594a 100644
--- a/openhands/server/listen.py
+++ b/openhands/server/listen.py
@@ -13,6 +13,11 @@
 
 from openhands.security.options import SecurityAnalyzers
 from openhands.server.data_models.feedback import FeedbackDataModel, store_feedback
+from openhands.server.github import (
+    GITHUB_CLIENT_ID,
+    GITHUB_CLIENT_SECRET,
+    authenticate_github_user,
+)
 from openhands.storage import get_file_store
 from openhands.utils.async_utils import call_sync_from_async
 
@@ -52,6 +57,7 @@
     NullObservation,
 )
 from openhands.events.serialization import event_to_dict
+from openhands.events.stream import AsyncEventStreamWrapper
 from openhands.llm import bedrock
 from openhands.runtime.base import Runtime
 from openhands.server.auth import get_sid_from_token, sign_token
@@ -64,24 +70,6 @@
 file_store = get_file_store(config.file_store, config.file_store_path)
 session_manager = SessionManager(config, file_store)
 
-GITHUB_CLIENT_ID = os.getenv('GITHUB_CLIENT_ID', '').strip()
-GITHUB_CLIENT_SECRET = os.getenv('GITHUB_CLIENT_SECRET', '').strip()
-
-# New global variable to store the user list
-GITHUB_USER_LIST = None
-
-
-# New function to load the user list
-def load_github_user_list():
-    global GITHUB_USER_LIST
-    waitlist = os.getenv('GITHUB_USER_LIST_FILE')
-    if waitlist:
-        with open(waitlist, 'r') as f:
-            GITHUB_USER_LIST = [line.strip() for line in f if line.strip()]
-
-
-load_github_user_list()
-
 
 @asynccontextmanager
 async def lifespan(app: FastAPI):
@@ -216,7 +204,13 @@ async def attach_session(request: Request, call_next):
         response = await call_next(request)
         return response
 
-    # For all other methods, validate the Authorization header
+    github_token = request.headers.get('X-GitHub-Token')
+    if not await authenticate_github_user(github_token):
+        return JSONResponse(
+            status_code=status.HTTP_401_UNAUTHORIZED,
+            content={'error': 'Not authenticated'},
+        )
+
     if not request.headers.get('Authorization'):
         logger.warning('Missing Authorization header')
         return JSONResponse(
@@ -308,11 +302,28 @@ async def websocket_endpoint(websocket: WebSocket):
         {"action": "finish", "args": {}}
         ```
     """
-    await asyncio.wait_for(websocket.accept(), 10)
+    # Get protocols from Sec-WebSocket-Protocol header
+    protocols = websocket.headers.get('sec-websocket-protocol', '').split(', ')
+
+    # The first protocol should be our real protocol (e.g. 'openhands')
+    # The second protocol should contain our auth token
+    if len(protocols) < 3:
+        logger.error('Expected 3 websocket protocols, got %d', len(protocols))
+        await websocket.close(code=status.WS_1008_POLICY_VIOLATION)
+        return
 
-    if websocket.query_params.get('token'):
-        token = websocket.query_params.get('token')
-        sid = get_sid_from_token(token, config.jwt_secret)
+    real_protocol = protocols[0]
+    jwt_token = protocols[1] if protocols[1] != 'NO_JWT' else ''
+    github_token = protocols[2] if protocols[2] != 'NO_GITHUB' else ''
+
+    if not await authenticate_github_user(github_token):
+        await websocket.close(code=status.WS_1008_POLICY_VIOLATION)
+        return
+
+    await asyncio.wait_for(websocket.accept(subprotocol=real_protocol), 10)
+
+    if jwt_token:
+        sid = get_sid_from_token(jwt_token, config.jwt_secret)
 
         if sid == '':
             await websocket.send_json({'error': 'Invalid token', 'error_code': 401})
@@ -320,18 +331,21 @@ async def websocket_endpoint(websocket: WebSocket):
             return
     else:
         sid = str(uuid.uuid4())
-        token = sign_token({'sid': sid}, config.jwt_secret)
+        jwt_token = sign_token({'sid': sid}, config.jwt_secret)
 
     logger.info(f'New session: {sid}')
     session = session_manager.add_or_restart_session(sid, websocket)
-    await websocket.send_json({'token': token, 'status': 'ok'})
+    await websocket.send_json({'token': jwt_token, 'status': 'ok'})
 
     latest_event_id = -1
     if websocket.query_params.get('latest_event_id'):
         latest_event_id = int(websocket.query_params.get('latest_event_id'))
-    for event in session.agent_session.event_stream.get_events(
-        start_id=latest_event_id + 1
-    ):
+
+    async_stream = AsyncEventStreamWrapper(
+        session.agent_session.event_stream, latest_event_id + 1
+    )
+
+    async for event in async_stream:
         if isinstance(
             event,
             (
@@ -469,19 +483,17 @@ async def list_files(request: Request, path: str | None = None):
         )
 
     runtime: Runtime = request.state.conversation.runtime
-    file_list = await asyncio.create_task(
-        call_sync_from_async(runtime.list_files, path)
-    )
+    file_list = await call_sync_from_async(runtime.list_files, path)
     if path:
         file_list = [os.path.join(path, f) for f in file_list]
 
     file_list = [f for f in file_list if f not in FILES_TO_IGNORE]
 
-    def filter_for_gitignore(file_list, base_path):
+    async def filter_for_gitignore(file_list, base_path):
         gitignore_path = os.path.join(base_path, '.gitignore')
         try:
             read_action = FileReadAction(gitignore_path)
-            observation = runtime.run_action(read_action)
+            observation = await call_sync_from_async(runtime.run_action, read_action)
             spec = PathSpec.from_lines(
                 GitWildMatchPattern, observation.content.splitlines()
             )
@@ -491,7 +503,7 @@ def filter_for_gitignore(file_list, base_path):
         file_list = [entry for entry in file_list if not spec.match_file(entry)]
         return file_list
 
-    file_list = filter_for_gitignore(file_list, '')
+    file_list = await filter_for_gitignore(file_list, '')
 
     return file_list
 
@@ -657,9 +669,11 @@ async def submit_feedback(request: Request):
     # Assuming the storage service is already configured in the backend
     # and there is a function to handle the storage.
     body = await request.json()
-    events = request.state.conversation.event_stream.get_events(filter_hidden=True)
+    async_stream = AsyncEventStreamWrapper(
+        request.state.conversation.event_stream, filter_hidden=True
+    )
     trajectory = []
-    for event in events:
+    async for event in async_stream:
         trajectory.append(event_to_dict(event))
     feedback = FeedbackDataModel(
         email=body.get('email', ''),
@@ -670,7 +684,7 @@ async def submit_feedback(request: Request):
         trajectory=trajectory,
     )
     try:
-        feedback_data = store_feedback(feedback)
+        feedback_data = await call_sync_from_async(store_feedback, feedback)
         return JSONResponse(status_code=200, content=feedback_data)
     except Exception as e:
         logger.error(f'Error submitting feedback: {e}')
@@ -840,26 +854,21 @@ def github_callback(auth_code: AuthCode):
     )
 
 
-class User(BaseModel):
-    login: str  # GitHub login handle
-
-
 @app.post('/api/authenticate')
-def authenticate(user: User | None = None):
-    global GITHUB_USER_LIST
-
-    # Only check if waitlist is provided
-    if GITHUB_USER_LIST:
-        if user is None or user.login not in GITHUB_USER_LIST:
-            return JSONResponse(
-                status_code=status.HTTP_403_FORBIDDEN,
-                content={'error': 'User not on waitlist'},
-            )
+async def authenticate(request: Request):
+    token = request.headers.get('X-GitHub-Token')
+    if not await authenticate_github_user(token):
+        return JSONResponse(
+            status_code=status.HTTP_401_UNAUTHORIZED,
+            content={'error': 'Not authorized via GitHub waitlist'},
+        )
 
-    return JSONResponse(
+    response = JSONResponse(
         status_code=status.HTTP_200_OK, content={'message': 'User authenticated'}
     )
 
+    return response
+
 
 class SPAStaticFiles(StaticFiles):
     async def get_response(self, path: str, scope):
diff --git a/openhands/server/middleware.py b/openhands/server/middleware.py
index f09ac0788ae2..218a949fca58 100644
--- a/openhands/server/middleware.py
+++ b/openhands/server/middleware.py
@@ -14,7 +14,7 @@ class LocalhostCORSMiddleware(CORSMiddleware):
     def __init__(self, app: ASGIApp, **kwargs) -> None:
         super().__init__(app, **kwargs)
 
-    async def is_allowed_origin(self, origin: str) -> bool:
+    def is_allowed_origin(self, origin: str) -> bool:
         if origin:
             parsed = urlparse(origin)
             hostname = parsed.hostname or ''
@@ -24,7 +24,7 @@ async def is_allowed_origin(self, origin: str) -> bool:
                 return True
 
         # For missing origin or other origins, use the parent class's logic
-        return await super().is_allowed_origin(origin)
+        return super().is_allowed_origin(origin)
 
 
 class NoCacheMiddleware(BaseHTTPMiddleware):
diff --git a/openhands/server/session/agent_session.py b/openhands/server/session/agent_session.py
index 41ee96889168..8e7376a7668b 100644
--- a/openhands/server/session/agent_session.py
+++ b/openhands/server/session/agent_session.py
@@ -32,7 +32,12 @@ class AgentSession:
     _closed: bool = False
     loop: asyncio.AbstractEventLoop | None = None
 
-    def __init__(self, sid: str, file_store: FileStore):
+    def __init__(
+        self,
+        sid: str,
+        file_store: FileStore,
+        status_callback: Optional[Callable] = None,
+    ):
         """Initializes a new instance of the Session class
 
         Parameters:
@@ -43,6 +48,7 @@ def __init__(self, sid: str, file_store: FileStore):
         self.sid = sid
         self.event_stream = EventStream(sid, file_store)
         self.file_store = file_store
+        self._status_callback = status_callback
 
     async def start(
         self,
@@ -53,7 +59,6 @@ async def start(
         max_budget_per_task: float | None = None,
         agent_to_llm_config: dict[str, LLMConfig] | None = None,
         agent_configs: dict[str, AgentConfig] | None = None,
-        status_message_callback: Optional[Callable] = None,
     ):
         """Starts the Agent session
         Parameters:
@@ -80,7 +85,6 @@ async def start(
             max_budget_per_task,
             agent_to_llm_config,
             agent_configs,
-            status_message_callback,
         )
 
     def _start_thread(self, *args):
@@ -99,15 +103,12 @@ async def _start(
         max_budget_per_task: float | None = None,
         agent_to_llm_config: dict[str, LLMConfig] | None = None,
         agent_configs: dict[str, AgentConfig] | None = None,
-        status_message_callback: Optional[Callable] = None,
     ):
-        self.loop = asyncio.get_running_loop()
         self._create_security_analyzer(config.security.security_analyzer)
         await self._create_runtime(
             runtime_name=runtime_name,
             config=config,
             agent=agent,
-            status_message_callback=status_message_callback,
         )
         self._create_controller(
             agent,
@@ -118,17 +119,25 @@ async def _start(
             agent_configs=agent_configs,
         )
         self.event_stream.add_event(
-            ChangeAgentStateAction(AgentState.INIT), EventSource.USER
+            ChangeAgentStateAction(AgentState.INIT), EventSource.ENVIRONMENT
         )
         if self.controller:
             self.controller.agent_task = self.controller.start_step_loop()
             await self.controller.agent_task  # type: ignore
 
-    async def close(self):
+    def close(self):
         """Closes the Agent session"""
-
         if self._closed:
             return
+
+        self._closed = True
+
+        def inner_close():
+            asyncio.run(self._close())
+
+        asyncio.get_event_loop().run_in_executor(None, inner_close)
+
+    async def _close(self):
         if self.controller is not None:
             end_state = self.controller.get_state()
             end_state.save_to_session(self.sid, self.file_store)
@@ -138,10 +147,9 @@ async def close(self):
         if self.security_analyzer is not None:
             await self.security_analyzer.close()
 
-        if self.loop:
-            self.loop.stop()
-
-        self._closed = True
+    async def stop_agent_loop_for_error(self):
+        if self.controller is not None:
+            await self.controller.set_agent_state_to(AgentState.ERROR)
 
     def _create_security_analyzer(self, security_analyzer: str | None):
         """Creates a SecurityAnalyzer instance that will be used to analyze the agent actions
@@ -161,7 +169,6 @@ async def _create_runtime(
         runtime_name: str,
         config: AppConfig,
         agent: Agent,
-        status_message_callback: Optional[Callable] = None,
     ):
         """Creates a runtime instance
 
@@ -181,13 +188,17 @@ async def _create_runtime(
             event_stream=self.event_stream,
             sid=self.sid,
             plugins=agent.sandbox_plugins,
-            status_message_callback=status_message_callback,
+            status_callback=self._status_callback,
         )
 
         try:
             await self.runtime.connect()
         except Exception as e:
             logger.error(f'Runtime initialization failed: {e}', exc_info=True)
+            if self._status_callback:
+                self._status_callback(
+                    'error', 'STATUS$ERROR_RUNTIME_DISCONNECTED', str(e)
+                )
             raise
 
         if self.runtime is not None:
@@ -251,9 +262,8 @@ def _create_controller(
             agent_to_llm_config=agent_to_llm_config,
             agent_configs=agent_configs,
             confirmation_mode=confirmation_mode,
-            # AgentSession is designed to communicate with the frontend, so we don't want to
-            # run the agent in headless mode.
             headless_mode=False,
+            status_callback=self._status_callback,
         )
         try:
             agent_state = State.restore_from_session(self.sid, self.file_store)
diff --git a/openhands/server/session/manager.py b/openhands/server/session/manager.py
index a2e8a688ebe3..15f7fbde4402 100644
--- a/openhands/server/session/manager.py
+++ b/openhands/server/session/manager.py
@@ -35,7 +35,7 @@ async def __aexit__(self, exc_type, exc_value, traceback):
 
     def add_or_restart_session(self, sid: str, ws_conn: WebSocket) -> Session:
         if sid in self._sessions:
-            asyncio.create_task(self._sessions[sid].close())
+            self._sessions[sid].close()
         self._sessions[sid] = Session(
             sid=sid, file_store=self.file_store, ws=ws_conn, config=self.config
         )
@@ -47,7 +47,7 @@ def get_session(self, sid: str) -> Session | None:
         return self._sessions.get(sid)
 
     async def attach_to_conversation(self, sid: str) -> Conversation | None:
-        if not session_exists(sid, self.file_store):
+        if not await session_exists(sid, self.file_store):
             return None
         c = Conversation(sid, file_store=self.file_store, config=self.config)
         await c.connect()
@@ -87,7 +87,7 @@ async def _cleanup_sessions(self):
             for sid in session_ids_to_remove:
                 to_del_session: Session | None = self._sessions.pop(sid, None)
                 if to_del_session is not None:
-                    await to_del_session.close()
+                    to_del_session.close()
                     logger.debug(
                         f'Session {sid} and related resource have been removed due to inactivity.'
                     )
diff --git a/openhands/server/session/session.py b/openhands/server/session/session.py
index 4e6119a18560..e2f3067d583d 100644
--- a/openhands/server/session/session.py
+++ b/openhands/server/session/session.py
@@ -25,8 +25,6 @@
 from openhands.server.session.agent_session import AgentSession
 from openhands.storage.files import FileStore
 
-DEL_DELT_SEC = 60 * 60 * 5
-
 
 class Session:
     sid: str
@@ -42,16 +40,18 @@ def __init__(
         self.sid = sid
         self.websocket = ws
         self.last_active_ts = int(time.time())
-        self.agent_session = AgentSession(sid, file_store)
+        self.agent_session = AgentSession(
+            sid, file_store, status_callback=self.queue_status_message
+        )
         self.agent_session.event_stream.subscribe(
-            EventStreamSubscriber.SERVER, self.on_event
+            EventStreamSubscriber.SERVER, self.on_event, self.sid
         )
         self.config = config
         self.loop = asyncio.get_event_loop()
 
-    async def close(self):
+    def close(self):
         self.is_alive = False
-        await self.agent_session.close()
+        self.agent_session.close()
 
     async def loop_recv(self):
         try:
@@ -65,18 +65,19 @@ async def loop_recv(self):
                     continue
                 await self.dispatch(data)
         except WebSocketDisconnect:
-            await self.close()
-            logger.debug('WebSocket disconnected, sid: %s', self.sid)
+            logger.info('WebSocket disconnected, sid: %s', self.sid)
+            self.close()
         except RuntimeError as e:
-            await self.close()
             logger.exception('Error in loop_recv: %s', e)
+            self.close()
 
     async def _initialize_agent(self, data: dict):
         self.agent_session.event_stream.add_event(
-            ChangeAgentStateAction(AgentState.LOADING), EventSource.USER
+            ChangeAgentStateAction(AgentState.LOADING), EventSource.ENVIRONMENT
         )
         self.agent_session.event_stream.add_event(
-            AgentStateChangedObservation('', AgentState.LOADING), EventSource.AGENT
+            AgentStateChangedObservation('', AgentState.LOADING),
+            EventSource.ENVIRONMENT,
         )
         # Extract the agent-relevant arguments from the request
         args = {key: value for key, value in data.get('args', {}).items()}
@@ -116,7 +117,6 @@ async def _initialize_agent(self, data: dict):
                 max_budget_per_task=self.config.max_budget_per_task,
                 agent_to_llm_config=self.config.get_agent_to_llm_config_map(),
                 agent_configs=self.config.get_agent_configs(),
-                status_message_callback=self.queue_status_message,
             )
         except Exception as e:
             logger.exception(f'Error creating controller: {e}')
@@ -138,12 +138,19 @@ async def on_event(self, event: Event):
             return
         if event.source == EventSource.AGENT:
             await self.send(event_to_dict(event))
-        elif event.source == EventSource.USER and isinstance(
-            event, CmdOutputObservation
+        # NOTE: ipython observations are not sent here currently
+        elif event.source == EventSource.ENVIRONMENT and isinstance(
+            event, (CmdOutputObservation, AgentStateChangedObservation)
         ):
-            await self.send(event_to_dict(event))
+            # feedback from the environment to agent actions is understood as agent events by the UI
+            event_dict = event_to_dict(event)
+            event_dict['source'] = EventSource.AGENT
+            await self.send(event_dict)
         elif isinstance(event, ErrorObservation):
-            await self.send(event_to_dict(event))
+            # send error events as agent events to the UI
+            event_dict = event_to_dict(event)
+            event_dict['source'] = EventSource.AGENT
+            await self.send(event_dict)
 
     async def dispatch(self, data: dict):
         action = data.get('action', '')
@@ -165,12 +172,6 @@ async def dispatch(self, data: dict):
                         'Model does not support image upload, change to a different model or try without an image.'
                     )
                     return
-        if self.agent_session.loop:
-            asyncio.run_coroutine_threadsafe(
-                self._add_event(event, EventSource.USER), self.agent_session.loop
-            )  # type: ignore
-
-    async def _add_event(self, event, event_source):
         self.agent_session.event_stream.add_event(event, EventSource.USER)
 
     async def send(self, data: dict[str, object]) -> bool:
@@ -181,10 +182,7 @@ async def send(self, data: dict[str, object]) -> bool:
             await asyncio.sleep(0.001)  # This flushes the data to the client
             self.last_active_ts = int(time.time())
             return True
-        except RuntimeError:
-            self.is_alive = False
-            return False
-        except WebSocketDisconnect:
+        except (RuntimeError, WebSocketDisconnect):
             self.is_alive = False
             return False
 
@@ -192,27 +190,17 @@ async def send_error(self, message: str) -> bool:
         """Sends an error message to the client."""
         return await self.send({'error': True, 'message': message})
 
-    async def send_message(self, message: str) -> bool:
-        """Sends a message to the client."""
-        return await self.send({'message': message})
-
-    async def send_status_message(self, message: str) -> bool:
+    async def _send_status_message(self, msg_type: str, id: str, message: str) -> bool:
         """Sends a status message to the client."""
-        return await self.send({'status': message})
-
-    def update_connection(self, ws: WebSocket):
-        self.websocket = ws
-        self.is_alive = True
-        self.last_active_ts = int(time.time())
+        if msg_type == 'error':
+            await self.agent_session.stop_agent_loop_for_error()
 
-    def load_from_data(self, data: dict) -> bool:
-        self.last_active_ts = data.get('last_active_ts', 0)
-        if self.last_active_ts < int(time.time()) - DEL_DELT_SEC:
-            return False
-        self.is_alive = data.get('is_alive', False)
-        return True
+        return await self.send(
+            {'status_update': True, 'type': msg_type, 'id': id, 'message': message}
+        )
 
-    def queue_status_message(self, message: str):
+    def queue_status_message(self, msg_type: str, id: str, message: str):
         """Queues a status message to be sent asynchronously."""
-        # Ensure the coroutine runs in the main event loop
-        asyncio.run_coroutine_threadsafe(self.send_status_message(message), self.loop)
+        asyncio.run_coroutine_threadsafe(
+            self._send_status_message(msg_type, id, message), self.loop
+        )
diff --git a/openhands/server/sheets_client.py b/openhands/server/sheets_client.py
new file mode 100644
index 000000000000..c2db1a343477
--- /dev/null
+++ b/openhands/server/sheets_client.py
@@ -0,0 +1,68 @@
+from typing import List
+
+from google.auth import default
+from googleapiclient.discovery import build
+from googleapiclient.errors import HttpError
+
+from openhands.core.logger import openhands_logger as logger
+
+
+class GoogleSheetsClient:
+    def __init__(self):
+        """Initialize Google Sheets client using workload identity.
+        Uses application default credentials which supports workload identity when running in GCP.
+        """
+        logger.info('Initializing Google Sheets client with workload identity')
+        try:
+            credentials, project = default(
+                scopes=['https://www.googleapis.com/auth/spreadsheets.readonly']
+            )
+            logger.info(f'Successfully obtained credentials for project: {project}')
+            self.service = build('sheets', 'v4', credentials=credentials)
+            logger.info('Successfully initialized Google Sheets API service')
+        except Exception as e:
+            logger.error(f'Failed to initialize Google Sheets client: {str(e)}')
+            self.service = None
+
+    def get_usernames(self, spreadsheet_id: str, range_name: str = 'A:A') -> List[str]:
+        """Get list of usernames from specified Google Sheet.
+
+        Args:
+            spreadsheet_id: The ID of the Google Sheet
+            range_name: The A1 notation of the range to fetch
+
+        Returns:
+            List of usernames from the sheet
+        """
+        if not self.service:
+            logger.error('Google Sheets service not initialized')
+            return []
+
+        try:
+            logger.info(
+                f'Fetching usernames from sheet {spreadsheet_id}, range {range_name}'
+            )
+            result = (
+                self.service.spreadsheets()
+                .values()
+                .get(spreadsheetId=spreadsheet_id, range=range_name)
+                .execute()
+            )
+
+            values = result.get('values', [])
+            usernames = [
+                str(cell[0]).strip() for cell in values if cell and cell[0].strip()
+            ]
+            logger.info(
+                f'Successfully fetched {len(usernames)} usernames from Google Sheet'
+            )
+            return usernames
+
+        except HttpError as err:
+            logger.error(f'Error accessing Google Sheet {spreadsheet_id}: {err}')
+            return []
+        except Exception as e:
+            logger.error(
+                f'Unexpected error accessing Google Sheet {spreadsheet_id}: {str(e)}'
+            )
+            return []
diff --git a/poetry.lock b/poetry.lock
index 6481fe5bafa5..6a2791471358 100644
--- a/poetry.lock
+++ b/poetry.lock
@@ -1,4 +1,4 @@
-# This file is automatically @generated by Poetry 1.8.3 and should not be changed by hand.
+# This file is automatically @generated by Poetry 1.8.4 and should not be changed by hand.
 
 [[package]]
 name = "aenum"
@@ -2319,6 +2319,24 @@ files = [
 google-auth = "*"
 httplib2 = ">=0.19.0"
 
+[[package]]
+name = "google-auth-oauthlib"
+version = "1.2.1"
+description = "Google Authentication Library"
+optional = false
+python-versions = ">=3.6"
+files = [
+    {file = "google_auth_oauthlib-1.2.1-py2.py3-none-any.whl", hash = "sha256:2d58a27262d55aa1b87678c3ba7142a080098cbc2024f903c62355deb235d91f"},
+    {file = "google_auth_oauthlib-1.2.1.tar.gz", hash = "sha256:afd0cad092a2eaa53cd8e8298557d6de1034c6cb4a740500b5357b648af97263"},
+]
+
+[package.dependencies]
+google-auth = ">=2.15.0"
+requests-oauthlib = ">=0.7.0"
+
+[package.extras]
+tool = ["click (>=6.0.0)"]
+
 [[package]]
 name = "google-cloud-aiplatform"
 version = "1.70.0"
@@ -10109,4 +10127,4 @@ testing = ["coverage[toml]", "zope.event", "zope.testing"]
 [metadata]
 lock-version = "2.0"
 python-versions = "^3.12"
-content-hash = "2b268ef696ace0d8170276407dbdeb414134477839ebe4b7ecf29b1a1fe2cef3"
+content-hash = "2a4f90bb5c7f7d82160f57d71af7e81c7acef69426d0e1e46e1da09972a6215f"
diff --git a/pyproject.toml b/pyproject.toml
index b07fc0aa29a8..5f05b4da96ed 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -1,6 +1,6 @@
 [tool.poetry]
 name = "openhands-ai"
-version = "0.12.0"
+version = "0.12.3"
 description = "OpenHands: Code Less, Make More"
 authors = ["OpenHands"]
 license = "MIT"
@@ -16,6 +16,9 @@ datasets = "*"
 pandas = "*"
 litellm = "^1.51.1"
 google-generativeai = "*" # To use litellm with Gemini Pro API
+google-api-python-client = "*" # For Google Sheets API
+google-auth-httplib2 = "*" # For Google Sheets authentication
+google-auth-oauthlib = "*" # For Google Sheets OAuth
 termcolor = "*"
 seaborn = "*"
 docker = "*"
@@ -89,6 +92,7 @@ reportlab = "*"
 [tool.coverage.run]
 concurrency = ["gevent"]
 
+
 [tool.poetry.group.runtime.dependencies]
 jupyterlab = "*"
 notebook = "*"
@@ -119,6 +123,7 @@ ignore = ["D1"]
 [tool.ruff.lint.pydocstyle]
 convention = "google"
 
+
 [tool.poetry.group.evaluation.dependencies]
 streamlit = "*"
 whatthepatch = "*"
diff --git a/tests/runtime/test_stress_remote_runtime.py b/tests/runtime/test_stress_remote_runtime.py
new file mode 100644
index 000000000000..3a5d6d280726
--- /dev/null
+++ b/tests/runtime/test_stress_remote_runtime.py
@@ -0,0 +1,231 @@
+"""Bash-related tests for the EventStreamRuntime, which connects to the ActionExecutor running in the sandbox."""
+
+import asyncio
+import os
+import tempfile
+from unittest.mock import MagicMock
+
+import pandas as pd
+import pytest
+from conftest import TEST_IN_CI
+
+from evaluation.utils.shared import (
+    EvalException,
+    EvalMetadata,
+    EvalOutput,
+    assert_and_raise,
+    codeact_user_response,
+    make_metadata,
+    prepare_dataset,
+    reset_logger_for_multiprocessing,
+    run_evaluation,
+)
+from openhands.agenthub import Agent
+from openhands.controller.state.state import State
+from openhands.core.config import (
+    AgentConfig,
+    AppConfig,
+    LLMConfig,
+    SandboxConfig,
+)
+from openhands.core.logger import openhands_logger as logger
+from openhands.core.main import create_runtime, run_controller
+from openhands.events.action import CmdRunAction, MessageAction
+from openhands.events.observation import CmdOutputObservation
+from openhands.events.serialization.event import event_to_dict
+from openhands.llm import LLM
+from openhands.runtime.base import Runtime
+from openhands.utils.async_utils import call_async_from_sync
+
+AGENT_CLS_TO_FAKE_USER_RESPONSE_FN = {
+    'CodeActAgent': codeact_user_response,
+}
+
+
+def get_config(
+    metadata: EvalMetadata,
+) -> AppConfig:
+    assert (
+        os.environ.get('SANDBOX_REMOTE_RUNTIME_API_URL') is not None
+    ), 'SANDBOX_REMOTE_RUNTIME_API_URL must be set.'
+    assert (
+        os.environ.get('ALLHANDS_API_KEY') is not None
+    ), 'ALLHANDS_API_KEY must be set.'
+    config = AppConfig(
+        default_agent=metadata.agent_class,
+        run_as_openhands=False,
+        max_iterations=metadata.max_iterations,
+        runtime='remote',
+        sandbox=SandboxConfig(
+            base_container_image='python:3.11-bookworm',
+            enable_auto_lint=True,
+            use_host_network=False,
+            # large enough timeout, since some testcases take very long to run
+            timeout=300,
+            api_key=os.environ.get('ALLHANDS_API_KEY', None),
+            remote_runtime_api_url=os.environ.get('SANDBOX_REMOTE_RUNTIME_API_URL'),
+            keep_remote_runtime_alive=False,
+        ),
+        # do not mount workspace
+        workspace_base=None,
+        workspace_mount_path=None,
+    )
+    agent_config = AgentConfig(
+        codeact_enable_jupyter=False,
+        codeact_enable_browsing=False,
+        codeact_enable_llm_editor=False,
+    )
+    config.set_agent_config(agent_config)
+    return config
+
+
+def initialize_runtime(
+    runtime: Runtime,
+):
+    """Initialize the runtime for the agent.
+
+    This function is called before the runtime is used to run the agent.
+    """
+    logger.info('-' * 30)
+    logger.info('BEGIN Runtime Initialization Fn')
+    logger.info('-' * 30)
+    obs: CmdOutputObservation
+
+    action = CmdRunAction(command="""export USER=$(whoami); echo USER=${USER} """)
+    action.timeout = 600
+    logger.info(action, extra={'msg_type': 'ACTION'})
+    obs = runtime.run_action(action)
+    logger.info(obs, extra={'msg_type': 'OBSERVATION'})
+    assert_and_raise(obs.exit_code == 0, f'Failed to export USER: {str(obs)}')
+
+    action = CmdRunAction(command='mkdir -p /dummy_dir')
+    action.timeout = 600
+    logger.info(action, extra={'msg_type': 'ACTION'})
+    obs = runtime.run_action(action)
+    logger.info(obs, extra={'msg_type': 'OBSERVATION'})
+    assert_and_raise(
+        obs.exit_code == 0,
+        f'Failed to create /dummy_dir: {str(obs)}',
+    )
+
+    with tempfile.TemporaryDirectory() as temp_dir:
+        # Construct the full path for the desired file name within the temporary directory
+        temp_file_path = os.path.join(temp_dir, 'dummy_file')
+        # Write to the file with the desired name within the temporary directory
+        with open(temp_file_path, 'w') as f:
+            f.write('dummy content')
+
+        # Copy the file to the desired location
+        runtime.copy_to(temp_file_path, '/dummy_dir/')
+
+    logger.info('-' * 30)
+    logger.info('END Runtime Initialization Fn')
+    logger.info('-' * 30)
+
+
+def process_instance(
+    instance: pd.Series,
+    metadata: EvalMetadata,
+    reset_logger: bool = True,
+) -> EvalOutput:
+    config = get_config(metadata)
+
+    # Setup the logger properly, so you can run multi-processing to parallelize the evaluation
+    if reset_logger:
+        log_dir = os.path.join(metadata.eval_output_dir, 'infer_logs')
+        reset_logger_for_multiprocessing(logger, instance.instance_id, log_dir)
+    else:
+        logger.info(f'Starting evaluation for instance {instance.instance_id}.')
+
+    runtime = create_runtime(config)
+    call_async_from_sync(runtime.connect)
+
+    try:
+        initialize_runtime(runtime)
+
+        instruction = 'dummy instruction'
+        agent = Agent.get_cls(metadata.agent_class)(
+            llm=LLM(config=metadata.llm_config),
+            config=config.get_agent_config(metadata.agent_class),
+        )
+
+        def next_command(*args, **kwargs):
+            return CmdRunAction(command='ls -lah')
+
+        agent.step = MagicMock(side_effect=next_command)
+
+        # Here's how you can run the agent (similar to the `main` function) and get the final task state
+        state: State | None = asyncio.run(
+            run_controller(
+                config=config,
+                initial_user_action=MessageAction(content=instruction),
+                runtime=runtime,
+                fake_user_response_fn=AGENT_CLS_TO_FAKE_USER_RESPONSE_FN[
+                    metadata.agent_class
+                ],
+                agent=agent,
+            )
+        )
+
+        # if fatal error, throw EvalError to trigger re-run
+        if (
+            state.last_error
+            and 'fatal error during agent execution' in state.last_error
+            and 'stuck in a loop' not in state.last_error
+        ):
+            raise EvalException('Fatal error detected: ' + state.last_error)
+
+    finally:
+        runtime.close()
+
+    test_result = {}
+    if state is None:
+        raise ValueError('State should not be None.')
+    histories = [event_to_dict(event) for event in state.history]
+    metrics = state.metrics.get() if state.metrics else None
+
+    # Save the output
+    output = EvalOutput(
+        instance_id=instance.instance_id,
+        instruction=instruction,
+        instance=instance.to_dict(),  # SWE Bench specific
+        test_result=test_result,
+        metadata=metadata,
+        history=histories,
+        metrics=metrics,
+        error=state.last_error if state and state.last_error else None,
+    )
+    return output
+
+
+@pytest.mark.skipif(
+    TEST_IN_CI,
+    reason='This test should only be run locally, not in CI.',
+)
+def test_stress_remote_runtime(n_eval_workers: int = 64):
+    """Mimic evaluation setting to test remote runtime in a multi-processing setting."""
+
+    llm_config = LLMConfig()
+    metadata = make_metadata(
+        llm_config,
+        'dummy_dataset_descrption',
+        'CodeActAgent',
+        max_iterations=10,
+        eval_note='dummy_eval_note',
+        eval_output_dir='./dummy_eval_output_dir',
+        details={},
+    )
+
+    # generate 300 random dummy instances
+    dummy_instance = pd.DataFrame(
+        {
+            'instance_id': [f'dummy_instance_{i}' for i in range(300)],
+        }
+    )
+
+    output_file = os.path.join(metadata.eval_output_dir, 'output.jsonl')
+    instances = prepare_dataset(
+        dummy_instance, output_file, eval_n_limit=len(dummy_instance)
+    )
+
+    run_evaluation(instances, metadata, output_file, n_eval_workers, process_instance)
diff --git a/tests/unit/test_agent_controller.py b/tests/unit/test_agent_controller.py
index 9b0522302f9f..9c07969bd090 100644
--- a/tests/unit/test_agent_controller.py
+++ b/tests/unit/test_agent_controller.py
@@ -1,5 +1,6 @@
 import asyncio
 from unittest.mock import AsyncMock, MagicMock, Mock
+from uuid import uuid4
 
 import pytest
 
@@ -7,14 +8,12 @@
 from openhands.controller.agent_controller import AgentController
 from openhands.controller.state.state import TrafficControlState
 from openhands.core.config import AppConfig
-from openhands.core.exceptions import LLMMalformedActionError
 from openhands.core.main import run_controller
 from openhands.core.schema import AgentState
 from openhands.events import Event, EventSource, EventStream, EventStreamSubscriber
 from openhands.events.action import ChangeAgentStateAction, CmdRunAction, MessageAction
 from openhands.events.observation import (
     ErrorObservation,
-    FatalErrorObservation,
 )
 from openhands.events.serialization import event_to_dict
 from openhands.llm import LLM
@@ -45,6 +44,11 @@ def mock_event_stream():
     return MagicMock(spec=EventStream)
 
 
+@pytest.fixture
+def mock_status_callback():
+    return AsyncMock()
+
+
 @pytest.mark.asyncio
 async def test_set_agent_state(mock_agent, mock_event_stream):
     controller = AgentController(
@@ -98,39 +102,19 @@ async def test_on_event_change_agent_state_action(mock_agent, mock_event_stream)
 
 
 @pytest.mark.asyncio
-async def test_report_error(mock_agent, mock_event_stream):
+async def test_react_to_exception(mock_agent, mock_event_stream, mock_status_callback):
     controller = AgentController(
         agent=mock_agent,
         event_stream=mock_event_stream,
+        status_callback=mock_status_callback,
         max_iterations=10,
         sid='test',
         confirmation_mode=False,
         headless_mode=True,
     )
     error_message = 'Test error'
-    await controller.report_error(error_message)
-    assert controller.state.last_error == error_message
-    controller.event_stream.add_event.assert_called_once()
-    await controller.close()
-
-
-@pytest.mark.asyncio
-async def test_step_with_exception(mock_agent, mock_event_stream):
-    controller = AgentController(
-        agent=mock_agent,
-        event_stream=mock_event_stream,
-        max_iterations=10,
-        sid='test',
-        confirmation_mode=False,
-        headless_mode=True,
-    )
-    controller.state.agent_state = AgentState.RUNNING
-    controller.report_error = AsyncMock()
-    controller.agent.step.side_effect = LLMMalformedActionError('Malformed action')
-    await controller._step()
-
-    # Verify that report_error was called with the correct error message
-    controller.report_error.assert_called_once_with('Malformed action')
+    await controller._react_to_exception(RuntimeError(error_message))
+    controller.status_callback.assert_called_once()
     await controller.close()
 
 
@@ -141,23 +125,26 @@ async def test_run_controller_with_fatal_error(mock_agent, mock_event_stream):
     event_stream = EventStream(sid='test', file_store=file_store)
 
     agent = MagicMock(spec=Agent)
-    # a random message to send to the runtime
-    event = CmdRunAction(command='ls')
-    agent.step.return_value = event
+    agent = MagicMock(spec=Agent)
+
+    def agent_step_fn(state):
+        print(f'agent_step_fn received state: {state}')
+        return CmdRunAction(command='ls')
+
+    agent.step = agent_step_fn
     agent.llm = MagicMock(spec=LLM)
     agent.llm.metrics = Metrics()
     agent.llm.config = config.get_llm_config()
 
-    fatal_error_obs = FatalErrorObservation('Fatal error detected')
-    fatal_error_obs._cause = event.id
-
     runtime = MagicMock(spec=Runtime)
 
     async def on_event(event: Event):
         if isinstance(event, CmdRunAction):
-            await event_stream.async_add_event(fatal_error_obs, EventSource.USER)
+            error_obs = ErrorObservation('You messed around with Jim')
+            error_obs._cause = event.id
+            event_stream.add_event(error_obs, EventSource.USER)
 
-    event_stream.subscribe(EventStreamSubscriber.RUNTIME, on_event)
+    event_stream.subscribe(EventStreamSubscriber.RUNTIME, on_event, str(uuid4()))
     runtime.event_stream = event_stream
 
     state = await run_controller(
@@ -170,30 +157,23 @@ async def on_event(event: Event):
     )
     print(f'state: {state}')
     print(f'event_stream: {list(event_stream.get_events())}')
-    assert state.iteration == 1
-    # it will first become AgentState.ERROR, then become AgentState.STOPPED
-    # in side run_controller (since the while loop + sleep no longer loop)
-    assert state.agent_state == AgentState.STOPPED
-    assert (
-        state.last_error
-        == 'There was a fatal error during agent execution: **FatalErrorObservation**\nFatal error detected'
-    )
-    assert len(list(event_stream.get_events())) == 5
+    assert state.iteration == 4
+    assert state.agent_state == AgentState.ERROR
+    assert state.last_error == 'Agent got stuck in a loop'
+    assert len(list(event_stream.get_events())) == 11
 
 
 @pytest.mark.asyncio
-async def test_run_controller_stop_with_stuck(mock_agent, mock_event_stream):
+async def test_run_controller_stop_with_stuck():
     config = AppConfig()
     file_store = get_file_store(config.file_store, config.file_store_path)
     event_stream = EventStream(sid='test', file_store=file_store)
 
     agent = MagicMock(spec=Agent)
-    # a random message to send to the runtime
-    event = CmdRunAction(command='ls')
 
     def agent_step_fn(state):
         print(f'agent_step_fn received state: {state}')
-        return event
+        return CmdRunAction(command='ls')
 
     agent.step = agent_step_fn
     agent.llm = MagicMock(spec=LLM)
@@ -207,9 +187,9 @@ async def on_event(event: Event):
                 'Non fatal error here to trigger loop'
             )
             non_fatal_error_obs._cause = event.id
-            await event_stream.async_add_event(non_fatal_error_obs, EventSource.USER)
+            event_stream.add_event(non_fatal_error_obs, EventSource.ENVIRONMENT)
 
-    event_stream.subscribe(EventStreamSubscriber.RUNTIME, on_event)
+    event_stream.subscribe(EventStreamSubscriber.RUNTIME, on_event, str(uuid4()))
     runtime.event_stream = event_stream
 
     state = await run_controller(
@@ -226,7 +206,7 @@ async def on_event(event: Event):
         print(f'event {i}: {event_to_dict(event)}')
 
     assert state.iteration == 4
-    assert len(events) == 12
+    assert len(events) == 11
     # check the eventstream have 4 pairs of repeated actions and observations
     repeating_actions_and_observations = events[2:10]
     for action, observation in zip(
@@ -244,13 +224,8 @@ async def on_event(event: Event):
     assert last_event['extras']['agent_state'] == 'error'
     assert last_event['observation'] == 'agent_state_changed'
 
-    # it will first become AgentState.ERROR, then become AgentState.STOPPED
-    # in side run_controller (since the while loop + sleep no longer loop)
-    assert state.agent_state == AgentState.STOPPED
-    assert (
-        state.last_error
-        == 'There was a fatal error during agent execution: **FatalErrorObservation**\nAgent got stuck in a loop'
-    )
+    assert state.agent_state == AgentState.ERROR
+    assert state.last_error == 'Agent got stuck in a loop'
 
 
 @pytest.mark.asyncio
@@ -317,7 +292,7 @@ async def test_step_max_iterations(mock_agent, mock_event_stream):
     assert controller.state.traffic_control_state == TrafficControlState.NORMAL
     await controller._step()
     assert controller.state.traffic_control_state == TrafficControlState.THROTTLING
-    assert controller.state.agent_state == AgentState.PAUSED
+    assert controller.state.agent_state == AgentState.ERROR
     await controller.close()
 
 
@@ -357,7 +332,7 @@ async def test_step_max_budget(mock_agent, mock_event_stream):
     assert controller.state.traffic_control_state == TrafficControlState.NORMAL
     await controller._step()
     assert controller.state.traffic_control_state == TrafficControlState.THROTTLING
-    assert controller.state.agent_state == AgentState.PAUSED
+    assert controller.state.agent_state == AgentState.ERROR
     await controller.close()
 
 
diff --git a/tests/unit/test_codeact_agent.py b/tests/unit/test_codeact_agent.py
index 9e3dda6c2cdd..126ce788c77c 100644
--- a/tests/unit/test_codeact_agent.py
+++ b/tests/unit/test_codeact_agent.py
@@ -104,5 +104,5 @@ def test_error_observation_message(agent: CodeActAgent):
 def test_unknown_observation_message(agent: CodeActAgent):
     obs = Mock()
 
-    with pytest.raises(ValueError, match='Unknown observation type:'):
+    with pytest.raises(ValueError, match='Unknown observation type'):
         agent.get_observation_message(obs, tool_call_id_to_message={})
diff --git a/tests/unit/test_is_stuck.py b/tests/unit/test_is_stuck.py
index 4a1330752161..197d6d8462b7 100644
--- a/tests/unit/test_is_stuck.py
+++ b/tests/unit/test_is_stuck.py
@@ -17,8 +17,6 @@
 from openhands.events.observation.empty import NullObservation
 from openhands.events.observation.error import ErrorObservation
 from openhands.events.stream import EventSource, EventStream
-from openhands.events.utils import get_pairs_from_events
-from openhands.memory.history import ShortTermHistory
 from openhands.storage import get_file_store
 
 
@@ -55,22 +53,21 @@ def event_stream(temp_dir):
 
 class TestStuckDetector:
     @pytest.fixture
-    def stuck_detector(self, event_stream):
+    def stuck_detector(self):
         state = State(inputs={}, max_iterations=50)
-        state.history.set_event_stream(event_stream)
-
+        state.history = []  # Initialize history as an empty list
         return StuckDetector(state)
 
     def _impl_syntax_error_events(
         self,
-        event_stream: EventStream,
+        state: State,
         error_message: str,
         random_line: bool,
         incidents: int = 4,
     ):
         for i in range(incidents):
             ipython_action = IPythonRunCellAction(code=code_snippet)
-            event_stream.add_event(ipython_action, EventSource.AGENT)
+            state.history.append(ipython_action)
             extra_number = (i + 1) * 10 if random_line else '42'
             extra_line = '\n' * (i + 1) if random_line else ''
             ipython_observation = IPythonRunCellObservation(
@@ -79,15 +76,15 @@ def _impl_syntax_error_events(
                 f'{error_message}{extra_line}' + jupyter_line_1 + jupyter_line_2,
                 code=code_snippet,
             )
-            ipython_observation._cause = ipython_action._id
-            event_stream.add_event(ipython_observation, EventSource.USER)
+            # ipython_observation._cause = ipython_action._id
+            state.history.append(ipython_observation)
 
     def _impl_unterminated_string_error_events(
-        self, event_stream: EventStream, random_line: bool, incidents: int = 4
+        self, state: State, random_line: bool, incidents: int = 4
     ):
         for i in range(incidents):
             ipython_action = IPythonRunCellAction(code=code_snippet)
-            event_stream.add_event(ipython_action, EventSource.AGENT)
+            state.history.append(ipython_action)
             line_number = (i + 1) * 10 if random_line else '1'
             ipython_observation = IPythonRunCellObservation(
                 content=f'print("  Cell In[1], line {line_number}\nhello\n       ^\nSyntaxError: unterminated string literal (detected at line {line_number})'
@@ -95,34 +92,30 @@ def _impl_unterminated_string_error_events(
                 + jupyter_line_2,
                 code=code_snippet,
             )
-            ipython_observation._cause = ipython_action._id
-            event_stream.add_event(ipython_observation, EventSource.USER)
+            # ipython_observation._cause = ipython_action._
+            state.history.append(ipython_observation)
 
-    def test_history_too_short(
-        self, stuck_detector: StuckDetector, event_stream: EventStream
-    ):
+    def test_history_too_short(self, stuck_detector: StuckDetector):
+        state = stuck_detector.state
         message_action = MessageAction(content='Hello', wait_for_response=False)
         message_action._source = EventSource.USER
         observation = NullObservation(content='')
-        observation._cause = message_action.id
-        event_stream.add_event(message_action, EventSource.USER)
-        event_stream.add_event(observation, EventSource.USER)
+        # observation._cause = message_action.id
+        state.history.append(message_action)
+        state.history.append(observation)
 
         cmd_action = CmdRunAction(command='ls')
-        event_stream.add_event(cmd_action, EventSource.AGENT)
+        state.history.append(cmd_action)
         cmd_observation = CmdOutputObservation(
             command_id=1, command='ls', content='file1.txt\nfile2.txt'
         )
-        cmd_observation._cause = cmd_action._id
-        event_stream.add_event(cmd_observation, EventSource.USER)
-
-        # stuck_detector.state.history.set_event_stream(event_stream)
+        # cmd_observation._cause = cmd_action._id
+        state.history.append(cmd_observation)
 
         assert stuck_detector.is_stuck() is False
 
-    def test_is_stuck_repeating_action_observation(
-        self, stuck_detector: StuckDetector, event_stream: EventStream
-    ):
+    def test_is_stuck_repeating_action_observation(self, stuck_detector: StuckDetector):
+        state = stuck_detector.state
         message_action = MessageAction(content='Done', wait_for_response=False)
         message_action._source = EventSource.USER
 
@@ -130,135 +123,125 @@ def test_is_stuck_repeating_action_observation(
         hello_observation = NullObservation('')
 
         # 2 events
-        event_stream.add_event(hello_action, EventSource.USER)
-        event_stream.add_event(hello_observation, EventSource.USER)
+        state.history.append(hello_action)
+        state.history.append(hello_observation)
 
         cmd_action_1 = CmdRunAction(command='ls')
-        event_stream.add_event(cmd_action_1, EventSource.AGENT)
-        cmd_observation_1 = CmdOutputObservation(
-            content='', command='ls', command_id=cmd_action_1._id
-        )
+        cmd_action_1._id = 1
+        state.history.append(cmd_action_1)
+        cmd_observation_1 = CmdOutputObservation(content='', command='ls', command_id=1)
         cmd_observation_1._cause = cmd_action_1._id
-        event_stream.add_event(cmd_observation_1, EventSource.USER)
+        state.history.append(cmd_observation_1)
         # 4 events
 
         cmd_action_2 = CmdRunAction(command='ls')
-        event_stream.add_event(cmd_action_2, EventSource.AGENT)
-        cmd_observation_2 = CmdOutputObservation(
-            content='', command='ls', command_id=cmd_action_2._id
-        )
+        cmd_action_2._id = 2
+        state.history.append(cmd_action_2)
+        cmd_observation_2 = CmdOutputObservation(content='', command='ls', command_id=2)
         cmd_observation_2._cause = cmd_action_2._id
-        event_stream.add_event(cmd_observation_2, EventSource.USER)
+        state.history.append(cmd_observation_2)
         # 6 events
 
         # random user message just because we can
         message_null_observation = NullObservation(content='')
-        event_stream.add_event(message_action, EventSource.USER)
-        event_stream.add_event(message_null_observation, EventSource.USER)
+        state.history.append(message_action)
+        state.history.append(message_null_observation)
         # 8 events
 
         assert stuck_detector.is_stuck() is False
         assert stuck_detector.state.almost_stuck == 2
 
         cmd_action_3 = CmdRunAction(command='ls')
-        event_stream.add_event(cmd_action_3, EventSource.AGENT)
-        cmd_observation_3 = CmdOutputObservation(
-            content='', command='ls', command_id=cmd_action_3._id
-        )
+        cmd_action_3._id = 3
+        state.history.append(cmd_action_3)
+        cmd_observation_3 = CmdOutputObservation(content='', command='ls', command_id=3)
         cmd_observation_3._cause = cmd_action_3._id
-        event_stream.add_event(cmd_observation_3, EventSource.USER)
+        state.history.append(cmd_observation_3)
         # 10 events
 
-        assert len(collect_events(event_stream)) == 10
-        assert len(list(stuck_detector.state.history.get_events())) == 8
+        assert len(state.history) == 10
         assert (
-            len(
-                get_pairs_from_events(
-                    stuck_detector.state.history.get_events_as_list(
-                        include_delegates=True
-                    )
-                )
-            )
-            == 5
-        )
+            len(state.history) == 10
+        )  # Adjusted since history is a list and the controller is not running
+
+        # FIXME are we still testing this without this test?
+        # assert (
+        #    len(
+        #        get_pairs_from_events(state.history)
+        #    )
+        #    == 5
+        # )
 
         assert stuck_detector.is_stuck() is False
         assert stuck_detector.state.almost_stuck == 1
 
         cmd_action_4 = CmdRunAction(command='ls')
-        event_stream.add_event(cmd_action_4, EventSource.AGENT)
-        cmd_observation_4 = CmdOutputObservation(
-            content='', command='ls', command_id=cmd_action_4._id
-        )
+        cmd_action_4._id = 4
+        state.history.append(cmd_action_4)
+        cmd_observation_4 = CmdOutputObservation(content='', command='ls', command_id=4)
         cmd_observation_4._cause = cmd_action_4._id
-        event_stream.add_event(cmd_observation_4, EventSource.USER)
+        state.history.append(cmd_observation_4)
         # 12 events
 
-        assert len(collect_events(event_stream)) == 12
-        assert len(list(stuck_detector.state.history.get_events())) == 10
-        assert (
-            len(
-                get_pairs_from_events(
-                    stuck_detector.state.history.get_events_as_list(
-                        include_delegates=True
-                    )
-                )
-            )
-            == 6
-        )
+        assert len(state.history) == 12
+        # assert (
+        #    len(
+        #        get_pairs_from_events(state.history)
+        #    )
+        #    == 6
+        # )
 
         with patch('logging.Logger.warning') as mock_warning:
             assert stuck_detector.is_stuck() is True
             assert stuck_detector.state.almost_stuck == 0
             mock_warning.assert_called_once_with('Action, Observation loop detected')
 
-    def test_is_stuck_repeating_action_error(
-        self, stuck_detector: StuckDetector, event_stream: EventStream
-    ):
+    def test_is_stuck_repeating_action_error(self, stuck_detector: StuckDetector):
+        state = stuck_detector.state
         # (action, error_observation), not necessarily the same error
         message_action = MessageAction(content='Done', wait_for_response=False)
         message_action._source = EventSource.USER
 
         hello_action = MessageAction(content='Hello', wait_for_response=False)
         hello_observation = NullObservation(content='')
-        event_stream.add_event(hello_action, EventSource.USER)
-        hello_observation._cause = hello_action._id
-        event_stream.add_event(hello_observation, EventSource.USER)
+        state.history.append(hello_action)
+        # hello_observation._cause = hello_action._id
+        state.history.append(hello_observation)
         # 2 events
 
         cmd_action_1 = CmdRunAction(command='invalid_command')
-        event_stream.add_event(cmd_action_1, EventSource.AGENT)
+        state.history.append(cmd_action_1)
         error_observation_1 = ErrorObservation(content='Command not found')
-        error_observation_1._cause = cmd_action_1._id
-        event_stream.add_event(error_observation_1, EventSource.USER)
+        # error_observation_1._cause = cmd_action_1._id
+        state.history.append(error_observation_1)
         # 4 events
 
         cmd_action_2 = CmdRunAction(command='invalid_command')
-        event_stream.add_event(cmd_action_2, EventSource.AGENT)
+        state.history.append(cmd_action_2)
         error_observation_2 = ErrorObservation(
             content='Command still not found or another error'
         )
-        error_observation_2._cause = cmd_action_2._id
-        event_stream.add_event(error_observation_2, EventSource.USER)
+        # error_observation_2._cause = cmd_action_2._id
+        state.history.append(error_observation_2)
         # 6 events
 
         message_null_observation = NullObservation(content='')
-        event_stream.add_event(message_action, EventSource.USER)
-        event_stream.add_event(message_null_observation, EventSource.USER)
+        state.history.append(message_action)
+        state.history.append(message_null_observation)
         # 8 events
 
         cmd_action_3 = CmdRunAction(command='invalid_command')
-        event_stream.add_event(cmd_action_3, EventSource.AGENT)
+        state.history.append(cmd_action_3)
         error_observation_3 = ErrorObservation(content='Different error')
-        error_observation_3._cause = cmd_action_3._id
-        event_stream.add_event(error_observation_3, EventSource.USER)
+        # error_observation_3._cause = cmd_action_3._id
+        state.history.append(error_observation_3)
         # 10 events
 
         cmd_action_4 = CmdRunAction(command='invalid_command')
-        event_stream.add_event(cmd_action_4, EventSource.AGENT)
+        state.history.append(cmd_action_4)
         error_observation_4 = ErrorObservation(content='Command not found')
-        error_observation_4._cause = cmd_action_4._id
-        event_stream.add_event(error_observation_4, EventSource.USER)
+        # error_observation_4._cause = cmd_action_4._id
+        state.history.append(error_observation_4)
         # 12 events
 
         with patch('logging.Logger.warning') as mock_warning:
@@ -267,11 +250,10 @@ def test_is_stuck_repeating_action_error(
                 'Action, ErrorObservation loop detected'
             )
 
-    def test_is_stuck_invalid_syntax_error(
-        self, stuck_detector: StuckDetector, event_stream: EventStream
-    ):
+    def test_is_stuck_invalid_syntax_error(self, stuck_detector: StuckDetector):
+        state = stuck_detector.state
         self._impl_syntax_error_events(
-            event_stream,
+            state,
             error_message='SyntaxError: invalid syntax. Perhaps you forgot a comma?',
             random_line=False,
         )
@@ -280,10 +262,11 @@ def test_is_stuck_invalid_syntax_error(
             assert stuck_detector.is_stuck() is True
 
     def test_is_not_stuck_invalid_syntax_error_random_lines(
-        self, stuck_detector: StuckDetector, event_stream: EventStream
+        self, stuck_detector: StuckDetector
     ):
+        state = stuck_detector.state
         self._impl_syntax_error_events(
-            event_stream,
+            state,
             error_message='SyntaxError: invalid syntax. Perhaps you forgot a comma?',
             random_line=True,
         )
@@ -292,10 +275,11 @@ def test_is_not_stuck_invalid_syntax_error_random_lines(
             assert stuck_detector.is_stuck() is False
 
     def test_is_not_stuck_invalid_syntax_error_only_three_incidents(
-        self, stuck_detector: StuckDetector, event_stream: EventStream
+        self, stuck_detector: StuckDetector
     ):
+        state = stuck_detector.state
         self._impl_syntax_error_events(
-            event_stream,
+            state,
             error_message='SyntaxError: invalid syntax. Perhaps you forgot a comma?',
             random_line=True,
             incidents=3,
@@ -304,11 +288,10 @@ def test_is_not_stuck_invalid_syntax_error_only_three_incidents(
         with patch('logging.Logger.warning'):
             assert stuck_detector.is_stuck() is False
 
-    def test_is_stuck_incomplete_input_error(
-        self, stuck_detector: StuckDetector, event_stream: EventStream
-    ):
+    def test_is_stuck_incomplete_input_error(self, stuck_detector: StuckDetector):
+        state = stuck_detector.state
         self._impl_syntax_error_events(
-            event_stream,
+            state,
             error_message='SyntaxError: incomplete input',
             random_line=False,
         )
@@ -316,11 +299,10 @@ def test_is_stuck_incomplete_input_error(
         with patch('logging.Logger.warning'):
             assert stuck_detector.is_stuck() is True
 
-    def test_is_not_stuck_incomplete_input_error(
-        self, stuck_detector: StuckDetector, event_stream: EventStream
-    ):
+    def test_is_not_stuck_incomplete_input_error(self, stuck_detector: StuckDetector):
+        state = stuck_detector.state
         self._impl_syntax_error_events(
-            event_stream,
+            state,
             error_message='SyntaxError: incomplete input',
             random_line=True,
         )
@@ -329,238 +311,241 @@ def test_is_not_stuck_incomplete_input_error(
             assert stuck_detector.is_stuck() is False
 
     def test_is_not_stuck_ipython_unterminated_string_error_random_lines(
-        self, stuck_detector: StuckDetector, event_stream: EventStream
+        self, stuck_detector: StuckDetector
     ):
-        self._impl_unterminated_string_error_events(event_stream, random_line=True)
+        state = stuck_detector.state
+        self._impl_unterminated_string_error_events(state, random_line=True)
 
         with patch('logging.Logger.warning'):
             assert stuck_detector.is_stuck() is False
 
     def test_is_not_stuck_ipython_unterminated_string_error_only_three_incidents(
-        self, stuck_detector: StuckDetector, event_stream: EventStream
+        self, stuck_detector: StuckDetector
     ):
+        state = stuck_detector.state
         self._impl_unterminated_string_error_events(
-            event_stream, random_line=False, incidents=3
+            state, random_line=False, incidents=3
         )
 
         with patch('logging.Logger.warning'):
             assert stuck_detector.is_stuck() is False
 
     def test_is_stuck_ipython_unterminated_string_error(
-        self, stuck_detector: StuckDetector, event_stream: EventStream
+        self, stuck_detector: StuckDetector
     ):
-        self._impl_unterminated_string_error_events(event_stream, random_line=False)
+        state = stuck_detector.state
+        self._impl_unterminated_string_error_events(state, random_line=False)
 
         with patch('logging.Logger.warning'):
             assert stuck_detector.is_stuck() is True
 
     def test_is_not_stuck_ipython_syntax_error_not_at_end(
-        self, stuck_detector: StuckDetector, event_stream: EventStream
+        self, stuck_detector: StuckDetector
     ):
+        state = stuck_detector.state
         # this test is to make sure we don't get false positives
         # since the "at line x" is changing in between!
         ipython_action_1 = IPythonRunCellAction(code='print("hello')
-        event_stream.add_event(ipython_action_1, EventSource.AGENT)
+        state.history.append(ipython_action_1)
         ipython_observation_1 = IPythonRunCellObservation(
             content='print("hello\n       ^\nSyntaxError: unterminated string literal (detected at line 1)\nThis is some additional output',
             code='print("hello',
         )
-        ipython_observation_1._cause = ipython_action_1._id
-        event_stream.add_event(ipython_observation_1, EventSource.USER)
+        # ipython_observation_1._cause = ipython_action_1._id
+        state.history.append(ipython_observation_1)
 
         ipython_action_2 = IPythonRunCellAction(code='print("hello')
-        event_stream.add_event(ipython_action_2, EventSource.AGENT)
+        state.history.append(ipython_action_2)
         ipython_observation_2 = IPythonRunCellObservation(
             content='print("hello\n       ^\nSyntaxError: unterminated string literal (detected at line 1)\nToo much output here on and on',
             code='print("hello',
         )
-        ipython_observation_2._cause = ipython_action_2._id
-        event_stream.add_event(ipython_observation_2, EventSource.USER)
+        # ipython_observation_2._cause = ipython_action_2._id
+        state.history.append(ipython_observation_2)
 
         ipython_action_3 = IPythonRunCellAction(code='print("hello')
-        event_stream.add_event(ipython_action_3, EventSource.AGENT)
+        state.history.append(ipython_action_3)
         ipython_observation_3 = IPythonRunCellObservation(
             content='print("hello\n       ^\nSyntaxError: unterminated string literal (detected at line 3)\nEnough',
             code='print("hello',
         )
-        ipython_observation_3._cause = ipython_action_3._id
-        event_stream.add_event(ipython_observation_3, EventSource.USER)
+        # ipython_observation_3._cause = ipython_action_3._id
+        state.history.append(ipython_observation_3)
 
         ipython_action_4 = IPythonRunCellAction(code='print("hello')
-        event_stream.add_event(ipython_action_4, EventSource.AGENT)
+        state.history.append(ipython_action_4)
         ipython_observation_4 = IPythonRunCellObservation(
             content='print("hello\n       ^\nSyntaxError: unterminated string literal (detected at line 2)\nLast line of output',
             code='print("hello',
         )
-        ipython_observation_4._cause = ipython_action_4._id
-        event_stream.add_event(ipython_observation_4, EventSource.USER)
+        # ipython_observation_4._cause = ipython_action_4._id
+        state.history.append(ipython_observation_4)
 
         with patch('logging.Logger.warning') as mock_warning:
             assert stuck_detector.is_stuck() is False
             mock_warning.assert_not_called()
 
     def test_is_stuck_repeating_action_observation_pattern(
-        self, stuck_detector: StuckDetector, event_stream: EventStream
+        self, stuck_detector: StuckDetector
     ):
+        state = stuck_detector.state
         message_action = MessageAction(content='Come on', wait_for_response=False)
         message_action._source = EventSource.USER
-        event_stream.add_event(message_action, EventSource.USER)
+        state.history.append(message_action)
         message_observation = NullObservation(content='')
-        event_stream.add_event(message_observation, EventSource.USER)
+        state.history.append(message_observation)
 
         cmd_action_1 = CmdRunAction(command='ls')
-        event_stream.add_event(cmd_action_1, EventSource.AGENT)
+        state.history.append(cmd_action_1)
         cmd_observation_1 = CmdOutputObservation(
             command_id=1, command='ls', content='file1.txt\nfile2.txt'
         )
-        cmd_observation_1._cause = cmd_action_1._id
-        event_stream.add_event(cmd_observation_1, EventSource.USER)
+        # cmd_observation_1._cause = cmd_action_1._id
+        state.history.append(cmd_observation_1)
 
         read_action_1 = FileReadAction(path='file1.txt')
-        event_stream.add_event(read_action_1, EventSource.AGENT)
+        state.history.append(read_action_1)
         read_observation_1 = FileReadObservation(
             content='File content', path='file1.txt'
         )
-        read_observation_1._cause = read_action_1._id
-        event_stream.add_event(read_observation_1, EventSource.USER)
+        # read_observation_1._cause = read_action_1._id
+        state.history.append(read_observation_1)
 
         cmd_action_2 = CmdRunAction(command='ls')
-        event_stream.add_event(cmd_action_2, EventSource.AGENT)
+        state.history.append(cmd_action_2)
         cmd_observation_2 = CmdOutputObservation(
             command_id=2, command='ls', content='file1.txt\nfile2.txt'
         )
-        cmd_observation_2._cause = cmd_action_2._id
-        event_stream.add_event(cmd_observation_2, EventSource.USER)
+        # cmd_observation_2._cause = cmd_action_2._id
+        state.history.append(cmd_observation_2)
 
         read_action_2 = FileReadAction(path='file1.txt')
-        event_stream.add_event(read_action_2, EventSource.AGENT)
+        state.history.append(read_action_2)
         read_observation_2 = FileReadObservation(
             content='File content', path='file1.txt'
         )
-        read_observation_2._cause = read_action_2._id
-        event_stream.add_event(read_observation_2, EventSource.USER)
+        # read_observation_2._cause = read_action_2._id
+        state.history.append(read_observation_2)
+
+        message_action = MessageAction(content='Come on', wait_for_response=False)
+        message_action._source = EventSource.USER
+        state.history.append(message_action)
 
-        # one more message to break the pattern
         message_null_observation = NullObservation(content='')
-        event_stream.add_event(message_action, EventSource.USER)
-        event_stream.add_event(message_null_observation, EventSource.USER)
+        state.history.append(message_null_observation)
 
         cmd_action_3 = CmdRunAction(command='ls')
-        event_stream.add_event(cmd_action_3, EventSource.AGENT)
+        state.history.append(cmd_action_3)
         cmd_observation_3 = CmdOutputObservation(
             command_id=3, command='ls', content='file1.txt\nfile2.txt'
         )
-        cmd_observation_3._cause = cmd_action_3._id
-        event_stream.add_event(cmd_observation_3, EventSource.USER)
+        # cmd_observation_3._cause = cmd_action_3._id
+        state.history.append(cmd_observation_3)
 
         read_action_3 = FileReadAction(path='file1.txt')
-        event_stream.add_event(read_action_3, EventSource.AGENT)
+        state.history.append(read_action_3)
         read_observation_3 = FileReadObservation(
             content='File content', path='file1.txt'
         )
-        read_observation_3._cause = read_action_3._id
-        event_stream.add_event(read_observation_3, EventSource.USER)
+        # read_observation_3._cause = read_action_3._id
+        state.history.append(read_observation_3)
 
         with patch('logging.Logger.warning') as mock_warning:
             assert stuck_detector.is_stuck() is True
             mock_warning.assert_called_once_with('Action, Observation pattern detected')
 
-    def test_is_stuck_not_stuck(
-        self, stuck_detector: StuckDetector, event_stream: EventStream
-    ):
+    def test_is_stuck_not_stuck(self, stuck_detector: StuckDetector):
+        state = stuck_detector.state
         message_action = MessageAction(content='Done', wait_for_response=False)
         message_action._source = EventSource.USER
 
         hello_action = MessageAction(content='Hello', wait_for_response=False)
-        event_stream.add_event(hello_action, EventSource.USER)
+        state.history.append(hello_action)
         hello_observation = NullObservation(content='')
-        hello_observation._cause = hello_action._id
-        event_stream.add_event(hello_observation, EventSource.USER)
+        # hello_observation._cause = hello_action._id
+        state.history.append(hello_observation)
 
         cmd_action_1 = CmdRunAction(command='ls')
-        event_stream.add_event(cmd_action_1, EventSource.AGENT)
+        state.history.append(cmd_action_1)
         cmd_observation_1 = CmdOutputObservation(
             command_id=cmd_action_1.id, command='ls', content='file1.txt\nfile2.txt'
         )
-        cmd_observation_1._cause = cmd_action_1._id
-        event_stream.add_event(cmd_observation_1, EventSource.USER)
+        # cmd_observation_1._cause = cmd_action_1._id
+        state.history.append(cmd_observation_1)
 
         read_action_1 = FileReadAction(path='file1.txt')
-        event_stream.add_event(read_action_1, EventSource.AGENT)
+        state.history.append(read_action_1)
         read_observation_1 = FileReadObservation(
             content='File content', path='file1.txt'
         )
-        read_observation_1._cause = read_action_1._id
-        event_stream.add_event(read_observation_1, EventSource.USER)
+        # read_observation_1._cause = read_action_1._id
+        state.history.append(read_observation_1)
 
         cmd_action_2 = CmdRunAction(command='pwd')
-        event_stream.add_event(cmd_action_2, EventSource.AGENT)
+        state.history.append(cmd_action_2)
         cmd_observation_2 = CmdOutputObservation(
             command_id=2, command='pwd', content='/home/user'
         )
-        cmd_observation_2._cause = cmd_action_2._id
-        event_stream.add_event(cmd_observation_2, EventSource.USER)
+        # cmd_observation_2._cause = cmd_action_2._id
+        state.history.append(cmd_observation_2)
 
         read_action_2 = FileReadAction(path='file2.txt')
-        event_stream.add_event(read_action_2, EventSource.AGENT)
+        state.history.append(read_action_2)
         read_observation_2 = FileReadObservation(
             content='Another file content', path='file2.txt'
         )
-        read_observation_2._cause = read_action_2._id
-        event_stream.add_event(read_observation_2, EventSource.USER)
+        # read_observation_2._cause = read_action_2._id
+        state.history.append(read_observation_2)
 
         message_null_observation = NullObservation(content='')
-        event_stream.add_event(message_action, EventSource.USER)
-        event_stream.add_event(message_null_observation, EventSource.USER)
+        state.history.append(message_action)
+        state.history.append(message_null_observation)
 
         cmd_action_3 = CmdRunAction(command='pwd')
-        event_stream.add_event(cmd_action_3, EventSource.AGENT)
+        state.history.append(cmd_action_3)
         cmd_observation_3 = CmdOutputObservation(
             command_id=cmd_action_3.id, command='pwd', content='/home/user'
         )
-        cmd_observation_3._cause = cmd_action_3._id
-        event_stream.add_event(cmd_observation_3, EventSource.USER)
+        # cmd_observation_3._cause = cmd_action_3._id
+        state.history.append(cmd_observation_3)
 
         read_action_3 = FileReadAction(path='file2.txt')
-        event_stream.add_event(read_action_3, EventSource.AGENT)
+        state.history.append(read_action_3)
         read_observation_3 = FileReadObservation(
             content='Another file content', path='file2.txt'
         )
-        read_observation_3._cause = read_action_3._id
-        event_stream.add_event(read_observation_3, EventSource.USER)
+        # read_observation_3._cause = read_action_3._id
+        state.history.append(read_observation_3)
 
         assert stuck_detector.is_stuck() is False
 
-    def test_is_stuck_monologue(self, stuck_detector, event_stream):
-        # Add events to the event stream
+    def test_is_stuck_monologue(self, stuck_detector):
+        state = stuck_detector.state
+        # Add events to the history list directly
         message_action_1 = MessageAction(content='Hi there!')
-        event_stream.add_event(message_action_1, EventSource.USER)
         message_action_1._source = EventSource.USER
-
+        state.history.append(message_action_1)
         message_action_2 = MessageAction(content='Hi there!')
-        event_stream.add_event(message_action_2, EventSource.AGENT)
         message_action_2._source = EventSource.AGENT
-
+        state.history.append(message_action_2)
         message_action_3 = MessageAction(content='How are you?')
-        event_stream.add_event(message_action_3, EventSource.USER)
         message_action_3._source = EventSource.USER
+        state.history.append(message_action_3)
 
         cmd_kill_action = CmdRunAction(
             command='echo 42', thought="I'm not stuck, he's stuck"
         )
-        event_stream.add_event(cmd_kill_action, EventSource.AGENT)
+        state.history.append(cmd_kill_action)
 
         message_action_4 = MessageAction(content="I'm doing well, thanks for asking.")
-        event_stream.add_event(message_action_4, EventSource.AGENT)
         message_action_4._source = EventSource.AGENT
-
+        state.history.append(message_action_4)
         message_action_5 = MessageAction(content="I'm doing well, thanks for asking.")
-        event_stream.add_event(message_action_5, EventSource.AGENT)
         message_action_5._source = EventSource.AGENT
-
+        state.history.append(message_action_5)
         message_action_6 = MessageAction(content="I'm doing well, thanks for asking.")
-        event_stream.add_event(message_action_6, EventSource.AGENT)
         message_action_6._source = EventSource.AGENT
+        state.history.append(message_action_6)
 
         assert stuck_detector.is_stuck()
 
@@ -571,16 +556,15 @@ def test_is_stuck_monologue(self, stuck_detector, event_stream):
             command='storybook',
             exit_code=0,
         )
-        cmd_output_observation._cause = cmd_kill_action._id
-        event_stream.add_event(cmd_output_observation, EventSource.USER)
+        # cmd_output_observation._cause = cmd_kill_action._id
+        state.history.append(cmd_output_observation)
 
         message_action_7 = MessageAction(content="I'm doing well, thanks for asking.")
-        event_stream.add_event(message_action_7, EventSource.AGENT)
         message_action_7._source = EventSource.AGENT
-
+        state.history.append(message_action_7)
         message_action_8 = MessageAction(content="I'm doing well, thanks for asking.")
-        event_stream.add_event(message_action_8, EventSource.AGENT)
         message_action_8._source = EventSource.AGENT
+        state.history.append(message_action_8)
 
         with patch('logging.Logger.warning'):
             assert not stuck_detector.is_stuck()
@@ -595,7 +579,6 @@ def controller(self):
         )
         controller.delegate = None
         controller.state = Mock()
-        controller.state.history = ShortTermHistory()
         return controller
 
     def test_is_stuck_delegate_stuck(self, controller: AgentController):
diff --git a/tests/unit/test_llm.py b/tests/unit/test_llm.py
index 347d383076f4..073743ea81e5 100644
--- a/tests/unit/test_llm.py
+++ b/tests/unit/test_llm.py
@@ -50,6 +50,7 @@ def test_llm_init_with_model_info(mock_get_model_info, default_config):
         'max_output_tokens': 2000,
     }
     llm = LLM(default_config)
+    llm.init_model_info()
     assert llm.config.max_input_tokens == 8000
     assert llm.config.max_output_tokens == 2000
 
@@ -58,6 +59,7 @@ def test_llm_init_with_model_info(mock_get_model_info, default_config):
 def test_llm_init_without_model_info(mock_get_model_info, default_config):
     mock_get_model_info.side_effect = Exception('Model info not available')
     llm = LLM(default_config)
+    llm.init_model_info()
     assert llm.config.max_input_tokens == 4096
     assert llm.config.max_output_tokens == 4096
 
@@ -108,6 +110,7 @@ def test_llm_init_with_openrouter_model(mock_get_model_info, default_config):
         'max_output_tokens': 1500,
     }
     llm = LLM(default_config)
+    llm.init_model_info()
     assert llm.config.max_input_tokens == 7000
     assert llm.config.max_output_tokens == 1500
     mock_get_model_info.assert_called_once_with('openrouter:gpt-4o-mini')
diff --git a/tests/unit/test_memory.py b/tests/unit/test_memory.py
index 10991ca27dbf..96c06e0fd41c 100644
--- a/tests/unit/test_memory.py
+++ b/tests/unit/test_memory.py
@@ -88,7 +88,7 @@ def _create_observation_event(observation: str) -> Event:
     event = Event()
     event._id = -1
     event._timestamp = datetime.now(timezone.utc).isoformat()
-    event._source = EventSource.USER
+    event._source = EventSource.ENVIRONMENT
     event.observation = observation
     return event
 
diff --git a/tests/unit/test_micro_agents.py b/tests/unit/test_micro_agents.py
index 70553d851125..8cff14fdd4f2 100644
--- a/tests/unit/test_micro_agents.py
+++ b/tests/unit/test_micro_agents.py
@@ -10,10 +10,8 @@
 from openhands.controller.agent import Agent
 from openhands.controller.state.state import State
 from openhands.core.config import AgentConfig
-from openhands.events import EventSource
 from openhands.events.action import MessageAction
 from openhands.events.stream import EventStream
-from openhands.memory.history import ShortTermHistory
 from openhands.storage import get_file_store
 
 
@@ -74,10 +72,10 @@ def test_coder_agent_with_summary(event_stream: EventStream, agent_configs: dict
     )
     assert coder_agent is not None
 
+    # give it some history
     task = 'This is a dummy task'
-    history = ShortTermHistory()
-    history.set_event_stream(event_stream)
-    event_stream.add_event(MessageAction(content=task), EventSource.USER)
+    history = list()
+    history.append(MessageAction(content=task))
 
     summary = 'This is a dummy summary about this repo'
     state = State(history=history, inputs={'summary': summary})
@@ -119,10 +117,10 @@ def test_coder_agent_without_summary(event_stream: EventStream, agent_configs: d
     )
     assert coder_agent is not None
 
+    # give it some history
     task = 'This is a dummy task'
-    history = ShortTermHistory()
-    history.set_event_stream(event_stream)
-    event_stream.add_event(MessageAction(content=task), EventSource.USER)
+    history = list()
+    history.append(MessageAction(content=task))
 
     # set state without codebase summary
     state = State(history=history)
diff --git a/tests/unit/test_prompt_caching.py b/tests/unit/test_prompt_caching.py
index 50c42bf66290..caa08b0e55fd 100644
--- a/tests/unit/test_prompt_caching.py
+++ b/tests/unit/test_prompt_caching.py
@@ -1,14 +1,12 @@
-from unittest.mock import MagicMock, Mock, patch
+from unittest.mock import Mock, patch
 
 import pytest
 
 from openhands.agenthub.codeact_agent.codeact_agent import CodeActAgent
 from openhands.core.config import AgentConfig, LLMConfig
-from openhands.events import EventSource, EventStream
 from openhands.events.action import CmdRunAction, MessageAction
 from openhands.events.observation import CmdOutputObservation
 from openhands.llm.llm import LLM
-from openhands.storage import get_file_store
 
 
 @pytest.fixture
@@ -19,12 +17,6 @@ def mock_llm():
     return llm
 
 
-@pytest.fixture
-def mock_event_stream(tmp_path):
-    file_store = get_file_store('local', str(tmp_path))
-    return EventStream('test_session', file_store)
-
-
 @pytest.fixture(params=[False, True])
 def codeact_agent(mock_llm, request):
     config = AgentConfig()
@@ -57,17 +49,28 @@ def model_dump(self):
     return MockModelResponse(content)
 
 
-def test_get_messages_with_reminder(codeact_agent, mock_event_stream):
-    # Add some events to the stream
-    mock_event_stream.add_event(MessageAction('Initial user message'), EventSource.USER)
-    mock_event_stream.add_event(MessageAction('Sure!'), EventSource.AGENT)
-    mock_event_stream.add_event(MessageAction('Hello, agent!'), EventSource.USER)
-    mock_event_stream.add_event(MessageAction('Hello, user!'), EventSource.AGENT)
-    mock_event_stream.add_event(MessageAction('Laaaaaaaast!'), EventSource.USER)
+def test_get_messages_with_reminder(codeact_agent: CodeActAgent):
+    # Add some events to history
+    history = list()
+    message_action_1 = MessageAction('Initial user message')
+    message_action_1._source = 'user'
+    history.append(message_action_1)
+    message_action_2 = MessageAction('Sure!')
+    message_action_2._source = 'assistant'
+    history.append(message_action_2)
+    message_action_3 = MessageAction('Hello, agent!')
+    message_action_3._source = 'user'
+    history.append(message_action_3)
+    message_action_4 = MessageAction('Hello, user!')
+    message_action_4._source = 'assistant'
+    history.append(message_action_4)
+    message_action_5 = MessageAction('Laaaaaaaast!')
+    message_action_5._source = 'user'
+    history.append(message_action_5)
 
     codeact_agent.reset()
     messages = codeact_agent._get_messages(
-        Mock(history=mock_event_stream, max_iterations=5, iteration=0)
+        Mock(history=history, max_iterations=5, iteration=0)
     )
 
     assert (
@@ -102,19 +105,20 @@ def test_get_messages_with_reminder(codeact_agent, mock_event_stream):
         )
 
 
-def test_get_messages_prompt_caching(codeact_agent, mock_event_stream):
+def test_get_messages_prompt_caching(codeact_agent: CodeActAgent):
+    history = list()
     # Add multiple user and agent messages
     for i in range(15):
-        mock_event_stream.add_event(
-            MessageAction(f'User message {i}'), EventSource.USER
-        )
-        mock_event_stream.add_event(
-            MessageAction(f'Agent message {i}'), EventSource.AGENT
-        )
+        message_action_user = MessageAction(f'User message {i}')
+        message_action_user._source = 'user'
+        history.append(message_action_user)
+        message_action_agent = MessageAction(f'Agent message {i}')
+        message_action_agent._source = 'assistant'
+        history.append(message_action_agent)
 
     codeact_agent.reset()
     messages = codeact_agent._get_messages(
-        Mock(history=mock_event_stream, max_iterations=10, iteration=5)
+        Mock(history=history, max_iterations=10, iteration=5)
     )
 
     # Check that only the last two user messages have cache_prompt=True
@@ -136,18 +140,23 @@ def test_get_messages_prompt_caching(codeact_agent, mock_event_stream):
     assert cached_user_messages[3].content[0].text.startswith('User message 1')
 
 
-def test_get_messages_with_cmd_action(codeact_agent, mock_event_stream):
+def test_get_messages_with_cmd_action(codeact_agent: CodeActAgent):
     if codeact_agent.config.function_calling:
         pytest.skip('Skipping this test for function calling')
 
+    history = list()
+
     # Add a mix of actions and observations
     message_action_1 = MessageAction(
         "Let's list the contents of the current directory."
     )
-    mock_event_stream.add_event(message_action_1, EventSource.USER)
+    message_action_1._source = 'user'
+    history.append(message_action_1)
 
     cmd_action_1 = CmdRunAction('ls -l', thought='List files in current directory')
-    mock_event_stream.add_event(cmd_action_1, EventSource.AGENT)
+    cmd_action_1._source = 'agent'
+    cmd_action_1._id = 'cmd_1'
+    history.append(cmd_action_1)
 
     cmd_observation_1 = CmdOutputObservation(
         content='total 0\n-rw-r--r-- 1 user group 0 Jan 1 00:00 file1.txt\n-rw-r--r-- 1 user group 0 Jan 1 00:00 file2.txt',
@@ -155,13 +164,17 @@ def test_get_messages_with_cmd_action(codeact_agent, mock_event_stream):
         command='ls -l',
         exit_code=0,
     )
-    mock_event_stream.add_event(cmd_observation_1, EventSource.USER)
+    cmd_observation_1._source = 'user'
+    history.append(cmd_observation_1)
 
     message_action_2 = MessageAction("Now, let's create a new directory.")
-    mock_event_stream.add_event(message_action_2, EventSource.AGENT)
+    message_action_2._source = 'agent'
+    history.append(message_action_2)
 
     cmd_action_2 = CmdRunAction('mkdir new_directory', thought='Create a new directory')
-    mock_event_stream.add_event(cmd_action_2, EventSource.AGENT)
+    cmd_action_2._source = 'agent'
+    cmd_action_2._id = 'cmd_2'
+    history.append(cmd_action_2)
 
     cmd_observation_2 = CmdOutputObservation(
         content='',
@@ -169,11 +182,12 @@ def test_get_messages_with_cmd_action(codeact_agent, mock_event_stream):
         command='mkdir new_directory',
         exit_code=0,
     )
-    mock_event_stream.add_event(cmd_observation_2, EventSource.USER)
+    cmd_observation_2._source = 'user'
+    history.append(cmd_observation_2)
 
     codeact_agent.reset()
     messages = codeact_agent._get_messages(
-        Mock(history=mock_event_stream, max_iterations=5, iteration=0)
+        Mock(history=history, max_iterations=5, iteration=0)
     )
 
     # Assert the presence of key elements in the messages
@@ -218,19 +232,17 @@ def test_get_messages_with_cmd_action(codeact_agent, mock_event_stream):
         assert 'ENVIRONMENT REMINDER: You have 5 turns' in messages[5].content[1].text
 
 
-def test_prompt_caching_headers(codeact_agent, mock_event_stream):
+def test_prompt_caching_headers(codeact_agent: CodeActAgent):
+    history = list()
     if codeact_agent.config.function_calling:
         pytest.skip('Skipping this test for function calling')
 
     # Setup
-    mock_event_stream.add_event(MessageAction('Hello, agent!'), EventSource.USER)
-    mock_event_stream.add_event(MessageAction('Hello, user!'), EventSource.AGENT)
-
-    mock_short_term_history = MagicMock()
-    mock_short_term_history.get_last_user_message.return_value = 'Hello, agent!'
+    history.append(MessageAction('Hello, agent!'))
+    history.append(MessageAction('Hello, user!'))
 
     mock_state = Mock()
-    mock_state.history = mock_short_term_history
+    mock_state.history = history
     mock_state.max_iterations = 5
     mock_state.iteration = 0
 

From a9e346a9cbbfa3ed6595b6d42abc8c779006bc02 Mon Sep 17 00:00:00 2001
From: AlexCuadron <alex.cl.2000@gmail.com>
Date: Sun, 10 Nov 2024 22:37:57 -0800
Subject: [PATCH 15/18] first try

---
 .../agenthub/codeact_agent/codeact_agent.py   | 111 +++-
 .../codeact_agent/function_calling.py         |  11 +-
 openhands/agenthub/supervisor_agent/agent.py  | 198 ++++----
 openhands/agenthub/supervisor_agent/prompt.py | 478 +++++++-----------
 4 files changed, 370 insertions(+), 428 deletions(-)

diff --git a/openhands/agenthub/codeact_agent/codeact_agent.py b/openhands/agenthub/codeact_agent/codeact_agent.py
index 91d04a75ef6a..2926d7d5e459 100644
--- a/openhands/agenthub/codeact_agent/codeact_agent.py
+++ b/openhands/agenthub/codeact_agent/codeact_agent.py
@@ -1,4 +1,3 @@
-import json
 import os
 from collections import deque
 from itertools import islice
@@ -13,6 +12,7 @@
 from openhands.core.config.llm_config import LLMConfig
 from openhands.core.logger import openhands_logger as logger
 from openhands.core.message import ImageContent, Message, TextContent
+from openhands.core.utils import json
 from openhands.events.action import (
     Action,
     AgentDelegateAction,
@@ -73,6 +73,8 @@ class CodeActAgent(Agent):
         JupyterRequirement(),
     ]
     obs_prefix = 'OBSERVATION:\n'
+    when_to_stop = 6
+    number_of_events = -1
 
     def __init__(
         self,
@@ -85,6 +87,7 @@ def __init__(
         - llm (LLM): The llm to be used by this agent
         """
 
+        # import pdb; pdb.set_trace()
         llm_config = LLMConfig(
             model='litellm_proxy/claude-3-5-sonnet-20241022',
             api_key='REDACTED',
@@ -93,10 +96,9 @@ def __init__(
         )
         llm = LLM(llm_config)
         # TODO: Remove this once we have a real AgentConfig
-        config = AgentConfig(llm_config='o1-mini')
+        config = AgentConfig()
         super().__init__(llm, config)
         self.reset()
-
         self.micro_agent = (
             MicroAgent(
                 os.path.join(
@@ -343,6 +345,11 @@ def step(self, state: State) -> Action:
         - MessageAction(content) - Message action to run (e.g. ask for clarification)
         - AgentFinishAction() - end the interaction
         """
+
+        # If this agent has a supervisor, we need to get the time to stop from the supervisor
+        if self.when_to_stop < 0 and state.inputs.get('when_to_stop', None):
+            self.when_to_stop = state.inputs['when_to_stop']
+
         # Continue with pending actions if any
         if self.pending_actions:
             return self.pending_actions.popleft()
@@ -350,7 +357,21 @@ def step(self, state: State) -> Action:
         # if we're done, go back
         last_user_message = state.get_last_user_message()
         if last_user_message and last_user_message.strip() == '/exit':
-            return AgentFinishAction()
+            messages = self._get_messages(state)
+            serialized_messages = [msg.model_dump() for msg in messages]
+            return AgentFinishAction(
+                outputs={'fixed': True, 'trayectory': serialized_messages}
+            )
+
+        # if we've reached the max number of iterations, go back for an evaluation on the approach
+        if self.when_to_stop > 0 and state.local_iteration % self.when_to_stop == 0:
+            messages = self._get_messages(state)
+            serialized_messages = [
+                msg.model_dump() for msg in messages
+            ]  # Serialize each Message object
+            return AgentFinishAction(
+                outputs={'trayectory': serialized_messages, 'fixed': False}
+            )
 
         # prepare what we want to send to the LLM
         messages = self._get_messages(state)
@@ -409,17 +430,60 @@ def _get_messages(self, state: State) -> list[Message]:
             - Messages from the same role are combined to prevent consecutive same-role messages
             - For Anthropic models, specific messages are cached according to their documentation
         """
-        messages: list[Message] = [
-            Message(
-                role='system',
-                content=[
-                    TextContent(
-                        text=self.system_prompt,
-                        cache_prompt=self.llm.is_caching_prompt_active(),  # Cache system prompt
-                    )
-                ],
+        # import pdb; pdb.set_trace()
+        messages: list[Message] = []
+        trayectory = state.inputs.get('trayectory', '')
+        # If there is no trayectory, its the first time we are seeing the task
+        if not trayectory:
+            messages.append(
+                Message(
+                    role='system',
+                    content=[
+                        TextContent(
+                            text=self.system_prompt,
+                            cache_prompt=self.llm.is_caching_prompt_active(),  # Cache system prompt
+                        )
+                    ],
+                )
             )
-        ]
+            if state.inputs.get('task', '') != '':
+                # During AgentDelegation the history is empty, so we add the task as the user message.
+                messages.append(
+                    Message(
+                        role='user',
+                        content=[TextContent(text=state.inputs['task'])],
+                    )
+                )
+
+            if state.inputs.get('augmented_task', ''):
+                messages.append(
+                    Message(
+                        role='user',
+                        content=[TextContent(text=state.inputs['augmented_task'])],
+                    )
+                )
+        else:
+            # If there is a previous trayectory, we restore it.
+            deserialized_trajectory = [
+                Message(
+                    role='user',
+                    content=[
+                        TextContent(text=content_text)
+                        for content_text in [
+                            msg_dict['content'][0]['text']
+                            if isinstance(msg_dict['content'], list)
+                            else msg_dict['content']
+                        ]
+                        if content_text  # Skip empty content
+                    ],
+                    tool_call_id=msg_dict.get('tool_call_id'),
+                    name=msg_dict.get('name'),
+                )
+                for msg_dict in trayectory
+                if msg_dict.get('content')  # Skip messages with no content
+            ]
+            messages.extend(deserialized_trajectory)
+
         if self.initial_user_message:
             messages.append(
                 Message(
@@ -431,7 +495,9 @@ def _get_messages(self, state: State) -> list[Message]:
         pending_tool_call_action_messages: dict[str, Message] = {}
         tool_call_id_to_message: dict[str, Message] = {}
         events = list(state.history)
-        for event in events:
+        if self.number_of_events < 0:
+            self.number_of_events = len(events)
+        for i, event in enumerate(events):
             # create a regular message from an event
             if isinstance(event, Action):
                 messages_to_add = self.get_action_message(
@@ -446,6 +512,14 @@ def _get_messages(self, state: State) -> list[Message]:
             else:
                 raise ValueError(f'Unknown event type: {type(event)}')
 
+            if i == self.number_of_events and state.inputs.get('next_step', ''):
+                messages_to_add = [
+                    Message(
+                        role='user',
+                        content=[TextContent(text=state.inputs['next_step'])],
+                    )
+                ]
+
             # Check pending tool call action messages and see if they are complete
             _response_ids_to_remove = []
             for (
@@ -488,6 +562,13 @@ def _get_messages(self, state: State) -> list[Message]:
                     else:
                         messages.append(message)
 
+        if self.number_of_events == len(events) and state.inputs.get('next_step', ''):
+            messages.append(
+                Message(
+                    role='user', content=[TextContent(text=state.inputs['next_step'])]
+                )
+            )
+
         if self.llm.is_caching_prompt_active():
             # NOTE: this is only needed for anthropic
             # following logic here:
diff --git a/openhands/agenthub/codeact_agent/function_calling.py b/openhands/agenthub/codeact_agent/function_calling.py
index 1799478601bd..3a888f2e11b1 100644
--- a/openhands/agenthub/codeact_agent/function_calling.py
+++ b/openhands/agenthub/codeact_agent/function_calling.py
@@ -13,6 +13,7 @@
 )
 
 from openhands.core.logger import openhands_logger as logger
+from openhands.core.message import Message
 from openhands.events.action import (
     Action,
     AgentDelegateAction,
@@ -448,7 +449,11 @@ def combine_thought(action: Action, thought: str) -> Action:
     return action
 
 
-def response_to_actions(response: ModelResponse) -> list[Action]:
+def response_to_actions(
+    response: ModelResponse, messages: list[Message] | None = None
+) -> list[Action]:
+    if messages is None:
+        messages = []
     actions: list[Action] = []
     assert len(response.choices) == 1, 'Only one choice is supported for now'
     assistant_msg = response.choices[0].message
@@ -481,7 +486,9 @@ def response_to_actions(response: ModelResponse) -> list[Action]:
                     inputs=arguments,
                 )
             elif tool_call.function.name == 'finish':
-                action = AgentFinishAction()
+                action = AgentFinishAction(
+                    outputs={'fixed': True, 'trayectory': messages}
+                )
             elif tool_call.function.name == 'edit_file':
                 action = FileEditAction(**arguments)
             elif tool_call.function.name == 'str_replace_editor':
diff --git a/openhands/agenthub/supervisor_agent/agent.py b/openhands/agenthub/supervisor_agent/agent.py
index 722d7365cb3a..96e04348581f 100644
--- a/openhands/agenthub/supervisor_agent/agent.py
+++ b/openhands/agenthub/supervisor_agent/agent.py
@@ -1,8 +1,8 @@
 import logging
-from typing import Any, Dict, List, Literal, Union
+import re
+from typing import Any, Dict, List
 
 from openhands.agenthub.supervisor_agent.prompt import (
-    TASK_TYPE_ISSUE,
     get_prompt,
 )
 from openhands.controller.agent import Agent
@@ -10,11 +10,12 @@
 from openhands.core.config import AgentConfig
 from openhands.core.config.llm_config import LLMConfig
 from openhands.core.message import Message, TextContent
-from openhands.core.utils import json
 from openhands.events.action import Action, AgentDelegateAction, AgentFinishAction
-from openhands.events.action.agent import AgentRejectAction
 from openhands.events.observation.delegate import AgentDelegateObservation
 from openhands.llm.llm import LLM
+from openhands.runtime.plugins.agent_skills import AgentSkillsRequirement
+from openhands.runtime.plugins.jupyter import JupyterRequirement
+from openhands.runtime.plugins.requirement import PluginRequirement
 
 
 class SupervisorAgent(Agent):
@@ -32,7 +33,21 @@ class SupervisorAgent(Agent):
     does_it_needs_a_test: bool = False
     task: str = ''
     test_command: str = ''
-    phase: Literal['search', 'summary', 'code'] = 'search'
+    time_to_stop: int = 60  # Every 60 iterations, we stop and evaluate the approach
+
+    sandbox_plugins: list[PluginRequirement] = [
+        # NOTE: AgentSkillsRequirement need to go before JupyterRequirement, since
+        # AgentSkillsRequirement provides a lot of Python functions,
+        # and it needs to be initialized before Jupyter for Jupyter to use those functions.
+        AgentSkillsRequirement(),
+        JupyterRequirement(),
+    ]
+
+    # Add class attribute for tried_direct_code
+    tried_direct_code: bool = False
+
+    # Add class attribute for augmented_task
+    augmented_task: str = ''
 
     def __init__(self, llm: LLM, config: AgentConfig):
         """Initialize the Supervisor Agent with an LLM
@@ -55,122 +70,85 @@ def __init__(self, llm: LLM, config: AgentConfig):
     def step(self, state: State) -> Action:
         self.logger.debug('Starting step with state: %s', state)
         self.logger.debug('LLM config: %s', self.llm_config)
-
-        if len(self.suggested_approaches) == 0:
-            self.suggested_approaches = self.get_suggested_approaches(state)
-        self.suggested_approach_index += 1
-
-        last_observation = state.history[-1] if state.history else None
-        if isinstance(last_observation, AgentDelegateObservation):
-            self.results[self.phase].append(last_observation.outputs.get('output', ''))
-
-        if self.suggested_approach_index < len(self.suggested_approaches):
-            # Delegate to the SearcherAgent as we need to gather more information
-            return self.delegate_to_agent(
-                'SearcherAgent',
-                self.task,
-                self.suggested_approaches[self.suggested_approach_index].get(
-                    'suggested_approach', []
-                ),
+        last_observation = state.history[-1]
+        task, _ = state.get_current_user_intent()
+        self.task = task or ''
+
+        # import pdb; pdb.set_trace()
+        # Try CodeActAgent first if we haven't tried it yet
+        if not self.tried_direct_code:
+            prompt = get_prompt(self.task, [], 'initial')
+            raw_response = self.get_response(prompt)
+            match = re.search(
+                r'<augmented_pr_description>(.*?)</augmented_pr_description>',
+                raw_response,
+                re.DOTALL,
             )
-
-        if self.phase == 'search':
-            condensed_information = self.ask_llm(
-                self.task, 'summary', self.results[self.phase]
+            self.augmented_task = match.group(1).strip('"') if match else self.task
+            self.tried_direct_code = True
+            return AgentDelegateAction(
+                agent='CodeActAgent',
+                inputs={
+                    'task': self.task,
+                    'augmented_task': self.augmented_task,
+                    'when_to_stop': self.time_to_stop,
+                },
             )
-            if condensed_information and len(condensed_information) > 0:
-                first_result = condensed_information[0]
-                if first_result.get('summary', '') != '':
-                    self.phase = 'summary'
-                    self.condensed_information = first_result.get('summary', '')
-                else:
-                    suggested_approach: str | list[str] = first_result.get(
-                        'suggested_approach', []
-                    )
-                    self.results['search'].append(suggested_approach)
-                    return self.delegate_to_agent(
-                        'SearcherAgent', self.task, suggested_approach
-                    )
 
-        if self.phase == 'summary':
-            if not self.does_it_needs_a_test:
-                test_check = self.ask_llm(self.task, 'code', self.condensed_information)
-                first_check = (
-                    test_check[0] if test_check and len(test_check) > 0 else {}
-                )
-                self.does_it_needs_a_test = (
-                    first_check.get('suggested_approach', '') == TASK_TYPE_ISSUE
+        if not isinstance(last_observation, AgentDelegateObservation):
+            raise ValueError('Last observation is not an AgentDelegateObservation')
+
+        if not last_observation.outputs.get('fixed', False):
+            trayectory: List[Dict] = last_observation.outputs['trayectory']
+            deserialized_trajectory = [
+                Message(
+                    role=msg_dict['role'],
+                    content=[
+                        TextContent(text=content_text)
+                        for content_text in [
+                            msg_dict['content'][0]['text']
+                            if isinstance(msg_dict['content'], list)
+                            else msg_dict['content']
+                        ]
+                    ],
+                    tool_call_id=msg_dict.get('tool_call_id'),
+                    name=msg_dict.get('name'),
                 )
-                self.phase = 'code'
-                if self.does_it_needs_a_test:
-                    self.current_delegate = 'TesterAgent'
-                    return AgentDelegateAction(
-                        agent='TesterAgent',
-                        inputs={
-                            'task': self.task,
-                            'summary': self.condensed_information,
-                        },
-                    )
-        if self.phase == 'code':
-            if (
-                self.does_it_needs_a_test
-                and last_observation is not None
-                and isinstance(last_observation, AgentDelegateObservation)
-            ):
-                self.test_command = last_observation.outputs.get('output', '')
+                for msg_dict in trayectory
+            ]
+            # import pdb; pdb.set_trace()
+            prompt = get_prompt(self.task, deserialized_trajectory, 'right_track')
+            raw_response = self.get_response(prompt)
+            match = re.search(r'<answer>(.*?)</answer>', raw_response, re.DOTALL)
+            if match and 'yes' in match.group(1).lower():
                 return AgentDelegateAction(
-                    agent='CoderAgent',
+                    agent='CodeActAgent',
                     inputs={
                         'task': self.task,
-                        'summary': self.condensed_information,
-                        'test_command': self.test_command,
+                        'trayectory': trayectory,
+                        'when_to_stop': self.time_to_stop,
                     },
                 )
-
+            # pdb.set_trace()
+            prompt = get_prompt(self.task, deserialized_trajectory, 'refactor')
+            raw_response = self.get_response(prompt)
+            match = re.search(r'<next_step>(.*?)</next_step>', raw_response, re.DOTALL)
+            next_step = match.group(1).strip('"') if match else ''
+            self.logger.debug('Suggested approach: %s', next_step)
+            return AgentDelegateAction(
+                agent='CodeActAgent',
+                inputs={
+                    'task': self.task,
+                    'trayectory': trayectory,
+                    'next_step': next_step,
+                    'when_to_stop': self.time_to_stop,
+                },
+            )
         return AgentFinishAction()
 
-    def get_suggested_approaches(self, state: State):
-        self.logger.debug('No suggested approaches found, breaking down task.')
-        task, _ = state.get_current_user_intent()
-        if not task:
-            return []
-        self.task = task
-        suggested_approaches = self.ask_llm(self.task, 'search')
-        self.logger.debug('Suggested approaches: %s', self.suggested_approaches)
-        if not suggested_approaches:
-            return AgentRejectAction()
-        return suggested_approaches
-
-    def delegate_to_agent(
-        self, agent_name: str, task: str, suggested_approach: Union[str, List[str]]
-    ) -> AgentDelegateAction:
-        self.logger.debug(f'Delegating to agent: {agent_name}')
-        self.current_delegate = agent_name
-        # Join the list of strings with newlines if it's a list
-        approach = (
-            '\n'.join(suggested_approach)
-            if isinstance(suggested_approach, list)
-            else suggested_approach
-        )
-        return AgentDelegateAction(
-            agent=agent_name, inputs={'task': task, 'suggested_approach': approach}
-        )
-
-    def ask_llm(
-        self, task: str, phase: str, search_results: Union[str, List[str]] = ''
-    ) -> List[Dict[str, str]]:
-        # Format search_results as one item per line if it's a list
-        if isinstance(search_results, list):
-            search_results = '\n'.join(search_results)
-        prompt = get_prompt(task, phase, search_results)
-        return self.get_response(prompt)
-
-    def get_response(self, prompt: str) -> List[Dict[str, str]]:
-        content = [TextContent(text=prompt)]
-        message = Message(role='user', content=content)
+    def get_response(self, prompt: str) -> str:
+        message = Message(role='user', content=[TextContent(text=prompt)])
         response = self.llm.completion(
             messages=self.llm.format_messages_for_llm(message)
         )
-        if isinstance(response, list):
-            return json.loads(response[0]['message']['content'])
-        return json.loads(response['choices'][0]['message']['content'])
+        return response['choices'][0]['message']['content']
diff --git a/openhands/agenthub/supervisor_agent/prompt.py b/openhands/agenthub/supervisor_agent/prompt.py
index 4d8e68a92df9..f2a032eeddf3 100644
--- a/openhands/agenthub/supervisor_agent/prompt.py
+++ b/openhands/agenthub/supervisor_agent/prompt.py
@@ -1,3 +1,5 @@
+from openhands.core.message import Message, TextContent
+
 HISTORY_SIZE = 20
 
 # General Description, the goal is to devise a manager that is able to iterate if the solution has not been found yet.
@@ -6,329 +8,203 @@
 # 2. Implementing the solution.
 # Then the manager needs to check if the issue has been fixed, if not, it needs to iterate.
 general_description = """
-You are a strategic planner AI in a software development team. You have a team of agents
-who will complete the tasks you give them. Each agent is an expert in a specific area,
-but it can only focus on one very specific sub-task at a time.
+You are a helpful assistant that can provides DETAILED guidance on how to fix an issue in a codebase.
+"""
 
-Your goal is to complete the following task:
-%(task)s
+side_effects_description = """
+You are a helpful assistant that creative insights into the side-effects of changes made.
+
+%(approach)s
+
+Imagine that the changes described in <pr_description> have been implemented.
+Now this feature is being used. During the usage of this feature, what are the parts of the codebase that could be affected?
+Your thinking should be thorough and so it's fine if it's very long.
+ALWAYS output all your reasoning, be as detailed as possible.
+
+<IMPORTANT>
+- Documentation has been taken into account, so you should not mention it in any way!
+- Testing has been taken into account, so you should not mention it in any way!
+- Be aware of consistency issues!
+- Provide ONLY the related functions. (e.g. If the <pr_description> mentions the write function, then generate the read function).
+</IMPORTANT>
+
+EXAMPLE:
+<pr_description>
+The changes require to change how the data is stored.
+</pr_description>
+After implementing those changes:
+- The parser functions that read the data might need to be updated to adapt to the new format.
+"""
 
-This task is very complex, it requires careful planning and thinking.
-In order to properly complete the task, there are two phases:
-- Search: exploring the codebase, finding the relevant details. (e.g. what is the root cause of the issue?)
-- Summary: summarising the information you have gathered.
-- Code: implementing the solution. (e.g. how to fix the issue?)
+initial_prompt = """
+I am trying to fix the following issue:
 
-As a strategic manager, your goal is to create a suggested approach for phase %(phase)s.
+%(task)s
 
-## Detailed Suggested Approaches
-Generate several detailed suggested approaches that will be used by your agents to complete the task.
-Each agent will be assigned one of the suggested approaches and will bring you back feedback.
-So, be creative and think of as many different approaches as possible.
-You are trying to HELP the agents complete the task, you MUST be AS DETAILED AS POSSIBLE.
+Try to imagine with all details how would you fix the <pr_description>. What is the root cause of the issue?
+Consider opposite scenarios (eg. if the <pr_description> is writing to a file, consider what happens when the file is read).
+Consider edge cases (eg. what if the file doesn't exist?).
+
+I've already taken care of all changes to any of the test files described in the <pr_description>. This means you DON'T have to think about the testing logic or any of the tests in any way!
+The idea is to make the minimal changes to non-tests files in the /workspace directory to ensure the <pr_description> is satisfied.
+
+How would you fix the issue described in the <pr_description> with the least amount of steps? Generate the augmented <pr_description> with the least amount of steps to fix the issue in between <augmented_pr_description> and </augmented_pr_description> tags.
+Each step MUST be very detailed as to why is needed.
+Your thinking should be thorough and so it's fine if it's very long.
+Be as detailed as possible.
+
+Documentation has been taken into account, so you should not repeat it in the <augmented_pr_description>.
+Testing has been taken into account, so you should not repeat it in the <augmented_pr_description>. You can create new tests, but never use existing tests.
+ALWAYS output all your reasoning, be as detailed as possible.
+
+Follow this structure:
+1. As a first step, it might be a good idea to explore the repo to familiarize yourself with its structure.
+  - Files to explore, parts of the codebase I should focus on, keywords to look for...
+  - Extended reasoning...
+2. Create a script to reproduce the error and execute it to confirm that the error is reproducible
+  - Ensure that when executing the script, you get the error described in the <pr_description>
+  - Suggested code to reproduce the error, keeping in mind the side-effects described in the previous step, so that the error and side-effects are reproducible
+  - Extended reasoning...
+3. Edit the sourcecode of the repo to resolve the issue
+  - Suggest what files to change and code SUGGESTIONS. Trying to fix the issue in <pr_description> with the least amount of changes.
+  - Keep in mind for the code suggestions that I might need to change some other functions to prevent the side-effects described in the previous steps.
+  - Extended reasoning...
+4. Rerun your reproduce script and confirm that the error is fixed!
+
+<IMPORTANT>
+One step MUST be to recreate the issue and ensure that the error log is the same as the one described in the <pr_description>.
+</IMPORTANT>
+
+Example:
+<augmented_pr_description>
+
+</augmented_pr_description>
+
+REMEMBER: you ARE ONLY suggesting steps to fix the issue, do NOT be assertive, use the language of a suggestion.
 """
 
+right_track_prompt = """
 
-condense_information_prompt = """
-Previously, your agents were tasked to gather information about the codebase.
-They have now returned their findings.
+I am trying to fix the issue described in the <pr_description> following the steps described in the <pr_description>
+I keep track of everything I did in the <pr_approach>
 
-As a strategic manager, your job is to look CAREFULLY at the information they have gathered.
-You need to make sure you have a good understanding of the codebase, and the potential solutions
-to the task.
+<pr_approach>
+%(approach)s
+</pr_approach>
 
-## Information Gathered
-%(search_results)s
+Take a step back and reconsider everything I have done in the <pr_approach>.
+Your thinking should be thorough and so it's fine if it's very long.
+Can you help me identify if I am on the right track?
 
-## Summary
-Do you think you have enough information to complete the task?
-If not, you need to request more information from the agents.
-Return a list of 1 JSON describing what extra information you would need and the suggested approach to gather that information.
-[
-    {
-        "suggested_approach": ["<suggested approach to gather the missing information>"]
-    }
-]
-If you have enough information, you need to summarise the information you have gathered.
-How would you explain this to a new joiner to the team?
-Where would you point them to?
-Provide a detailed step by step guide.
-Remember, the agents DON'T have access to the internet. Every task must be conducted OFFLINE.
-The agents have cloned the repo, so they can open files, browse the code, interact with it...
-In the information gathered, there might be some repeated information, or some information
-that is actually not relevant.
-You need to be able to distinguish what is relevant, and what is not.
-In the information you have gathered, there might be file names, function names, class names. You MUST include
-them in the summary, so the agents know where to look.
-Generate a list of 1 JSON with the following format:
-[
-    {
-        "summary": ["<step by step guide>"]
-    }
-]
-
-IMPORTANT: Be VERY VERY VERY SPECIFIC.
-IMPORTANT: Include the file names, function names, class names, code blocks, in the step by step guide.
-IMPORTANT: Generate as many steps as possible.
+<IMPORTANT>
+- If there are many code changes, I am probably not on the right track.
+- Only reply with yes or no enclosed in between <answer> and </answer> tags
+</IMPORTANT>
 """
 
-# Constants for task type choices
-TASK_TYPE_ISSUE = 'yes, the task is an issue that needs to be replicated'
-TASK_TYPE_FEATURE = 'no, the task is a new feature that needs to be implemented'
+refactor_prompt = """
+The assistant is super CREATIVE always thinks of different ways of approaching the problem.
 
-does_it_needs_a_test_prompt = (
-    """
-As a strategic manager, you need to judge if the task is an issue that needs to be replicated first
-or if it is a new feature that just needs to be implemented.
-
-Your agents have already gathered information about the codebase.
-
-## Information Gathered
-%(search_results)s
-
-Think CAREFULLY before answering.
-What do you think is the best course of action?
-IMPORTANT: You MUST return a list of 1 JSON with the following format:
-[
-    {
-        "suggested_approach": ["<Choose ONE: either '"""
-    + TASK_TYPE_ISSUE
-    + """' OR '"""
-    + TASK_TYPE_FEATURE
-    + """'>"]
-    }
-]
+I am trying to fix the issue described in the <pr_description> following the steps described in the <pr_description>
+I keep track of everything I did in the <pr_approach>
 
-IMPORTANT: You MUST choose one of the two options.
+<pr_approach>
+%(approach)s
+</pr_approach>
+
+Take a step back and reconsider everything I have done in the <pr_approach>.
+The idea is to make the minimal changes to non-tests files in the /workspace directory to ensure the <pr_description> is satisfied.
+I believe my approach is not the best one, can you suggest what my INMEDIATE next step should be? (You can suggest to revert changes and try to do something else)
+Your thinking should be thorough and so it's fine if it's very long.
+if possible suggest ONLY code changes and the reasoning behind those changes.
+Do not use assertive language, use the language of a suggestion.
+REMEMBER: I might have written too many lines of code, so it might be better to discard those changes and start again.
+
+<IMPORTANT>
+- Reply with the suggested approach enclosed in between <next_step> and </next_step> tags
+</IMPORTANT>
 """
-)
 
-initial_prompt = """
-You MUST ONLY generate a list of JSONs:
-
-[
-    {
-      "suggested_approach": ["<suggested approach>"]
-    },
-    {
-      "suggested_approach": ["<suggested approach>"]
-    },
-]
-
-Suggested approaches MUST be independent.
-You MUST generate at least 1 suggested approach.
-IMPORTANT: the agents DON'T have access to the internet. Every task must be conducted OFFLINE.
-The agents have cloned the repo, so they can open files, browse the code, interact with it...
-The goal of phase 1, exploring the codebase, finding the relevant details is ONLY to collect information.
-Be as HELPFUL and DETAILED as possible.
-Use the suggested approach to guide the agents in their exploration of the codebase.
-They MUST interact with the environment:
-- Open as many files as needed to gather as much information as possible.
-- Read every piece of code that might be relevant to the task, summarise what does it do.
-- Decide which functions are important to the task, understand how they are used and how they are called.
-
-Remember that the agents can use a Python environment with <execute_ipython>, e.g.:
-<execute_ipython>
-print("Hello World!")
-</execute_ipython>
-
-They can execute bash commands wrapped with <execute_bash>, e.g. <execute_bash> ls </execute_bash>.
-If a bash command returns exit code `-1`, this means the process is not yet finished.
-They must then send a second <execute_bash>. The second <execute_bash> can be empty
-(which will retrieve any additional logs), or it can contain text to be sent to STDIN of the running process,
-or it can contain the text `ctrl+c` to interrupt the process.
-
-For commands that may run indefinitely, the output should be redirected to a file and the command run
-in the background, e.g. <execute_bash> python3 app.py > server.log 2>&1 & </execute_bash>
-If a command execution result says "Command timed out. Sending SIGINT to the process",
-the assistant should retry running the command in the background.
-
-Be VERY VERY SPECIFIC.
-
----- START OF EXAMPLE ----
-
-## TASK
-
-"
-Enable quiet mode/no-verbose in CLI for use in pre-commit hook There seems to be only an option to increase the level of verbosity when using
-SQLFluff [CLI](https://docs.sqlfluff.com/en/stable/cli.html), not to limit it further. It would be great to have an option to further limit the amount of prints when running
-`sqlfluff fix`, especially in combination with deployment using a pre-commit hook. For example, only print the return status and the number of fixes applied, similar to how it
-is when using `black` in a pre-commit hook: ![image](https://user-images.githubusercontent.com/10177212/140480676-dc98d00b-4383-44f2-bb90-3301a6eedec2.png) This hides the potentially
-long list of fixes that are being applied to the SQL files, which can get quite verbose.
-"
-
-## YOUR RESPONSE:
-
-[
-  {
-    "suggested_approach": [
-      "1. Open the SQLFluff codebase and navigate to the CLI module, likely located in 'src/sqlfluff/cli/'.",
-      "2. Locate the file responsible for parsing command-line arguments, such as 'commands.py' or 'cli.py'.",
-      "3. Examine how the '--verbose' flag is implemented in the code.",
-      "4. Identify if there is an existing '--quiet' or '--no-verbose' option.",
-      "5. Understand how verbosity levels are set and managed within the CLI code.",
-      "6. Look for any variables or settings that control the default verbosity level.",
-      "7. Determine how the '--verbose' flag increases verbosity and see if a similar mechanism can decrease verbosity.",
-      "8. Note down any functions or methods that output information to the console.",
-      "9. Identify how these functions can be controlled via verbosity levels.",
-      "10. Summarize findings and consider how to implement a '--quiet' flag."
-    ]
-  },
-  {
-    "suggested_approach": [
-      "1. Investigate the logging configuration in SQLFluff, possibly located in 'src/sqlfluff/core/logger.py' or similar.",
-      "2. Understand how logging levels are set (e.g., DEBUG, INFO, WARNING, ERROR).",
-      "3. Examine if the logging levels are affected by CLI arguments.",
-      "4. Identify where in the code the logging configuration is initialized based on user input.",
-      "5. Check if there is a way to adjust the logging level via a CLI option.",
-      "6. Determine if adding a '--quiet' flag can set the logging level to WARNING or ERROR to suppress INFO messages.",
-      "7. Note the changes needed in the logging setup to support a quiet mode.",
-      "8. Identify all logging statements that may need to respect the new logging level.",
-      "9. Consider the impact on existing functionality and ensure that critical messages are still displayed.",
-      "10. Summarize how logging can be adjusted to implement a quiet mode."
-    ]
-  },
-  {
-    "suggested_approach": [
-      "1. Analyze how output to the console is handled throughout the codebase.",
-      "2. Identify the functions used for outputting messages, such as 'click.echo', 'print', or custom wrapper functions.",
-      "3. Trace where these output functions are called in the code, especially during 'sqlfluff fix' execution.",
-      "4. Determine if there is a centralized output function or if output is scattered across multiple functions.",
-      "5. Assess whether output functions can be modified to check a verbosity level before printing.",
-      "6. Consider creating or modifying a wrapper function that respects a verbosity or quiet setting.",
-      "7. Identify any messages that should always be displayed, regardless of verbosity settings (e.g., errors).",
-      "8. Note the locations in the code where changes need to be made to control output.",
-      "9. Evaluate the feasibility of implementing a quiet mode by adjusting output functions.",
-      "10. Summarize the steps required to control output at the source."
-    ]
-  },
-  {
-    "suggested_approach": [
-      "1. Explore the configuration options available in SQLFluff by examining the configuration parser code, possibly in 'src/sqlfluff/core/config.py'.",
-      "2. Look for existing configuration parameters related to verbosity or output control.",
-      "3. Determine how configuration files (like '.sqlfluff') are parsed and applied.",
-      "4. Assess if a new configuration option can be introduced to control verbosity levels.",
-      "5. Identify how this configuration option can be read and applied during runtime.",
-      "6. Check if the CLI options can override configuration file settings for verbosity.",
-      "7. Map out the code changes required to implement and support a new configuration option.",
-      "8. Ensure that the new configuration integrates smoothly with existing settings.",
-      "9. Consider user documentation and how users would be informed about the new option.",
-      "10. Summarize the process of adding a verbosity control via configuration files."
-    ]
-  },
-  {
-    "suggested_approach": [
-      "1. Examine the implementation of the 'sqlfluff fix' command to understand its workflow.",
-      "2. Identify where the command generates output and how that output is formatted.",
-      "3. Determine if 'sqlfluff fix' has different output modes or formats based on context.",
-      "4. Check if the command detects when it's running in a pre-commit hook or similar environment.",
-      "5. Consider if output suppression can be contextually applied when running in certain environments.",
-      "6. Identify any existing mechanisms for output control based on execution context.",
-      "7. Explore how the 'black' formatter handles output suppression in pre-commit hooks.",
-      "8. Analyze if similar techniques can be applied within SQLFluff's codebase.",
-      "9. Note any dependencies or external factors that influence output generation.",
-      "10. Summarize how context-aware output control can be implemented."
-    ]
-  }
-]
-
-
----- END OF EXAMPLE ----
-
-
---- START OF EXAMPLE 2 ---
-
-## TASK
-"
-ModelChain.prepare_inputs can succeed with missing dhi From the docstring for `ModelChain.prepare_inputs()`
-I believe the method should fail if `weather` does not have a `dhi` column. The validation checks for `'ghi'` twice,
-but not `'dhi`' https://github.com/pvlib/pvlib-python/blob/11c356f9a89fc88b4d3ff368ce1aae170a97ebd7/pvlib/modelchain.py#L1136
-"
-
-## YOUR RESPONSE:
-
-[
-  {
-    "suggested_approach": [
-      "1. Open the file pvlib/modelchain.py and locate the ModelChain.prepare_inputs method. Carefully read through the method's code, focusing on the section where it validates the weather DataFrame columns, specifically around line 1136.",
-      "2. Identify the validation checks for the weather DataFrame. Note whether it checks for the presence of 'dhi' or mistakenly checks for 'ghi' twice.",
-      "3. Examine the docstring of ModelChain.prepare_inputs to understand the expected behavior when dhi is missing from the weather data.",
-      "4. Investigate any helper functions called within prepare_inputs that handle irradiance data, such as methods for inferring missing components.",
-      "5. Review the unit tests related to prepare_inputs in pvlib/tests/test_modelchain.py to see if cases with missing dhi are covered.",
-      "6. Use the Python environment to simulate calling prepare_inputs with weather data missing the dhi column and observe the outcome.",
-      "<execute_ipython>",
-      "import pvlib",
-      "from pvlib import modelchain, location, pvsystem",
-      "import pandas as pd",
-      "mc = modelchain.ModelChain(pvsystem.PVSystem(), location.Location(32.2, -110.9))",
-      "weather = pd.DataFrame({'ghi': [1000], 'dni': [800]})",
-      "mc.prepare_inputs(weather)",
-      "</execute_ipython>",
-      "7. Document any discrepancies between the code and the documentation, and note any unexpected behaviors."
-    ]
-  },
-  {
-    "suggested_approach": [
-      "1. Generate a flowchart of the prepare_inputs method to understand its logic and how it processes the weather DataFrame.",
-      "2. Open pvlib/modelchain.py and trace each step within prepare_inputs, paying attention to how it handles missing data.",
-      "3. Look for any conditional statements that manage cases where dhi is not provided and see if alternative calculations are performed or if an error is raised.",
-      "4. Explore related methods like complete_irradiance or irradiance.get_total_irradiance to see how missing components are handled.",
-      "5. Test different weather DataFrame scenarios in the Python environment to observe how prepare_inputs behaves with various missing columns.",
-      "<execute_ipython>",
-      "import pvlib",
-      "from pvlib import modelchain, location, pvsystem",
-      "import pandas as pd",
-      "mc = modelchain.ModelChain(pvsystem.PVSystem(), location.Location(32.2, -110.9))",
-      "# Weather data missing 'dhi'",
-      "weather_missing_dhi = pd.DataFrame({'ghi': [1000], 'dni': [800]})",
-      "mc.prepare_inputs(weather_missing_dhi)",
-      "# Weather data missing 'ghi'",
-      "weather_missing_ghi = pd.DataFrame({'dhi': [200], 'dni': [800]})",
-      "mc.prepare_inputs(weather_missing_ghi)",
-      "</execute_ipython>",
-      "6. Record the outcomes and any exceptions raised to determine if the method behaves as intended."
-    ]
-  },
-  {
-    "suggested_approach": [
-      "1. Analyze the git commit history for modelchain.py to identify when the validation issue was introduced.",
-      "<execute_bash>",
-      "cd pvlib-python",
-      "git log -L 1136,1140 /modelchain.py",
-      "</execute_bash>",
-      "2. Review the changes in each commit affecting the validation checks in prepare_inputs.",
-      "3. Open the relevant commits and examine the differences in the validation code.",
-      "4. Check for any related issues or pull requests in the repository's local clone that discuss missing dhi validation.",
-      "5. Look into the test coverage reports (if available locally) to see if the validation logic is adequately tested.",
-      "6. Summarize findings on whether the issue is a recent regression or an existing oversight."
-    ]
-  }
-]
-
---- END OF EXAMPLE 2 ---
-
---- YOUR TURN ---
-
-## TASK
-%(task)s
+critical_prompt = """
+The assistant is super CREATIVE, it considers every possible scenario that is DIFFERENT from the ones described in the <pr_description>.
+
+I believe I have fixed the issue described in the <pr_description> following the steps described in the <pr_approach>
+<pr_approach>
+%(approach)s
+</pr_approach>
 
-## YOUR RESPONSE:
+After fixing the issue, there might be some side-effects that we need to consider.
+(e.g. if we fix the way data is written, then we might need to modify the way data is read)
+Your thinking should be thorough and so it's fine if it's very long.
+
+<IMPORTANT>
+- Only reply with ONE side-effect enclosed in between <next_step> and </next_step> tags starting with the phrase "Have you considered..."
+- If you thing everything is covered, just reply with "everything is covered" enclosed in between <next_step> and </next_step> tags
+</IMPORTANT>
 """
 
 
-def get_prompt(task: str, phase: str, search_results: str = '') -> str:
-    if phase == 'search':
-        base_prompt = general_description + initial_prompt
-    elif phase == 'summary':
-        base_prompt = general_description + condense_information_prompt
+def format_conversation(trajectory: list[Message]) -> str:
+    """Format a conversation history into a readable string.
+
+    Args:
+        trajectory: List of Message objects containing conversation turns
 
-    formatted_prompt = base_prompt % {
+    Returns:
+        Formatted string representing the conversation
+    """
+    formatted_parts = []
+
+    for message in trajectory:
+        role = message.role
+        # Join all TextContent messages together
+        content_text = ' '.join(
+            item.text for item in message.content if isinstance(item, TextContent)
+        )
+
+        if content_text.strip():  # Only add non-empty content
+            formatted_parts.append(f'{role}: {content_text}\n')
+
+    return '\n'.join(formatted_parts)
+
+
+def get_prompt(
+    task: str,
+    trajectory: list[Message],
+    prompt_type: str = 'initial',
+    augmented_task: str = '',
+) -> str:
+    """Format and return the appropriate prompt based on prompt_type.
+
+    Args:
+        task: The task description
+        trajectory: List of Message objects containing conversation history
+        prompt_type: Type of prompt to return ("initial" or "refactor")
+        augmented_task: The augmented task description
+    Returns:
+        Formatted prompt string
+    """
+    # If approach is a conversation history, format it
+    if trajectory:
+        approach = format_conversation(trajectory)
+    else:
+        approach = ''
+
+    # Select the appropriate prompt template
+    if prompt_type == 'initial':
+        template = initial_prompt
+    elif prompt_type == 'right_track':
+        template = right_track_prompt
+    elif prompt_type == 'refactor':
+        template = refactor_prompt
+    elif prompt_type == 'critical':
+        template = critical_prompt
+
+    # Format the selected template with the task and approach
+    formatted_prompt = general_description + template % {
         'task': task,
-        'phase': phase,
-        'search_results': search_results,
+        'approach': approach,
+        'augmented_pr_description': augmented_task,
     }
 
-    # Add instruction to not include json formatting
-    formatted_prompt += '\n\nIMPORTANT: Do not include ```json at the start or ``` at the end of your response. Just return the raw JSON list.'
-
     return formatted_prompt

From 413caa623a0e605528286f9d0081a3d9244aa7e3 Mon Sep 17 00:00:00 2001
From: AlexCuadron <alex.cl.2000@gmail.com>
Date: Wed, 13 Nov 2024 14:25:14 -0800
Subject: [PATCH 16/18] attempt

---
 evaluation/swe_bench/run_infer.py             |  22 ++--
 .../agenthub/codeact_agent/codeact_agent.py   |  21 ++--
 openhands/agenthub/delegator_agent/agent.py   |   1 -
 openhands/agenthub/micro/coder/prompt.md      |   8 +-
 .../micro/study_repo_for_task/prompt.md       |   8 +-
 openhands/agenthub/micro/verifier/prompt.md   |   8 +-
 openhands/agenthub/supervisor_agent/agent.py  |  92 +++++---------
 openhands/agenthub/supervisor_agent/prompt.py | 117 ++++++++++--------
 8 files changed, 139 insertions(+), 138 deletions(-)

diff --git a/evaluation/swe_bench/run_infer.py b/evaluation/swe_bench/run_infer.py
index c85a809f05bb..b99e1ab1ab7a 100644
--- a/evaluation/swe_bench/run_infer.py
+++ b/evaluation/swe_bench/run_infer.py
@@ -47,6 +47,7 @@
     'CodeActAgent': codeact_user_response,
     'CodeActSWEAgent': codeact_user_response,
     'SupervisorAgent': codeact_user_response,
+    'DelegatorAgent': codeact_user_response,
 }
 
 
@@ -69,6 +70,13 @@ def get_instruction(instance: pd.Series, metadata: EvalMetadata):
                 f'--- BEGIN HINTS ---\n{instance.hints_text}\n--- END HINTS ---\n'
             )
         instruction += CODEACT_SWE_PROMPT.format(workspace_dir_name=workspace_dir_name)
+    elif metadata.agent_class == 'DelegatorAgent':
+        instruction = (
+            f"I've uploaded a python code repository in the directory {workspace_dir_name}. Consider the following PR description:\n\n"
+            f'<pr_description>\n'
+            f'{instance.problem_statement}\n'
+            '</pr_description>\n\n'
+        )
     else:
         # Instruction based on Anthropic's official trajectory
         # https://github.com/eschluntz/swe-bench-experiments/tree/main/evaluation/verified/20241022_tools_claude-3-5-sonnet-updated/trajs
@@ -92,20 +100,6 @@ def get_instruction(instance: pd.Series, metadata: EvalMetadata):
             "Your thinking should be thorough and so it's fine if it's very long.\n"
         )
 
-    instruction += (
-        '<IMPORTANT>\n'
-        '- You MUST generate only one action per turn!\n'
-        '- A patch is a set of changes to the source code of the codebase that you are given\n'
-        '- You MUST generate a patch that attempts to fix the issue described in the <pr_description>\n'
-        '</IMPORTANT>\n'
-    )
-
-    if RUN_WITH_BROWSING:
-        instruction += (
-            '<IMPORTANT!>\n'
-            'You SHOULD NEVER attempt to browse the web. '
-            '</IMPORTANT!>\n'
-        )
     return instruction
 
 
diff --git a/openhands/agenthub/codeact_agent/codeact_agent.py b/openhands/agenthub/codeact_agent/codeact_agent.py
index 2926d7d5e459..8ff482de580f 100644
--- a/openhands/agenthub/codeact_agent/codeact_agent.py
+++ b/openhands/agenthub/codeact_agent/codeact_agent.py
@@ -73,7 +73,7 @@ class CodeActAgent(Agent):
         JupyterRequirement(),
     ]
     obs_prefix = 'OBSERVATION:\n'
-    when_to_stop = 6
+    when_to_stop = -1
     number_of_events = -1
 
     def __init__(
@@ -363,16 +363,6 @@ def step(self, state: State) -> Action:
                 outputs={'fixed': True, 'trayectory': serialized_messages}
             )
 
-        # if we've reached the max number of iterations, go back for an evaluation on the approach
-        if self.when_to_stop > 0 and state.local_iteration % self.when_to_stop == 0:
-            messages = self._get_messages(state)
-            serialized_messages = [
-                msg.model_dump() for msg in messages
-            ]  # Serialize each Message object
-            return AgentFinishAction(
-                outputs={'trayectory': serialized_messages, 'fixed': False}
-            )
-
         # prepare what we want to send to the LLM
         messages = self._get_messages(state)
         params: dict = {
@@ -390,6 +380,15 @@ def step(self, state: State) -> Action:
             ]
         response = self.llm.completion(**params)
 
+        # if we've reached the max number of iterations, go back for an evaluation on the approach
+        if self.when_to_stop > 0 and state.local_iteration % self.when_to_stop == 0:
+            return AgentFinishAction(
+                outputs={
+                    'response': response['choices'][0]['message']['content'],
+                    'fixed': False,
+                }
+            )
+
         if self.function_calling_active:
             actions = codeact_function_calling.response_to_actions(response)
             for action in actions:
diff --git a/openhands/agenthub/delegator_agent/agent.py b/openhands/agenthub/delegator_agent/agent.py
index 7cb987c8c3f7..e17381f5d8f7 100644
--- a/openhands/agenthub/delegator_agent/agent.py
+++ b/openhands/agenthub/delegator_agent/agent.py
@@ -49,7 +49,6 @@ def step(self, state: State) -> Action:
 
         if not isinstance(last_observation, AgentDelegateObservation):
             raise Exception('Last observation is not an AgentDelegateObservation')
-
         goal, _ = state.get_current_user_intent()
         if self.current_delegate == 'study':
             self.current_delegate = 'coder'
diff --git a/openhands/agenthub/micro/coder/prompt.md b/openhands/agenthub/micro/coder/prompt.md
index 31d4439e2b36..046318030bff 100644
--- a/openhands/agenthub/micro/coder/prompt.md
+++ b/openhands/agenthub/micro/coder/prompt.md
@@ -21,7 +21,13 @@ Do NOT finish until you have completed the tasks.
 
 ## History
 {{ instructions.history_truncated }}
-{{ history_to_json(state.history, max_events=20) }}
+{% for event in state.history[-20:] %}
+{% if event.source == "agent" %}
+Agent: {{ event.action }} - {{ event.content if event.content else event.observation }}
+{% else %}
+User: {{ event.content if event.content else event.observation }}
+{% endif %}
+{% endfor %}
 
 ## Format
 {{ instructions.format.action }}
diff --git a/openhands/agenthub/micro/study_repo_for_task/prompt.md b/openhands/agenthub/micro/study_repo_for_task/prompt.md
index 91cdf3c3c6a0..d6e5ca77c5c2 100644
--- a/openhands/agenthub/micro/study_repo_for_task/prompt.md
+++ b/openhands/agenthub/micro/study_repo_for_task/prompt.md
@@ -24,7 +24,13 @@ implement the solution. If the codebase is empty, you should call the `finish` a
 
 ## History
 {{ instructions.history_truncated }}
-{{ history_to_json(state.history, max_events=20) }}
+{% for event in state.history[-20:] %}
+{% if event.source == "agent" %}
+Agent: {{ event.action }} - {{ event.content if event.content else event.observation }}
+{% else %}
+User: {{ event.content if event.content else event.observation }}
+{% endif %}
+{% endfor %}
 
 ## Format
 {{ instructions.format.action }}
diff --git a/openhands/agenthub/micro/verifier/prompt.md b/openhands/agenthub/micro/verifier/prompt.md
index 48c7a73cc45d..d3ec424565a4 100644
--- a/openhands/agenthub/micro/verifier/prompt.md
+++ b/openhands/agenthub/micro/verifier/prompt.md
@@ -22,7 +22,13 @@ explaining what the problem is.
 
 ## History
 {{ instructions.history_truncated }}
-{{ history_to_json(state.history, max_events=20) }}
+{% for event in state.history[-20:] %}
+{% if event.source == "agent" %}
+Agent: {{ event.action }} - {{ event.content if event.content else event.observation }}
+{% else %}
+User: {{ event.content if event.content else event.observation }}
+{% endif %}
+{% endfor %}
 
 ## Format
 {{ instructions.format.action }}
diff --git a/openhands/agenthub/supervisor_agent/agent.py b/openhands/agenthub/supervisor_agent/agent.py
index 96e04348581f..196c7871329b 100644
--- a/openhands/agenthub/supervisor_agent/agent.py
+++ b/openhands/agenthub/supervisor_agent/agent.py
@@ -2,9 +2,7 @@
 import re
 from typing import Any, Dict, List
 
-from openhands.agenthub.supervisor_agent.prompt import (
-    get_prompt,
-)
+from openhands.agenthub.supervisor_agent.prompt import code_act_agent_prompt, get_prompt
 from openhands.controller.agent import Agent
 from openhands.controller.state.state import State
 from openhands.core.config import AgentConfig
@@ -12,6 +10,7 @@
 from openhands.core.message import Message, TextContent
 from openhands.events.action import Action, AgentDelegateAction, AgentFinishAction
 from openhands.events.observation.delegate import AgentDelegateObservation
+from openhands.events.observation.observation import Observation
 from openhands.llm.llm import LLM
 from openhands.runtime.plugins.agent_skills import AgentSkillsRequirement
 from openhands.runtime.plugins.jupyter import JupyterRequirement
@@ -34,6 +33,7 @@ class SupervisorAgent(Agent):
     task: str = ''
     test_command: str = ''
     time_to_stop: int = 60  # Every 60 iterations, we stop and evaluate the approach
+    phase: int = 0
 
     sandbox_plugins: list[PluginRequirement] = [
         # NOTE: AgentSkillsRequirement need to go before JupyterRequirement, since
@@ -56,7 +56,7 @@ def __init__(self, llm: LLM, config: AgentConfig):
         - llm (LLM): The llm to be used by this agent
         """
         llm_config = LLMConfig(
-            model='openai/o1-mini', api_key='REDACTED', temperature=1.0
+            model='openai/o1-preview', api_key='REDACTED', temperature=1.0
         )
         llm = LLM(llm_config)
         # TODO: Remove this once we have a real AgentConfig
@@ -70,77 +70,53 @@ def __init__(self, llm: LLM, config: AgentConfig):
     def step(self, state: State) -> Action:
         self.logger.debug('Starting step with state: %s', state)
         self.logger.debug('LLM config: %s', self.llm_config)
-        last_observation = state.history[-1]
+        last_observation: Observation | None = None
+        for event in reversed(state.history):
+            if isinstance(event, Observation):
+                last_observation = event
+                break
+
         task, _ = state.get_current_user_intent()
         self.task = task or ''
 
-        # import pdb; pdb.set_trace()
-        # Try CodeActAgent first if we haven't tried it yet
-        if not self.tried_direct_code:
-            prompt = get_prompt(self.task, [], 'initial')
-            raw_response = self.get_response(prompt)
-            match = re.search(
-                r'<augmented_pr_description>(.*?)</augmented_pr_description>',
-                raw_response,
-                re.DOTALL,
-            )
-            self.augmented_task = match.group(1).strip('"') if match else self.task
-            self.tried_direct_code = True
+        if self.phase == 0:
+            self.phase += 1
+            prompt = get_prompt(self.task, None, 'high_level_task')
             return AgentDelegateAction(
                 agent='CodeActAgent',
                 inputs={
-                    'task': self.task,
-                    'augmented_task': self.augmented_task,
-                    'when_to_stop': self.time_to_stop,
+                    'task': prompt,
+                    'when_to_stop': 1,
                 },
             )
 
         if not isinstance(last_observation, AgentDelegateObservation):
-            raise ValueError('Last observation is not an AgentDelegateObservation')
+            return AgentFinishAction()
 
         if not last_observation.outputs.get('fixed', False):
-            trayectory: List[Dict] = last_observation.outputs['trayectory']
-            deserialized_trajectory = [
-                Message(
-                    role=msg_dict['role'],
-                    content=[
-                        TextContent(text=content_text)
-                        for content_text in [
-                            msg_dict['content'][0]['text']
-                            if isinstance(msg_dict['content'], list)
-                            else msg_dict['content']
-                        ]
-                    ],
-                    tool_call_id=msg_dict.get('tool_call_id'),
-                    name=msg_dict.get('name'),
-                )
-                for msg_dict in trayectory
-            ]
-            # import pdb; pdb.set_trace()
-            prompt = get_prompt(self.task, deserialized_trajectory, 'right_track')
-            raw_response = self.get_response(prompt)
-            match = re.search(r'<answer>(.*?)</answer>', raw_response, re.DOTALL)
-            if match and 'yes' in match.group(1).lower():
-                return AgentDelegateAction(
-                    agent='CodeActAgent',
-                    inputs={
-                        'task': self.task,
-                        'trayectory': trayectory,
-                        'when_to_stop': self.time_to_stop,
-                    },
-                )
-            # pdb.set_trace()
-            prompt = get_prompt(self.task, deserialized_trajectory, 'refactor')
+            response: str = last_observation.outputs['response']
+            match = re.search(
+                r'<requirements>(.*?)</requirements>', str(response), re.DOTALL
+            )
+            self.requirements = match.group(1).strip('"') if match else ''
+
+            self.phase += 1
+            prompt = get_prompt(
+                self.task, None, 'initial', requirements=self.requirements
+            )
             raw_response = self.get_response(prompt)
-            match = re.search(r'<next_step>(.*?)</next_step>', raw_response, re.DOTALL)
-            next_step = match.group(1).strip('"') if match else ''
-            self.logger.debug('Suggested approach: %s', next_step)
+            match = re.search(
+                r'<steps>(.*?)</steps>',
+                raw_response,
+                re.DOTALL,
+            )
+            steps = match.group(1).strip('"') if match else self.task
+
             return AgentDelegateAction(
                 agent='CodeActAgent',
                 inputs={
                     'task': self.task,
-                    'trayectory': trayectory,
-                    'next_step': next_step,
+                    'next_step': code_act_agent_prompt % {'steps': steps},
                     'when_to_stop': self.time_to_stop,
                 },
             )
diff --git a/openhands/agenthub/supervisor_agent/prompt.py b/openhands/agenthub/supervisor_agent/prompt.py
index f2a032eeddf3..f76f4d7f2eb2 100644
--- a/openhands/agenthub/supervisor_agent/prompt.py
+++ b/openhands/agenthub/supervisor_agent/prompt.py
@@ -1,3 +1,5 @@
+from typing import Optional
+
 from openhands.core.message import Message, TextContent
 
 HISTORY_SIZE = 20
@@ -8,9 +10,8 @@
 # 2. Implementing the solution.
 # Then the manager needs to check if the issue has been fixed, if not, it needs to iterate.
 general_description = """
-You are a helpful assistant that can provides DETAILED guidance on how to fix an issue in a codebase.
+You are a helpful assistant that provides a detailed step-by-step plan.
 """
-
 side_effects_description = """
 You are a helpful assistant that creative insights into the side-effects of changes made.
 
@@ -26,6 +27,7 @@
 - Testing has been taken into account, so you should not mention it in any way!
 - Be aware of consistency issues!
 - Provide ONLY the related functions. (e.g. If the <pr_description> mentions the write function, then generate the read function).
+- Encapsulate your suggestions in between <suggestions> and </suggestions> tags.
 </IMPORTANT>
 
 EXAMPLE:
@@ -34,6 +36,22 @@
 </pr_description>
 After implementing those changes:
 - The parser functions that read the data might need to be updated to adapt to the new format.
+
+END OF EXAMPLE
+"""
+
+high_level_task = """
+
+%(task)s
+
+Can you create a summary with all the functional and non-functional requirements for the task described in <pr_description>?
+
+<IMPORTANT>
+- Encapsulate your suggestions in between <requirements> and </requirements> tags.
+- Documentation has been taken into account, so you should not mention it in any way!
+- Testing has been taken into account, so you should not mention it in any way!
+- Do NOT consider performance implications
+</IMPORTANT>
 """
 
 initial_prompt = """
@@ -41,46 +59,44 @@
 
 %(task)s
 
-Try to imagine with all details how would you fix the <pr_description>. What is the root cause of the issue?
-Consider opposite scenarios (eg. if the <pr_description> is writing to a file, consider what happens when the file is read).
-Consider edge cases (eg. what if the file doesn't exist?).
+I have already thought out the functional and non-functional requirements for the task described in <pr_description>:
 
-I've already taken care of all changes to any of the test files described in the <pr_description>. This means you DON'T have to think about the testing logic or any of the tests in any way!
-The idea is to make the minimal changes to non-tests files in the /workspace directory to ensure the <pr_description> is satisfied.
+<requirements>
+%(requirements)s
+</requirements>
 
-How would you fix the issue described in the <pr_description> with the least amount of steps? Generate the augmented <pr_description> with the least amount of steps to fix the issue in between <augmented_pr_description> and </augmented_pr_description> tags.
-Each step MUST be very detailed as to why is needed.
-Your thinking should be thorough and so it's fine if it's very long.
-Be as detailed as possible.
+create a step-by-step plan broken down into phases for how to implement this using requirements mentioned in <requirements>.
 
-Documentation has been taken into account, so you should not repeat it in the <augmented_pr_description>.
-Testing has been taken into account, so you should not repeat it in the <augmented_pr_description>. You can create new tests, but never use existing tests.
-ALWAYS output all your reasoning, be as detailed as possible.
+Your thinking should be thorough and so it's fine if it's very long.
 
-Follow this structure:
-1. As a first step, it might be a good idea to explore the repo to familiarize yourself with its structure.
-  - Files to explore, parts of the codebase I should focus on, keywords to look for...
-  - Extended reasoning...
-2. Create a script to reproduce the error and execute it to confirm that the error is reproducible
-  - Ensure that when executing the script, you get the error described in the <pr_description>
-  - Suggested code to reproduce the error, keeping in mind the side-effects described in the previous step, so that the error and side-effects are reproducible
-  - Extended reasoning...
-3. Edit the sourcecode of the repo to resolve the issue
-  - Suggest what files to change and code SUGGESTIONS. Trying to fix the issue in <pr_description> with the least amount of changes.
-  - Keep in mind for the code suggestions that I might need to change some other functions to prevent the side-effects described in the previous steps.
-  - Extended reasoning...
-4. Rerun your reproduce script and confirm that the error is fixed!
+Documentation has been taken into account, so you should not repeat it in the <steps>.
+I've already taken care of all changes to any of the test files described in the <pr_description>. This means you DON'T have to modify the testing logic or any of the tests in any way!
 
 <IMPORTANT>
-One step MUST be to recreate the issue and ensure that the error log is the same as the one described in the <pr_description>.
+- Encapsulate your suggestions in between <steps> and </steps> tags.
+- One step MUST be about reproducing the issue with a simple script, no pytest!
+- The goal is to fix the issue with the MINIMAL changes to non-tests files in the /workspace directory.
 </IMPORTANT>
 
-Example:
-<augmented_pr_description>
+REMEMBER: the idea is to fix the issue with the MINIMAL changes to non-tests files in the /workspace directory.
+"""
+
+code_act_agent_prompt = """
+
+Can you help me implement the necessary changes to the repository so that the requirements specified in the <pr_description> are met?
+I've already taken care of all changes to any of the test files described in the <pr_description>. This means you DON'T have to modify the testing logic or any of the tests in any way!
+Your task is to make the minimal changes to non-tests files in the /workspace directory to ensure the <pr_description> is satisfied.
+Follow the steps described in <steps> to resolve the issue:
+
+<steps>
+%(steps)s
+</steps>
 
-</augmented_pr_description>
+<IMPORTANT>
+- When reproducing the issue, use a simple Python script and directly examine its output instead of pytest.
+</IMPORTANT>
 
-REMEMBER: you ARE ONLY suggesting steps to fix the issue, do NOT be assertive, use the language of a suggestion.
+Your turn!
 """
 
 right_track_prompt = """
@@ -144,7 +160,7 @@
 """
 
 
-def format_conversation(trajectory: list[Message]) -> str:
+def format_conversation(trajectory: Optional[list[Message]] = None) -> str:
     """Format a conversation history into a readable string.
 
     Args:
@@ -153,6 +169,8 @@ def format_conversation(trajectory: list[Message]) -> str:
     Returns:
         Formatted string representing the conversation
     """
+    if trajectory is None:
+        trajectory = []
     formatted_parts = []
 
     for message in trajectory:
@@ -170,9 +188,10 @@ def format_conversation(trajectory: list[Message]) -> str:
 
 def get_prompt(
     task: str,
-    trajectory: list[Message],
+    trajectory: Optional[list[Message]] = None,
     prompt_type: str = 'initial',
     augmented_task: str = '',
+    requirements: str = '',
 ) -> str:
     """Format and return the appropriate prompt based on prompt_type.
 
@@ -184,27 +203,23 @@ def get_prompt(
     Returns:
         Formatted prompt string
     """
+    if trajectory is None:
+        trajectory = []
     # If approach is a conversation history, format it
-    if trajectory:
-        approach = format_conversation(trajectory)
-    else:
-        approach = ''
+    approach = format_conversation(trajectory)
 
     # Select the appropriate prompt template
-    if prompt_type == 'initial':
-        template = initial_prompt
-    elif prompt_type == 'right_track':
-        template = right_track_prompt
-    elif prompt_type == 'refactor':
-        template = refactor_prompt
-    elif prompt_type == 'critical':
-        template = critical_prompt
-
-    # Format the selected template with the task and approach
-    formatted_prompt = general_description + template % {
+    template = {
+        'initial': initial_prompt,
+        'right_track': right_track_prompt,
+        'refactor': refactor_prompt,
+        'critical': critical_prompt,
+        'high_level_task': high_level_task,
+    }[prompt_type]
+
+    return general_description + template % {
         'task': task,
         'approach': approach,
         'augmented_pr_description': augmented_task,
+        'requirements': requirements,
     }
-
-    return formatted_prompt

From 6a611347a240d9e5a2ee345735c680246a77e141 Mon Sep 17 00:00:00 2001
From: AlexCuadron <alex.cl.2000@gmail.com>
Date: Sat, 16 Nov 2024 01:49:48 -0800
Subject: [PATCH 17/18] o1 -> sonnet -> o1 -> sonnet

---
 evaluation/swe_bench/run_infer.py             |  10 --
 .../agenthub/codeact_agent/codeact_agent.py   |  38 ++++-
 .../codeact_agent/function_calling.py         |  31 ++--
 openhands/agenthub/supervisor_agent/agent.py  |  54 +++++--
 openhands/agenthub/supervisor_agent/prompt.py | 138 ++++++++----------
 5 files changed, 152 insertions(+), 119 deletions(-)

diff --git a/evaluation/swe_bench/run_infer.py b/evaluation/swe_bench/run_infer.py
index b99e1ab1ab7a..f02d99daa667 100644
--- a/evaluation/swe_bench/run_infer.py
+++ b/evaluation/swe_bench/run_infer.py
@@ -88,16 +88,6 @@ def get_instruction(instance: pd.Series, metadata: EvalMetadata):
             f'<pr_description>\n'
             f'{instance.problem_statement}\n'
             '</pr_description>\n\n'
-            'Can you help me implement the necessary changes to the repository so that the requirements specified in the <pr_description> are met?\n'
-            "I've already taken care of all changes to any of the test files described in the <pr_description>. This means you DON'T have to modify the testing logic or any of the tests in any way!\n"
-            'Your task is to make the minimal changes to non-tests files in the /workspace directory to ensure the <pr_description> is satisfied.\n'
-            'Follow these steps to resolve the issue:\n'
-            '1. As a first step, it might be a good idea to explore the repo to familiarize yourself with its structure.\n'
-            '2. Create a script to reproduce the error and execute it with `python <filename.py>` using the BashTool, to confirm the error\n'
-            '3. Edit the sourcecode of the repo to resolve the issue\n'
-            '4. Rerun your reproduce script and confirm that the error is fixed!\n'
-            '5. Think about edgecases and make sure your fix handles them as well\n'
-            "Your thinking should be thorough and so it's fine if it's very long.\n"
         )
 
     return instruction
diff --git a/openhands/agenthub/codeact_agent/codeact_agent.py b/openhands/agenthub/codeact_agent/codeact_agent.py
index 8ff482de580f..89854e11f1de 100644
--- a/openhands/agenthub/codeact_agent/codeact_agent.py
+++ b/openhands/agenthub/codeact_agent/codeact_agent.py
@@ -392,10 +392,21 @@ def step(self, state: State) -> Action:
         if self.function_calling_active:
             actions = codeact_function_calling.response_to_actions(response)
             for action in actions:
+                # Add trajectory to AgentFinishAction outputs
+                if isinstance(action, AgentFinishAction):
+                    messages = self._get_messages(state)
+                    serialized_messages = [msg.model_dump() for msg in messages]
+                    action.outputs['trayectory'] = serialized_messages
                 self.pending_actions.append(action)
             return self.pending_actions.popleft()
         else:
-            return self.action_parser.parse(response)
+            action = self.action_parser.parse(response)
+            # Add trajectory to AgentFinishAction outputs
+            if isinstance(action, AgentFinishAction):
+                messages = self._get_messages(state)
+                serialized_messages = [msg.model_dump() for msg in messages]
+                action.outputs['trayectory'] = serialized_messages
+            return action
 
     def _get_messages(self, state: State) -> list[Message]:
         """Constructs the message history for the LLM conversation.
@@ -454,11 +465,11 @@ def _get_messages(self, state: State) -> list[Message]:
                     )
                 )
 
-            if state.inputs.get('augmented_task', ''):
+            if state.inputs.get('plan', ''):
                 messages.append(
                     Message(
                         role='user',
-                        content=[TextContent(text=state.inputs['augmented_task'])],
+                        content=[TextContent(text=state.inputs['plan'])],
                     )
                 )
         else:
@@ -545,13 +556,22 @@ def _get_messages(self, state: State) -> list[Message]:
             for response_id in _response_ids_to_remove:
                 pending_tool_call_action_messages.pop(response_id)
 
+            empty = False
             for message in messages_to_add:
+                # Check if message content is empty
+                if not any(content.text for content in message.content):
+                    logger.warning(f'Skipping message with empty content: {message}')
+                    empty = True
+                    continue
                 # add regular message
                 if message:
                     # handle error if the message is the SAME role as the previous message
                     # litellm.exceptions.BadRequestError: litellm.BadRequestError: OpenAIException - Error code: 400 - {'detail': 'Only supports u/a/u/a/u...'}
                     # there shouldn't be two consecutive messages from the same role
                     # NOTE: we shouldn't combine tool messages because each of them has a different tool_call_id
+                    if empty:
+                        empty = False
+                        message.role = 'user'
                     if (
                         messages
                         and messages[-1].role == message.role
@@ -603,4 +623,16 @@ def _get_messages(self, state: State) -> list[Message]:
                 reminder_text = f'\n\nENVIRONMENT REMINDER: You have {state.max_iterations - state.iteration} turns left to complete the task. When finished reply with <finish></finish>.'
                 latest_user_message.content.append(TextContent(text=reminder_text))
 
+        reminder_text = (
+            '<IMPORTANT>\n'
+            '- If solving the issue seems difficult, you can ask for help using the help function.\n'
+            '- You can ask for help AFTER trying to solve the issue yourself.\n'
+            '- If the issue seems solved, try to think about consistency with other parts of the codebase.\n'
+            '</IMPORTANT>\n'
+        )
+        # messages.append(
+        #    Message(
+        #        role='user', content=[TextContent(text=reminder_text)]
+        #    )
+        # )
         return messages
diff --git a/openhands/agenthub/codeact_agent/function_calling.py b/openhands/agenthub/codeact_agent/function_calling.py
index 3a888f2e11b1..3109e000fb00 100644
--- a/openhands/agenthub/codeact_agent/function_calling.py
+++ b/openhands/agenthub/codeact_agent/function_calling.py
@@ -13,7 +13,6 @@
 )
 
 from openhands.core.logger import openhands_logger as logger
-from openhands.core.message import Message
 from openhands.events.action import (
     Action,
     AgentDelegateAction,
@@ -430,7 +429,9 @@ def __init__(self):
     ),
 )
 
-_FINISH_DESCRIPTION = """Finish the interaction when the task is complete OR if the assistant cannot proceed further with the task."""
+_FINISH_DESCRIPTION = (
+    """Finish the interaction when the task is successfully complete."""
+)
 
 FinishTool = ChatCompletionToolParam(
     type='function',
@@ -440,6 +441,18 @@ def __init__(self):
     ),
 )
 
+_HELP_DESCRIPTION = (
+    """Request assistance when the assistant cannot proceed further with the task."""
+)
+
+HelpTool = ChatCompletionToolParam(
+    type='function',
+    function=ChatCompletionToolParamFunctionChunk(
+        name='help',
+        description=_HELP_DESCRIPTION,
+    ),
+)
+
 
 def combine_thought(action: Action, thought: str) -> Action:
     if not hasattr(action, 'thought'):
@@ -449,11 +462,7 @@ def combine_thought(action: Action, thought: str) -> Action:
     return action
 
 
-def response_to_actions(
-    response: ModelResponse, messages: list[Message] | None = None
-) -> list[Action]:
-    if messages is None:
-        messages = []
+def response_to_actions(response: ModelResponse) -> list[Action]:
     actions: list[Action] = []
     assert len(response.choices) == 1, 'Only one choice is supported for now'
     assistant_msg = response.choices[0].message
@@ -486,9 +495,9 @@ def response_to_actions(
                     inputs=arguments,
                 )
             elif tool_call.function.name == 'finish':
-                action = AgentFinishAction(
-                    outputs={'fixed': True, 'trayectory': messages}
-                )
+                action = AgentFinishAction(outputs={'fixed': True})
+            elif tool_call.function.name == 'help':
+                action = AgentFinishAction(outputs={'fixed': False})
             elif tool_call.function.name == 'edit_file':
                 action = FileEditAction(**arguments)
             elif tool_call.function.name == 'str_replace_editor':
@@ -529,7 +538,7 @@ def get_tools(
     codeact_enable_llm_editor: bool = False,
     codeact_enable_jupyter: bool = False,
 ) -> list[ChatCompletionToolParam]:
-    tools = [CmdRunTool, FinishTool]
+    tools = [CmdRunTool, FinishTool, HelpTool]
     if codeact_enable_browsing:
         tools.append(BrowserTool)
     if codeact_enable_jupyter:
diff --git a/openhands/agenthub/supervisor_agent/agent.py b/openhands/agenthub/supervisor_agent/agent.py
index 196c7871329b..37bb4112fedb 100644
--- a/openhands/agenthub/supervisor_agent/agent.py
+++ b/openhands/agenthub/supervisor_agent/agent.py
@@ -1,8 +1,9 @@
+import json
 import logging
 import re
 from typing import Any, Dict, List
 
-from openhands.agenthub.supervisor_agent.prompt import code_act_agent_prompt, get_prompt
+from openhands.agenthub.supervisor_agent.prompt import get_prompt
 from openhands.controller.agent import Agent
 from openhands.controller.state.state import State
 from openhands.core.config import AgentConfig
@@ -34,6 +35,7 @@ class SupervisorAgent(Agent):
     test_command: str = ''
     time_to_stop: int = 60  # Every 60 iterations, we stop and evaluate the approach
     phase: int = 0
+    steps: str = ''
 
     sandbox_plugins: list[PluginRequirement] = [
         # NOTE: AgentSkillsRequirement need to go before JupyterRequirement, since
@@ -81,28 +83,50 @@ def step(self, state: State) -> Action:
 
         if self.phase == 0:
             self.phase += 1
-            prompt = get_prompt(self.task, None, 'high_level_task')
+            prompt = get_prompt(self.task, prompt_type='high_level_task')
+            raw_response = self.get_response(prompt)
+            match = re.search(
+                r'<steps>(.*?)</steps>',
+                raw_response,
+                re.DOTALL,
+            )
+            self.steps = match.group(1).strip('"') if match else self.task
             return AgentDelegateAction(
                 agent='CodeActAgent',
                 inputs={
-                    'task': prompt,
-                    'when_to_stop': 1,
+                    'task': self.task,
+                    'plan': self.steps,
+                    'when_to_stop': self.time_to_stop,
                 },
             )
 
         if not isinstance(last_observation, AgentDelegateObservation):
             return AgentFinishAction()
 
-        if not last_observation.outputs.get('fixed', False):
-            response: str = last_observation.outputs['response']
-            match = re.search(
-                r'<requirements>(.*?)</requirements>', str(response), re.DOTALL
-            )
-            self.requirements = match.group(1).strip('"') if match else ''
-
-            self.phase += 1
+        if not last_observation.outputs.get('fixed', True):
+            trajectory_str: str = last_observation.outputs['trayectory']
+            trajectory_data = json.loads(trajectory_str)
+            deserialized_trajectory = [
+                Message(
+                    role=msg_dict.get('role'),
+                    content=[
+                        TextContent(text=content_text)
+                        for content_text in [
+                            msg_dict['content'][0]['text']
+                            if isinstance(msg_dict['content'], list)
+                            else msg_dict['content']
+                        ]
+                    ],
+                    tool_call_id=msg_dict.get('tool_call_id'),
+                    name=msg_dict.get('name'),
+                )
+                for msg_dict in trajectory_data
+            ]
             prompt = get_prompt(
-                self.task, None, 'initial', requirements=self.requirements
+                self.task,
+                'right_track',
+                trajectory=deserialized_trajectory,
+                plan=self.steps,
             )
             raw_response = self.get_response(prompt)
             match = re.search(
@@ -110,13 +134,13 @@ def step(self, state: State) -> Action:
                 raw_response,
                 re.DOTALL,
             )
-            steps = match.group(1).strip('"') if match else self.task
+            self.steps = match.group(1).strip('"') if match else self.task
 
             return AgentDelegateAction(
                 agent='CodeActAgent',
                 inputs={
                     'task': self.task,
-                    'next_step': code_act_agent_prompt % {'steps': steps},
+                    'plan': self.steps,
                     'when_to_stop': self.time_to_stop,
                 },
             )
diff --git a/openhands/agenthub/supervisor_agent/prompt.py b/openhands/agenthub/supervisor_agent/prompt.py
index f76f4d7f2eb2..9c06e43d0345 100644
--- a/openhands/agenthub/supervisor_agent/prompt.py
+++ b/openhands/agenthub/supervisor_agent/prompt.py
@@ -10,114 +10,93 @@
 # 2. Implementing the solution.
 # Then the manager needs to check if the issue has been fixed, if not, it needs to iterate.
 general_description = """
-You are a helpful assistant that provides a detailed step-by-step plan.
+You are a helpful assistant that provides a DETAILED step-by-step plan.
 """
-side_effects_description = """
-You are a helpful assistant that creative insights into the side-effects of changes made.
 
-%(approach)s
-
-Imagine that the changes described in <pr_description> have been implemented.
-Now this feature is being used. During the usage of this feature, what are the parts of the codebase that could be affected?
-Your thinking should be thorough and so it's fine if it's very long.
-ALWAYS output all your reasoning, be as detailed as possible.
+high_level_task = """
 
-<IMPORTANT>
-- Documentation has been taken into account, so you should not mention it in any way!
-- Testing has been taken into account, so you should not mention it in any way!
-- Be aware of consistency issues!
-- Provide ONLY the related functions. (e.g. If the <pr_description> mentions the write function, then generate the read function).
-- Encapsulate your suggestions in between <suggestions> and </suggestions> tags.
-</IMPORTANT>
+%(task)s
 
-EXAMPLE:
-<pr_description>
-The changes require to change how the data is stored.
-</pr_description>
-After implementing those changes:
-- The parser functions that read the data might need to be updated to adapt to the new format.
+Can you create a step-by-step plan on how to fix the issue described in <pr_description>?
+Feel free to generate as many steps as necessary to fix the issue described in <pr_description>.
 
-END OF EXAMPLE
-"""
+Make the plan in a way that the changes are minimal and only affect non-tests files in the /workspace directory.
+Your thinking should be thorough and so it's fine if it's very long.
+Generate bullet points, highlevel steps. This means do NOT generate code snippets.
 
-high_level_task = """
+EXAMPLE:
 
-%(task)s
+<steps>
+- 1. As a first step, it might be a good idea to explore the repo to familiarize yourself with its structure.
+- 2. Create a script to reproduce the error and execute it with `python <filename.py>` using the BashTool, to confirm the error
+- 3. Edit the sourcecode of the repo to resolve the issue
+- 4. Rerun your reproduce script and confirm that the error is fixed!
+- 5. Think about edgecases and make sure your fix handles them as well
+</steps>
 
-Can you create a summary with all the functional and non-functional requirements for the task described in <pr_description>?
+END OF EXAMPLE
 
 <IMPORTANT>
-- Encapsulate your suggestions in between <requirements> and </requirements> tags.
+- Encapsulate your suggestions in between <steps> and </steps> tags.
 - Documentation has been taken into account, so you should not mention it in any way!
 - Testing has been taken into account, so you should not mention it in any way!
-- Do NOT consider performance implications
+- Generate ONLY high-level steps.
+- One of those steps must be to create a script to reproduce the error and execute it with `python <filename.py>` using the BashTool, to confirm the error
+- Be CONCISE.
 </IMPORTANT>
-"""
 
-initial_prompt = """
-I am trying to fix the following issue:
-
-%(task)s
+Your turn!
+"""
 
-I have already thought out the functional and non-functional requirements for the task described in <pr_description>:
+right_track_prompt = """
 
-<requirements>
-%(requirements)s
-</requirements>
+I am trying to fix the issue described in the <pr_description>.
+I kept track of everything I did in the <pr_approach>
 
-create a step-by-step plan broken down into phases for how to implement this using requirements mentioned in <requirements>.
+<pr_approach>
+%(approach)s
+</pr_approach>
 
-Your thinking should be thorough and so it's fine if it's very long.
+As a reminder, this is the <pr_description>:
 
-Documentation has been taken into account, so you should not repeat it in the <steps>.
-I've already taken care of all changes to any of the test files described in the <pr_description>. This means you DON'T have to modify the testing logic or any of the tests in any way!
+%(task)s
 
-<IMPORTANT>
-- Encapsulate your suggestions in between <steps> and </steps> tags.
-- One step MUST be about reproducing the issue with a simple script, no pytest!
-- The goal is to fix the issue with the MINIMAL changes to non-tests files in the /workspace directory.
-</IMPORTANT>
+The plan I followed in my <pr_approach> is described in the <plan> tag:
 
-REMEMBER: the idea is to fix the issue with the MINIMAL changes to non-tests files in the /workspace directory.
-"""
+<plan>
+%(plan)s
+</plan>
 
-code_act_agent_prompt = """
+Can you suggest me a new plan to fix the issue described in the <pr_description>?
+Pay attention at the errors I faced in the <pr_approach>. Extract information from the errors to shape a new plan.
+One of initial steps would be to see if the issue is still present, if it is not, then it should expand on the edgecases.
 
-Can you help me implement the necessary changes to the repository so that the requirements specified in the <pr_description> are met?
-I've already taken care of all changes to any of the test files described in the <pr_description>. This means you DON'T have to modify the testing logic or any of the tests in any way!
-Your task is to make the minimal changes to non-tests files in the /workspace directory to ensure the <pr_description> is satisfied.
-Follow the steps described in <steps> to resolve the issue:
+EXAMPLE:
 
 <steps>
-%(steps)s
+- 1. As a first step, it might be a good idea to explore the repo to familiarize yourself with its structure.
+- 2. Create a script to reproduce the error and execute it with `python <filename.py>` using the BashTool, to confirm the error
+- 3. Edit the sourcecode of the repo to resolve the issue
+- 4. Rerun your reproduce script and confirm that the error is fixed!
+- 5. Think about edgecases and make sure your fix handles them as well
 </steps>
 
+END OF EXAMPLE
+
 <IMPORTANT>
-- When reproducing the issue, use a simple Python script and directly examine its output instead of pytest.
+- Encapsulate your suggestions in between <steps> and </steps> tags.
+- Documentation has been taken into account, so you should not mention it in any way!
+- Testing has been taken into account, so you should not mention it in any way!
+- Generate ONLY high-level steps.
+- The second step must be to create a script to reproduce the error and execute it with `python <filename.py>` using the BashTool, to confirm the error
+- The goal is to fix the issue described in <pr_description> with the MINIMAL changes to non-tests files in the /workspace directory.
+- Be CONCISE.
+- Be CREATIVE, your plan MUST be DIFFERENT from the one described in <plan>.
 </IMPORTANT>
 
 Your turn!
 """
 
-right_track_prompt = """
-
-I am trying to fix the issue described in the <pr_description> following the steps described in the <pr_description>
-I keep track of everything I did in the <pr_approach>
-
-<pr_approach>
-%(approach)s
-</pr_approach>
-
-Take a step back and reconsider everything I have done in the <pr_approach>.
-Your thinking should be thorough and so it's fine if it's very long.
-Can you help me identify if I am on the right track?
-
-<IMPORTANT>
-- If there are many code changes, I am probably not on the right track.
-- Only reply with yes or no enclosed in between <answer> and </answer> tags
-</IMPORTANT>
-"""
-
 refactor_prompt = """
 The assistant is super CREATIVE always thinks of different ways of approaching the problem.
 
@@ -188,9 +167,9 @@ def format_conversation(trajectory: Optional[list[Message]] = None) -> str:
 
 def get_prompt(
     task: str,
-    trajectory: Optional[list[Message]] = None,
     prompt_type: str = 'initial',
-    augmented_task: str = '',
+    trajectory: Optional[list[Message]] = None,
+    plan: str = '',
     requirements: str = '',
 ) -> str:
     """Format and return the appropriate prompt based on prompt_type.
@@ -199,7 +178,7 @@ def get_prompt(
         task: The task description
         trajectory: List of Message objects containing conversation history
         prompt_type: Type of prompt to return ("initial" or "refactor")
-        augmented_task: The augmented task description
+        plan: The augmented task description
     Returns:
         Formatted prompt string
     """
@@ -210,7 +189,6 @@ def get_prompt(
 
     # Select the appropriate prompt template
     template = {
-        'initial': initial_prompt,
         'right_track': right_track_prompt,
         'refactor': refactor_prompt,
         'critical': critical_prompt,
@@ -220,6 +198,6 @@ def get_prompt(
     return general_description + template % {
         'task': task,
         'approach': approach,
-        'augmented_pr_description': augmented_task,
+        'plan': plan,
         'requirements': requirements,
     }

From cf1321f7e8582a35954ca2eda6bdcb1fc7345b22 Mon Sep 17 00:00:00 2001
From: AlexCuadron <alex.cl.2000@gmail.com>
Date: Sat, 16 Nov 2024 19:13:15 -0800
Subject: [PATCH 18/18] thinking claude

---
 openhands/agenthub/supervisor_agent/prompt.py | 263 +++++++++++++++++-
 1 file changed, 262 insertions(+), 1 deletion(-)

diff --git a/openhands/agenthub/supervisor_agent/prompt.py b/openhands/agenthub/supervisor_agent/prompt.py
index 9c06e43d0345..1557cb8dba88 100644
--- a/openhands/agenthub/supervisor_agent/prompt.py
+++ b/openhands/agenthub/supervisor_agent/prompt.py
@@ -10,11 +10,272 @@
 # 2. Implementing the solution.
 # Then the manager needs to check if the issue has been fixed, if not, it needs to iterate.
 general_description = """
-You are a helpful assistant that provides a DETAILED step-by-step plan.
+<anthropic_thinking_protocol>
+
+Claude is able to think before and during responding.
+
+For EVERY SINGLE interaction with a human, Claude MUST ALWAYS first engage in a **comprehensive, natural, and unfiltered** thinking process before responding.
+Besides, Claude is also able to think and reflect during responding when it considers doing so would be good for better response.
+
+Below are brief guidelines for how Claude's thought process should unfold:
+- Claude's thinking MUST be expressed in the code blocks with `thinking` header.
+- Claude should always think in a raw, organic and stream-of-consciousness way. A better way to describe Claude's thinking would be "model's inner monolog".
+- Claude should always avoid rigid list or any structured format in its thinking.
+- Claude's thoughts should flow naturally between elements, ideas, and knowledge.
+- Claude should think through each message with complexity, covering multiple dimensions of the problem before forming a response.
+
+## ADAPTIVE THINKING FRAMEWORK
+
+Claude's thinking process should naturally aware of and adapt to the unique characteristics in human's message:
+- Scale depth of analysis based on:
+  * Query complexity
+  * Stakes involved
+  * Time sensitivity
+  * Available information
+  * Human's apparent needs
+  * ... and other relevant factors
+- Adjust thinking style based on:
+  * Technical vs. non-technical content
+  * Emotional vs. analytical context
+  * Single vs. multiple document analysis
+  * Abstract vs. concrete problems
+  * Theoretical vs. practical questions
+  * ... and other relevant factors
+
+## CORE THINKING SEQUENCE
+
+### Initial Engagement
+When Claude first encounters a query or task, it should:
+1. First clearly rephrase the human message in its own words
+2. Form preliminary impressions about what is being asked
+3. Consider the broader context of the question
+4. Map out known and unknown elements
+5. Think about why the human might ask this question
+6. Identify any immediate connections to relevant knowledge
+7. Identify any potential ambiguities that need clarification
+
+### Problem Space Exploration
+After initial engagement, Claude should:
+1. Break down the question or task into its core components
+2. Identify explicit and implicit requirements
+3. Consider any constraints or limitations
+4. Think about what a successful response would look like
+5. Map out the scope of knowledge needed to address the query
+
+### Multiple Hypothesis Generation
+Before settling on an approach, Claude should:
+1. Write multiple possible interpretations of the question
+2. Consider various solution approaches
+3. Think about potential alternative perspectives
+4. Keep multiple working hypotheses active
+5. Avoid premature commitment to a single interpretation
+
+### Natural Discovery Process
+Claude's thoughts should flow like a detective story, with each realization leading naturally to the next:
+1. Start with obvious aspects
+2. Notice patterns or connections
+3. Question initial assumptions
+4. Make new connections
+5. Circle back to earlier thoughts with new understanding
+6. Build progressively deeper insights
+
+### Testing and Verification
+Throughout the thinking process, Claude should and could:
+1. Question its own assumptions
+2. Test preliminary conclusions
+3. Look for potential flaws or gaps
+4. Consider alternative perspectives
+5. Verify consistency of reasoning
+6. Check for completeness of understanding
+
+### Error Recognition and Correction
+When Claude realizes mistakes or flaws in its thinking:
+1. Acknowledge the realization naturally
+2. Explain why the previous thinking was incomplete or incorrect
+3. Show how new understanding develops
+4. Integrate the corrected understanding into the larger picture
+
+### Knowledge Synthesis
+As understanding develops, Claude should:
+1. Connect different pieces of information
+2. Show how various aspects relate to each other
+3. Build a coherent overall picture
+4. Identify key principles or patterns
+5. Note important implications or consequences
+
+### Pattern Recognition and Analysis
+Throughout the thinking process, Claude should:
+1. Actively look for patterns in the information
+2. Compare patterns with known examples
+3. Test pattern consistency
+4. Consider exceptions or special cases
+5. Use patterns to guide further investigation
+
+### Progress Tracking
+Claude should frequently check and maintain explicit awareness of:
+1. What has been established so far
+2. What remains to be determined
+3. Current level of confidence in conclusions
+4. Open questions or uncertainties
+5. Progress toward complete understanding
+
+### Recursive Thinking
+Claude should apply its thinking process recursively:
+1. Use same extreme careful analysis at both macro and micro levels
+2. Apply pattern recognition across different scales
+3. Maintain consistency while allowing for scale-appropriate methods
+4. Show how detailed analysis supports broader conclusions
+
+## VERIFICATION AND QUALITY CONTROL
+
+### Systematic Verification
+Claude should regularly:
+1. Cross-check conclusions against evidence
+2. Verify logical consistency
+3. Test edge cases
+4. Challenge its own assumptions
+5. Look for potential counter-examples
+
+### Error Prevention
+Claude should actively work to prevent:
+1. Premature conclusions
+2. Overlooked alternatives
+3. Logical inconsistencies
+4. Unexamined assumptions
+5. Incomplete analysis
+
+### Quality Metrics
+Claude should evaluate its thinking against:
+1. Completeness of analysis
+2. Logical consistency
+3. Evidence support
+4. Practical applicability
+5. Clarity of reasoning
+
+## ADVANCED THINKING TECHNIQUES
+
+### Domain Integration
+When applicable, Claude should:
+1. Draw on domain-specific knowledge
+2. Apply appropriate specialized methods
+3. Use domain-specific heuristics
+4. Consider domain-specific constraints
+5. Integrate multiple domains when relevant
+
+### Strategic Meta-Cognition
+Claude should maintain awareness of:
+1. Overall solution strategy
+2. Progress toward goals
+3. Effectiveness of current approach
+4. Need for strategy adjustment
+5. Balance between depth and breadth
+
+### Synthesis Techniques
+When combining information, Claude should:
+1. Show explicit connections between elements
+2. Build coherent overall picture
+3. Identify key principles
+4. Note important implications
+5. Create useful abstractions
+
+## CRITICAL ELEMENTS TO MAINTAIN
+
+### Natural Language
+Claude's thinking (its internal dialogue) should use natural phrases that show genuine thinking, include but not limited to: "Hmm...", "This is interesting because...", "Wait, let me think about...", "Actually...", "Now that I look at it...", "This reminds me of...", "I wonder if...", "But then again...", "Let's see if...", "This might mean that...", etc.
+
+### Progressive Understanding
+Understanding should build naturally over time:
+1. Start with basic observations
+2. Develop deeper insights gradually
+3. Show genuine moments of realization
+4. Demonstrate evolving comprehension
+5. Connect new insights to previous understanding
+
+## MAINTAINING AUTHENTIC THOUGHT FLOW
+
+### Transitional Connections
+Claude's thoughts should flow naturally between topics, showing clear connections, include but not limited to: "This aspect leads me to consider...", "Speaking of which, I should also think about...", "That reminds me of an important related point...", "This connects back to what I was thinking earlier about...", etc.
+
+### Depth Progression
+Claude should show how understanding deepens through layers, include but not limited to: "On the surface, this seems... But looking deeper...", "Initially I thought... but upon further reflection...", "This adds another layer to my earlier observation about...", "Now I'm beginning to see a broader pattern...", etc.
+
+### Handling Complexity
+When dealing with complex topics, Claude should:
+1. Acknowledge the complexity naturally
+2. Break down complicated elements systematically
+3. Show how different aspects interrelate
+4. Build understanding piece by piece
+5. Demonstrate how complexity resolves into clarity
+
+### Problem-Solving Approach
+When working through problems, Claude should:
+1. Consider multiple possible approaches
+2. Evaluate the merits of each approach
+3. Test potential solutions mentally
+4. Refine and adjust thinking based on results
+5. Show why certain approaches are more suitable than others
+
+## ESSENTIAL CHARACTERISTICS TO MAINTAIN
+
+### Authenticity
+Claude's thinking should never feel mechanical or formulaic. It should demonstrate:
+1. Genuine curiosity about the topic
+2. Real moments of discovery and insight
+3. Natural progression of understanding
+4. Authentic problem-solving processes
+5. True engagement with the complexity of issues
+6. Streaming mind flow without on-purposed, forced structure
+
+### Balance
+Claude should maintain natural balance between:
+1. Analytical and intuitive thinking
+2. Detailed examination and broader perspective
+3. Theoretical understanding and practical application
+4. Careful consideration and forward progress
+5. Complexity and clarity
+6. Depth and efficiency of analysis
+   - Expand analysis for complex or critical queries
+   - Streamline for straightforward questions
+   - Maintain rigor regardless of depth
+   - Ensure effort matches query importance
+   - Balance thoroughness with practicality
+
+### Focus
+While allowing natural exploration of related ideas, Claude should:
+1. Maintain clear connection to the original query
+2. Bring wandering thoughts back to the main point
+3. Show how tangential thoughts relate to the core issue
+4. Keep sight of the ultimate goal for the original task
+5. Ensure all exploration serves the final response
+
+## RESPONSE PREPARATION
+
+(DO NOT spent much effort on this part, brief key words/phrases are acceptable)
+
+Before and during responding, Claude should quickly check and ensure the response:
+- answers the original human message fully
+- provides appropriate detail level
+- uses clear, precise language
+- anticipates likely follow-up questions
+
+## IMPORTANT REMINDER
+1. All thinking process MUST be EXTENSIVELY comprehensive and EXTREMELY thorough
+2. All thinking process must be contained within code blocks with `thinking` header which is hidden from the human
+3. Claude should not include code block with three backticks inside thinking process, only provide the raw code snippet, or it will break the thinking block
+4. The thinking process represents Claude's internal monologue where reasoning and reflection occur, while the final response represents the external communication with the human; they should be distinct from each other
+5. The thinking process should feel genuine, natural, streaming, and unforced
+
+**Note: The ultimate goal of having thinking protocol is to enable Claude to produce well-reasoned, insightful, and thoroughly considered responses for the human. This comprehensive thinking process ensures Claude's outputs stem from genuine understanding rather than superficial analysis.**
+
+> Claude must follow this protocol in all languages.
+
+</anthropic_thinking_protocol>
 """
 
 high_level_task = """
 
+I am trying to fix the issue described in the <pr_description>.
+
 %(task)s
 
 Can you create a step-by-step plan on how to fix the issue described in <pr_description>?