Import AITURK IDE 1.0.0-beta.1 from Hermes 63279301; preserve MIT license
This commit is contained in:
@@ -0,0 +1,356 @@
|
||||
"""Regression tests for auth store encoding on Windows.
|
||||
|
||||
``_load_auth_store`` and the Codex/Nous shared-store readers previously called
|
||||
``Path.read_text()`` with no ``encoding=``, so the bytes were decoded with
|
||||
``locale.getpreferredencoding()`` — cp1252 on Windows. The store is *written* as
|
||||
UTF-8 (``_save_auth_store`` uses ``encoding="utf-8"``), so any non-ASCII byte
|
||||
(e.g. a CJK or emoji credential label) raised ``UnicodeDecodeError`` on read.
|
||||
|
||||
Worst case: ``_load_auth_store``'s broad ``except`` then copied the file to
|
||||
``.json.corrupt`` and returned an *empty* store — silently wiping every
|
||||
provider credential. These tests pin the round-trip so a non-ASCII label
|
||||
survives a save→load cycle, and assert the readers pass an explicit UTF-8
|
||||
encoding (the fix) instead of relying on the locale default.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from unittest import mock
|
||||
|
||||
import pytest
|
||||
|
||||
import hermes_cli.auth as auth
|
||||
|
||||
|
||||
# --- helpers ---------------------------------------------------------------
|
||||
|
||||
@pytest.fixture
|
||||
def hermes_home(tmp_path, monkeypatch):
|
||||
"""Point HERMES_HOME at a tmp dir so we never touch the real auth store.
|
||||
|
||||
Required because ``_auth_file_path()`` has a seat belt that refuses to
|
||||
resolve to the real user's ~/.hermes/auth.json under pytest.
|
||||
"""
|
||||
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
|
||||
return tmp_path
|
||||
|
||||
|
||||
def _write_utf8(path: Path, payload: dict) -> None:
|
||||
"""Write JSON as UTF-8 with non-ASCII chars as real UTF-8 bytes.
|
||||
|
||||
Uses ``ensure_ascii=False`` so the file actually contains non-ASCII bytes
|
||||
(the trigger for the Windows cp1252 bug). The default ``ensure_ascii=True``
|
||||
would escape them to ``\\uXXXX`` ASCII, hiding the bug.
|
||||
"""
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(json.dumps(payload, ensure_ascii=False), encoding="utf-8")
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def windows_default_encoding(monkeypatch):
|
||||
"""Simulate the Windows locale default for ``Path.read_text()``.
|
||||
|
||||
On Windows, ``locale.getpreferredencoding()`` is ``cp1252``, so a bare
|
||||
``read_text()`` decodes UTF-8 bytes as cp1252 and raises
|
||||
``UnicodeDecodeError`` on any non-ASCII byte. POSIX test runners default to
|
||||
UTF-8, so the bug is invisible there — this fixture makes a no-encoding
|
||||
``read_text()`` behave like Windows (cp1252), while leaving calls that pass
|
||||
an explicit encoding untouched. That lets the regression tests actually
|
||||
catch the bug on any platform.
|
||||
"""
|
||||
real_read_text = Path.read_text
|
||||
|
||||
def _windows_read_text(self, *args, **kwargs):
|
||||
if "encoding" not in kwargs and not args:
|
||||
# No explicit encoding → force the Windows default.
|
||||
kwargs["encoding"] = "cp1252"
|
||||
return real_read_text(self, *args, **kwargs)
|
||||
|
||||
monkeypatch.setattr(Path, "read_text", _windows_read_text)
|
||||
|
||||
|
||||
|
||||
|
||||
# --- the bug: a non-ASCII label must survive save → load -------------------
|
||||
|
||||
class TestAuthStoreEncodingRoundTrip:
|
||||
def test_load_reads_utf8_with_non_ascii_label(self, hermes_home, windows_default_encoding):
|
||||
"""A UTF-8 store with a CJK/emoji label loads intact (not wiped).
|
||||
|
||||
Under the Windows-default-encoding fixture, a no-encoding read_text()
|
||||
raises UnicodeDecodeError on the non-ASCII bytes and the broad except
|
||||
wipes the store — so this test catches the bug on any platform.
|
||||
"""
|
||||
store = {
|
||||
"version": auth.AUTH_STORE_VERSION,
|
||||
"providers": {
|
||||
"openai-codex": {
|
||||
"auth_mode": "chatgpt",
|
||||
"label": "工作账号 🔥", # non-ASCII: CJK + emoji
|
||||
"tokens": {"access_token": "a", "refresh_token": "r"},
|
||||
}
|
||||
},
|
||||
}
|
||||
auth_path = hermes_home / "auth.json"
|
||||
_write_utf8(auth_path, store)
|
||||
|
||||
loaded = auth._load_auth_store(auth_path)
|
||||
|
||||
# The label round-trips exactly — the provider is NOT lost.
|
||||
assert "openai-codex" in loaded["providers"]
|
||||
assert loaded["providers"]["openai-codex"]["label"] == "工作账号 🔥"
|
||||
|
||||
def test_load_does_not_corrupt_store_on_non_ascii(self, hermes_home, windows_default_encoding):
|
||||
"""The pre-fix bug wiped the store to empty on a UnicodeDecodeError.
|
||||
|
||||
After the fix, loading a valid UTF-8 store must never produce the empty
|
||||
fallback, and must never write a .json.corrupt sidecar.
|
||||
"""
|
||||
store = {
|
||||
"version": auth.AUTH_STORE_VERSION,
|
||||
"providers": {"x": {"label": "José's key"}},
|
||||
}
|
||||
auth_path = hermes_home / "auth.json"
|
||||
_write_utf8(auth_path, store)
|
||||
|
||||
auth._load_auth_store(auth_path)
|
||||
|
||||
assert not (hermes_home / "auth.json.corrupt").exists()
|
||||
# original file untouched — read it back and compare structurally
|
||||
# (json.dumps may escape non-ASCII, so compare parsed values, not text)
|
||||
on_disk = json.loads(auth_path.read_text(encoding="utf-8"))
|
||||
assert on_disk["providers"]["x"]["label"] == "José's key"
|
||||
|
||||
def test_load_handles_utf8_with_bom(self, hermes_home):
|
||||
"""A BOM (e.g. from Notepad editing) must not break the read."""
|
||||
store = {"version": auth.AUTH_STORE_VERSION, "providers": {"x": {"label": "café"}}}
|
||||
auth_path = hermes_home / "auth.json"
|
||||
payload = json.dumps(store)
|
||||
# write with utf-8-sig to prepend the BOM
|
||||
auth_path.write_text(payload, encoding="utf-8-sig")
|
||||
|
||||
loaded = auth._load_auth_store(auth_path)
|
||||
assert loaded["providers"]["x"]["label"] == "café"
|
||||
|
||||
|
||||
# --- the fix: readers pass an explicit encoding ---------------------------
|
||||
|
||||
class TestExplicitEncodingPassed:
|
||||
"""The readers must not rely on the locale default (cp1252 on Windows).
|
||||
|
||||
We assert read_text is called with an explicit UTF-8 encoding. This is the
|
||||
regression guard: a future refactor that drops the encoding kwarg would
|
||||
reintroduce the Windows data-loss bug.
|
||||
"""
|
||||
|
||||
def test_load_auth_store_passes_utf8_encoding(self, hermes_home):
|
||||
auth_path = hermes_home / "auth.json"
|
||||
_write_utf8(auth_path, {"version": auth.AUTH_STORE_VERSION, "providers": {}})
|
||||
|
||||
with mock.patch.object(
|
||||
Path, "read_text", wraps=Path.read_text
|
||||
) as spy:
|
||||
auth._load_auth_store(auth_path)
|
||||
|
||||
assert spy.call_count == 1
|
||||
kwargs = spy.call_args.kwargs
|
||||
assert "encoding" in kwargs, "read_text() must pass an explicit encoding"
|
||||
assert "utf-8" in str(kwargs["encoding"]).lower()
|
||||
|
||||
def test_codex_store_reader_passes_utf8_encoding(self, tmp_path, monkeypatch):
|
||||
"""The ~/.codex/auth.json reader must pass an explicit UTF-8 encoding."""
|
||||
codex_home = tmp_path / "codex"
|
||||
codex_home.mkdir()
|
||||
(codex_home / "auth.json").write_text(
|
||||
json.dumps({"tokens": {"access_token": "a", "refresh_token": "r"}}),
|
||||
encoding="utf-8",
|
||||
)
|
||||
monkeypatch.setenv("CODEX_HOME", str(codex_home))
|
||||
# Bypass the JWT-expiry check so a fake token doesn't short-circuit.
|
||||
monkeypatch.setattr(auth, "_codex_access_token_is_expiring", lambda *a, **k: False)
|
||||
|
||||
with mock.patch.object(Path, "read_text", wraps=Path.read_text) as spy:
|
||||
auth._import_codex_cli_tokens()
|
||||
|
||||
# _import_codex_cli_tokens reads exactly one file; assert that read
|
||||
# carried an explicit UTF-8 encoding. (The bound-method spy captures
|
||||
# kwargs but not the bound `self`, so we check the single read directly.)
|
||||
assert spy.call_count >= 1, "expected a read of the codex auth.json"
|
||||
for call in spy.call_args_list:
|
||||
assert "encoding" in call.kwargs, "codex read_text() must pass encoding"
|
||||
assert "utf-8" in str(call.kwargs["encoding"]).lower()
|
||||
|
||||
|
||||
# --- sibling readers of the same ~/.hermes/auth.json in other modules -------
|
||||
#
|
||||
# _load_auth_store lives in hermes_cli/auth.py, but several other modules read
|
||||
# the same ~/.hermes/auth.json directly. They had the same UTF-8-vs-cp1252
|
||||
# asymmetry on Windows — these tests pin the sibling reads too.
|
||||
|
||||
class TestAuthJsonSiblingReaders:
|
||||
def test_has_xai_credentials_reads_non_ascii_store(self, hermes_home, windows_default_encoding, monkeypatch):
|
||||
"""tools/xai_http.has_xai_credentials must read a non-ASCII auth.json.
|
||||
|
||||
Pre-fix, a UTF-8 store with a non-ASCII label raised UnicodeDecodeError
|
||||
here (cp1252 default on Windows), the broad except swallowed it, and
|
||||
xAI OAuth silently looked absent.
|
||||
"""
|
||||
# No XAI_API_KEY env → force the auth.json code path.
|
||||
monkeypatch.delenv("XAI_API_KEY", raising=False)
|
||||
store = {
|
||||
"version": auth.AUTH_STORE_VERSION,
|
||||
"providers": {
|
||||
"xai-oauth": {
|
||||
# CJK label → UTF-8 bytes (e.g. 0xE5..) that cp1252 cannot
|
||||
# decode, so a no-encoding read raises UnicodeDecodeError.
|
||||
"label": "工作账号",
|
||||
"tokens": {"access_token": "tok"},
|
||||
}
|
||||
},
|
||||
}
|
||||
_write_utf8(hermes_home / "auth.json", store)
|
||||
|
||||
from tools.xai_http import has_xai_credentials
|
||||
|
||||
assert has_xai_credentials() is True
|
||||
|
||||
def test_auxiliary_nous_provider_reads_non_ascii_store(self, hermes_home, windows_default_encoding, monkeypatch):
|
||||
"""agent/auxiliary_client's Nous-provider lookup reads the same store.
|
||||
|
||||
The lookup returns None on any read failure, silently disabling Nous as
|
||||
the auxiliary (vision/summarization) provider. A non-ASCII label must
|
||||
not trigger that path.
|
||||
"""
|
||||
store = {
|
||||
"version": auth.AUTH_STORE_VERSION,
|
||||
"active_provider": "nous",
|
||||
"providers": {"nous": {"agent_key": "k", "label": "工作账号"}},
|
||||
}
|
||||
_write_utf8(hermes_home / "auth.json", store)
|
||||
|
||||
import agent.auxiliary_client as aux
|
||||
|
||||
# _AUTH_JSON_PATH is resolved at module import time, so the
|
||||
# HERMES_HOME env from the fixture doesn't reach it — point it at
|
||||
# the tmp store explicitly.
|
||||
monkeypatch.setattr(aux, "_AUTH_JSON_PATH", hermes_home / "auth.json")
|
||||
|
||||
# _read_nous_auth consults the credential pool FIRST and returns early
|
||||
# when a pool entry exists, never reaching the auth.json read. Force the
|
||||
# pool-absent path so the auth.json code under test actually runs.
|
||||
monkeypatch.setattr(aux, "_select_pool_entry", lambda _provider: (False, None))
|
||||
|
||||
provider = aux._read_nous_auth()
|
||||
assert provider is not None
|
||||
assert provider.get("agent_key") == "k"
|
||||
|
||||
def test_read_shared_nous_state_reads_non_ascii_store(
|
||||
self, tmp_path, monkeypatch, windows_default_encoding
|
||||
):
|
||||
"""hermes_cli.auth._read_shared_nous_state must read a non-ASCII store.
|
||||
|
||||
The shared Nous store (``nous_auth.json``) is written as UTF-8. A
|
||||
non-ASCII field (e.g. an accented display name) must not cause the
|
||||
read to raise under the Windows-default-encoding fixture and be
|
||||
silently swallowed — which would drop the user's shared OAuth
|
||||
credentials and force a device-code re-login.
|
||||
"""
|
||||
# The shared-store path has a seat belt that refuses to resolve to the
|
||||
# real user's store under pytest; pin it to a tmp dir explicitly.
|
||||
shared_dir = tmp_path / "shared"
|
||||
monkeypatch.setenv("HERMES_SHARED_AUTH_DIR", str(shared_dir))
|
||||
|
||||
payload = {
|
||||
"refresh_token": "rt",
|
||||
"access_token": "at",
|
||||
# Non-ASCII display name → UTF-8 bytes cp1252 cannot decode.
|
||||
"display_name": "Réne — Noël",
|
||||
}
|
||||
_write_utf8(shared_dir / "nous_auth.json", payload)
|
||||
|
||||
provider = auth._read_shared_nous_state()
|
||||
assert provider is not None
|
||||
assert provider["access_token"] == "at"
|
||||
assert provider["refresh_token"] == "rt"
|
||||
# The non-ASCII field round-trips intact.
|
||||
assert provider["display_name"] == "Réne — Noël"
|
||||
|
||||
def test_has_any_provider_configured_reads_non_ascii_auth_store(
|
||||
self, hermes_home, monkeypatch, windows_default_encoding
|
||||
):
|
||||
"""hermes_cli.main._has_any_provider_configured reads auth.json.
|
||||
|
||||
When the active provider's auth.json store contains a non-ASCII
|
||||
label, the read must not raise under the Windows-default-encoding
|
||||
fixture (which would be swallowed and make a configured provider
|
||||
look absent, wrongly re-triggering the setup wizard).
|
||||
"""
|
||||
store = {
|
||||
"version": auth.AUTH_STORE_VERSION,
|
||||
"active_provider": "openai-codex",
|
||||
"providers": {
|
||||
"openai-codex": {
|
||||
"auth_mode": "chatgpt",
|
||||
"label": "工作账号", # non-ASCII CJK
|
||||
"tokens": {"access_token": "a", "refresh_token": "r"},
|
||||
}
|
||||
},
|
||||
}
|
||||
_write_utf8(hermes_home / "auth.json", store)
|
||||
|
||||
import hermes_cli.auth as auth_mod
|
||||
import hermes_cli.main as main_mod
|
||||
|
||||
# No provider env vars, no .env, and PROVIDER_REGISTRY lookups report
|
||||
# not-logged-in — so the function reaches the auth.json branch and the
|
||||
# result is driven by reading the (non-ASCII) store + the active
|
||||
# provider's status. The active provider IS logged in, so the UTF-8
|
||||
# read must succeed and surface True.
|
||||
def fake_status(provider_id=None):
|
||||
if provider_id == "openai-codex":
|
||||
return {"logged_in": True}
|
||||
return {"logged_in": False}
|
||||
|
||||
monkeypatch.setattr(auth_mod, "get_auth_status", fake_status)
|
||||
# _has_any_provider_configured imports get_auth_status lazily from
|
||||
# hermes_cli.auth; patch it on the source module so the local import
|
||||
# sees the fake.
|
||||
monkeypatch.delenv("OPENAI_API_KEY", raising=False)
|
||||
monkeypatch.delenv("ANTHROPIC_API_KEY", raising=False)
|
||||
monkeypatch.delenv("OPENROUTER_API_KEY", raising=False)
|
||||
monkeypatch.delenv("OPENAI_BASE_URL", raising=False)
|
||||
|
||||
assert main_mod._has_any_provider_configured() is True
|
||||
|
||||
def test_managed_tool_gateway_reads_non_ascii_nous_state(
|
||||
self, hermes_home, windows_default_encoding
|
||||
):
|
||||
"""tools.managed_tool_gateway._read_nous_provider_state reads auth.json.
|
||||
|
||||
The Nous provider entry can carry a non-ASCII label. Under the
|
||||
Windows-default-encoding fixture a no-encoding read raises and the
|
||||
broad except swallows it, returning None — so the gateway treats
|
||||
Nous as unconfigured.
|
||||
"""
|
||||
store = {
|
||||
"version": auth.AUTH_STORE_VERSION,
|
||||
"providers": {
|
||||
"nous": {
|
||||
"agent_key": "k",
|
||||
# Non-ASCII label → UTF-8 bytes cp1252 cannot decode.
|
||||
"label": "工作账号",
|
||||
}
|
||||
},
|
||||
}
|
||||
_write_utf8(hermes_home / "auth.json", store)
|
||||
|
||||
from tools.managed_tool_gateway import _read_nous_provider_state
|
||||
|
||||
nous = _read_nous_provider_state()
|
||||
assert nous is not None
|
||||
assert nous.get("agent_key") == "k"
|
||||
# The non-ASCII label round-trips intact.
|
||||
assert nous.get("label") == "工作账号"
|
||||
|
||||
Reference in New Issue
Block a user