387 lines
14 KiB
Python
387 lines
14 KiB
Python
"""Tests for jobsource/agent_fallback.py — all network-free via monkeypatching."""
|
|
from __future__ import annotations
|
|
|
|
import pytest
|
|
|
|
import jobsource.agent_fallback as _mod
|
|
from jobsource.agent_fallback import (
|
|
AgentFallbackResult,
|
|
_AgentOutput,
|
|
_site_key,
|
|
_valid_url,
|
|
find_and_extract,
|
|
)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Helpers and fakes
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def _placeholder_settings(**overrides):
|
|
"""Return a fake settings object with placeholder values (agent unavailable)."""
|
|
defaults = {
|
|
"enable_browser_agent": True,
|
|
"agent_model": "PLACEHOLDER_AGENT_MODEL",
|
|
"llm_api_key": "PLACEHOLDER_LLM_API_KEY",
|
|
"browser_headless": True,
|
|
}
|
|
defaults.update(overrides)
|
|
return type("S", (), defaults)()
|
|
|
|
|
|
def _real_settings(**overrides):
|
|
"""Return a fake settings object with non-placeholder values."""
|
|
defaults = {
|
|
"enable_browser_agent": True,
|
|
"agent_model": "anthropic:claude-sonnet-4-6",
|
|
"llm_api_key": "sk-ant-real",
|
|
"browser_headless": True,
|
|
}
|
|
defaults.update(overrides)
|
|
return type("S", (), defaults)()
|
|
|
|
|
|
def _patch_settings(monkeypatch, settings_obj):
|
|
monkeypatch.setattr(_mod, "get_settings", lambda: settings_obj)
|
|
|
|
|
|
def _patch_chat(monkeypatch, chat_obj=None):
|
|
"""Monkeypatch _build_chat_model to return chat_obj (default sentinel)."""
|
|
chat_obj = chat_obj or object()
|
|
monkeypatch.setattr(_mod, "_build_chat_model", lambda: chat_obj)
|
|
return chat_obj
|
|
|
|
|
|
def _patch_session(monkeypatch, output: _AgentOutput | None):
|
|
"""Monkeypatch _agent_session so no real browser launches."""
|
|
async def _fake_session(website, chat):
|
|
return output
|
|
monkeypatch.setattr(_mod, "_agent_session", _fake_session)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# _site_key
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestSiteKey:
|
|
def test_strips_www(self):
|
|
assert _site_key("https://www.acme.com/foo") == "acme.com"
|
|
|
|
def test_lowercases(self):
|
|
assert _site_key("https://ACME.COM") == "acme.com"
|
|
|
|
def test_no_www(self):
|
|
assert _site_key("https://acme.com/careers") == "acme.com"
|
|
|
|
def test_subdomain_preserved(self):
|
|
assert _site_key("https://jobs.acme.com") == "jobs.acme.com"
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# _valid_url
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestValidUrl:
|
|
def test_https_url_passes(self):
|
|
assert _valid_url("https://acme.com/jobs/1") == "https://acme.com/jobs/1"
|
|
|
|
def test_http_url_passes(self):
|
|
assert _valid_url("http://acme.com/jobs/1") == "http://acme.com/jobs/1"
|
|
|
|
def test_none_returns_none(self):
|
|
assert _valid_url(None) is None
|
|
|
|
def test_junk_string_returns_none(self):
|
|
assert _valid_url("null") is None
|
|
|
|
def test_relative_url_returns_none(self):
|
|
assert _valid_url("/jobs/1") is None
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Gate: enable_browser_agent=False
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestFeatureFlagGate:
|
|
def test_returns_none_when_flag_off(self, monkeypatch):
|
|
_patch_settings(monkeypatch, _placeholder_settings(enable_browser_agent=False))
|
|
# Clear memo so there's no cached entry from another test.
|
|
_mod._MEMO.clear()
|
|
result = find_and_extract("https://acme.com")
|
|
assert result is None
|
|
|
|
def test_does_not_cache_when_flag_off(self, monkeypatch):
|
|
_patch_settings(monkeypatch, _placeholder_settings(enable_browser_agent=False))
|
|
_mod._MEMO.clear()
|
|
find_and_extract("https://flagoff.com")
|
|
assert "flagoff.com" not in _mod._MEMO
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Gate: placeholder LLM config
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestPlaceholderGate:
|
|
def test_returns_none_for_placeholder_model(self, monkeypatch):
|
|
_patch_settings(monkeypatch, _placeholder_settings())
|
|
_mod._MEMO.clear()
|
|
result = find_and_extract("https://acme.com")
|
|
assert result is None
|
|
|
|
def test_returns_none_for_placeholder_key_only(self, monkeypatch):
|
|
_patch_settings(monkeypatch, _real_settings(llm_api_key="PLACEHOLDER_LLM_API_KEY"))
|
|
_mod._MEMO.clear()
|
|
result = find_and_extract("https://acme.com")
|
|
assert result is None
|
|
|
|
def test_returns_none_for_placeholder_model_only(self, monkeypatch):
|
|
_patch_settings(monkeypatch, _real_settings(agent_model="PLACEHOLDER_AGENT_MODEL"))
|
|
_mod._MEMO.clear()
|
|
result = find_and_extract("https://acme.com")
|
|
assert result is None
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Gate: _build_chat_model raises (missing package / bad config)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestChatModelBuildFailure:
|
|
def test_returns_none_and_memos_none_on_import_error(self, monkeypatch):
|
|
_patch_settings(monkeypatch, _real_settings())
|
|
_mod._MEMO.clear()
|
|
|
|
def _raise():
|
|
raise ImportError("browser_use not installed")
|
|
|
|
monkeypatch.setattr(_mod, "_build_chat_model", _raise)
|
|
result = find_and_extract("https://chatfail.com")
|
|
assert result is None
|
|
assert _mod._MEMO.get("chatfail.com") is None
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Gate: _agent_session raises / returns None
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestAgentSessionFailure:
|
|
def test_returns_none_on_session_exception(self, monkeypatch):
|
|
_patch_settings(monkeypatch, _real_settings())
|
|
_mod._MEMO.clear()
|
|
_patch_chat(monkeypatch)
|
|
|
|
async def _raise_session(website, chat):
|
|
raise RuntimeError("Chromium unavailable")
|
|
|
|
monkeypatch.setattr(_mod, "_agent_session", _raise_session)
|
|
result = find_and_extract("https://chromiumfail.com")
|
|
assert result is None
|
|
|
|
def test_returns_none_when_session_returns_none(self, monkeypatch):
|
|
_patch_settings(monkeypatch, _real_settings())
|
|
_mod._MEMO.clear()
|
|
_patch_chat(monkeypatch)
|
|
_patch_session(monkeypatch, None)
|
|
result = find_and_extract("https://nonefail.com")
|
|
assert result is None
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Success path: structured output parsed into AgentFallbackResult
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestSuccessPath:
|
|
def test_both_urls_returned(self, monkeypatch):
|
|
_patch_settings(monkeypatch, _real_settings())
|
|
_mod._MEMO.clear()
|
|
_patch_chat(monkeypatch)
|
|
_patch_session(
|
|
monkeypatch,
|
|
_AgentOutput(
|
|
careers_url="https://acme.com/careers",
|
|
position_url="https://acme.com/careers/jobs/42-swe",
|
|
),
|
|
)
|
|
result = find_and_extract("https://acme.com")
|
|
assert result is not None
|
|
assert result.careers_url == "https://acme.com/careers"
|
|
assert result.position_url == "https://acme.com/careers/jobs/42-swe"
|
|
|
|
def test_junk_careers_url_filtered(self, monkeypatch):
|
|
_patch_settings(monkeypatch, _real_settings())
|
|
_mod._MEMO.clear()
|
|
_patch_chat(monkeypatch)
|
|
_patch_session(
|
|
monkeypatch,
|
|
_AgentOutput(careers_url="null", position_url="https://acme.com/jobs/1"),
|
|
)
|
|
result = find_and_extract("https://acme.com")
|
|
assert result is not None
|
|
assert result.careers_url is None
|
|
assert result.position_url == "https://acme.com/jobs/1"
|
|
|
|
def test_result_stored_in_memo(self, monkeypatch):
|
|
_patch_settings(monkeypatch, _real_settings())
|
|
_mod._MEMO.clear()
|
|
_patch_chat(monkeypatch)
|
|
_patch_session(
|
|
monkeypatch,
|
|
_AgentOutput(
|
|
careers_url="https://memo-test.com/careers",
|
|
position_url="https://memo-test.com/jobs/1",
|
|
),
|
|
)
|
|
find_and_extract("https://memo-test.com")
|
|
cached = _mod._MEMO.get("memo-test.com")
|
|
assert cached is not None
|
|
assert cached.careers_url == "https://memo-test.com/careers"
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Memo: same-domain second call uses cache, not a second session
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestMemo:
|
|
def test_second_call_uses_cache(self, monkeypatch):
|
|
_patch_settings(monkeypatch, _real_settings())
|
|
_mod._MEMO.clear()
|
|
_patch_chat(monkeypatch)
|
|
|
|
call_count = 0
|
|
|
|
async def _counting_session(website, chat):
|
|
nonlocal call_count
|
|
call_count += 1
|
|
return _AgentOutput(
|
|
careers_url="https://sharetest.com/careers",
|
|
position_url="https://sharetest.com/jobs/1",
|
|
)
|
|
|
|
monkeypatch.setattr(_mod, "_agent_session", _counting_session)
|
|
|
|
r1 = find_and_extract("https://sharetest.com")
|
|
r2 = find_and_extract("https://sharetest.com/careers") # same host, careers path
|
|
|
|
assert call_count == 1, "Expected exactly one browser session for the same host"
|
|
assert r1 == r2
|
|
|
|
def test_memo_none_prevents_retry(self, monkeypatch):
|
|
_patch_settings(monkeypatch, _real_settings())
|
|
_mod._MEMO.clear()
|
|
_patch_chat(monkeypatch)
|
|
|
|
call_count = 0
|
|
|
|
async def _failing_session(website, chat):
|
|
nonlocal call_count
|
|
call_count += 1
|
|
raise RuntimeError("broken")
|
|
|
|
monkeypatch.setattr(_mod, "_agent_session", _failing_session)
|
|
|
|
find_and_extract("https://memonil.com")
|
|
find_and_extract("https://memonil.com") # should use cached None
|
|
|
|
assert call_count == 1
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Cascade/extract wiring integration (lightweight, no real browser)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
class TestCascadeWiring:
|
|
"""Verify find_and_extract is reachable from cascade.find_careers_page."""
|
|
|
|
def test_cascade_tier6_returns_browser_agent_method(self, monkeypatch):
|
|
from jobsource.careers.cascade import find_careers_page, CareersResult
|
|
|
|
# Patch all cheaper tiers to miss.
|
|
monkeypatch.setattr(
|
|
"jobsource.careers.cascade._ats.detect_and_fetch", lambda *a, **kw: None
|
|
)
|
|
monkeypatch.setattr(
|
|
"jobsource.careers.cascade._ats.recover_via_slug_guess", lambda *a, **kw: None
|
|
)
|
|
monkeypatch.setattr(
|
|
"jobsource.careers.cascade._heuristics.probe_url_patterns", lambda *a, **kw: None
|
|
)
|
|
monkeypatch.setattr(
|
|
"jobsource.careers.cascade._heuristics.scan_homepage_links", lambda *a, **kw: None
|
|
)
|
|
monkeypatch.setattr(
|
|
"jobsource.careers.cascade._heuristics.parse_sitemap", lambda *a, **kw: None
|
|
)
|
|
monkeypatch.setattr(
|
|
"jobsource.careers.cascade._classify_llm.extract_anchors", lambda *a, **kw: []
|
|
)
|
|
monkeypatch.setattr(
|
|
"jobsource.careers.cascade._classify_llm.classify_careers_link",
|
|
lambda *a, **kw: None,
|
|
)
|
|
# Patch the safe homepage fetch.
|
|
from jobsource.careers import cascade as _cascade
|
|
monkeypatch.setattr(_cascade, "_safe_get_html", lambda *a, **kw: None)
|
|
|
|
# Patch agent_fallback.
|
|
_mod._MEMO.clear()
|
|
monkeypatch.setattr(
|
|
"jobsource.careers.cascade._agent_fallback.find_and_extract",
|
|
lambda *a, **kw: AgentFallbackResult(
|
|
careers_url="https://wired.com/careers",
|
|
position_url="https://wired.com/jobs/1",
|
|
),
|
|
)
|
|
|
|
result = find_careers_page("https://wired.com")
|
|
assert result.method == "browser_agent"
|
|
assert result.careers_url == "https://wired.com/careers"
|
|
assert result.position_url == "https://wired.com/jobs/1"
|
|
assert result.confidence == 0.50
|
|
|
|
|
|
class TestExtractWiring:
|
|
"""Verify find_and_extract is reachable from extract.extract_open_position."""
|
|
|
|
def test_extract_tier5_returns_browser_agent_method(self, monkeypatch):
|
|
from jobsource.extract import extract_open_position
|
|
|
|
# Patch cheaper tiers to miss (ATS, JSON-LD, anchors, LLM).
|
|
monkeypatch.setattr(
|
|
"jobsource.extract._ats.detect_ats_in_url", lambda *a, **kw: None
|
|
)
|
|
monkeypatch.setattr(
|
|
"jobsource.extract._ats.detect_ats_in_html", lambda *a, **kw: None
|
|
)
|
|
|
|
import httpx
|
|
fake_response = type(
|
|
"R",
|
|
(),
|
|
{"status_code": 200, "text": "<html><body></body></html>"},
|
|
)()
|
|
|
|
monkeypatch.setattr(
|
|
"jobsource.extract.request_with_retries",
|
|
lambda *a, **kw: fake_response,
|
|
)
|
|
monkeypatch.setattr(
|
|
"jobsource.extract.extract_anchors", lambda *a, **kw: []
|
|
)
|
|
monkeypatch.setattr(
|
|
"jobsource.extract.classify_job_link", lambda *a, **kw: None
|
|
)
|
|
|
|
# Patch agent_fallback.
|
|
_mod._MEMO.clear()
|
|
monkeypatch.setattr(
|
|
"jobsource.extract._agent_fallback.find_and_extract",
|
|
lambda *a, **kw: AgentFallbackResult(
|
|
careers_url="https://extract-wired.com/careers",
|
|
position_url="https://extract-wired.com/jobs/99",
|
|
),
|
|
)
|
|
|
|
url, method = extract_open_position("https://extract-wired.com/careers")
|
|
assert method == "browser_agent"
|
|
assert url == "https://extract-wired.com/jobs/99"
|