Skip to content
Open
Show file tree
Hide file tree
Changes from 1 commit
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 13 additions & 0 deletions src/gpd/mcp/servers/_arxiv_ar5iv.py
Original file line number Diff line number Diff line change
Expand Up @@ -75,6 +75,16 @@ def _html_to_text(html: str) -> str:
return parser.get_text()


# ar5iv serves this page with HTTP 200 when LaTeXML failed on a paper (for
# example hep-th/9711200). It is not the paper, so callers fall through to PDF.
_CONVERSION_FAILED_MARKER = "Conversion to HTML had a Fatal error"


def is_conversion_failure(text: str) -> bool:
"""True when ``text`` is ar5iv's failed-conversion page, not a paper."""
return _CONVERSION_FAILED_MARKER in text


def fetch_html_content(paper_id: str) -> str | None:
# ar5iv first (follow_redirects=False so 307 = miss, not a hop into
# arxiv.org). arxiv.org/html as fallback for papers ar5iv lacks.
Expand Down Expand Up @@ -114,6 +124,9 @@ def fetch_html_content(paper_id: str) -> str | None:
continue

text = _html_to_text(body.decode("utf-8", errors="replace"))
if is_conversion_failure(text):
logger.info("html-%s conversion failed for %s; treating as miss", label, paper_id)
continue
if text.strip():
return text
except Exception:
Expand Down
21 changes: 14 additions & 7 deletions src/gpd/mcp/servers/arxiv_bridge.py
Original file line number Diff line number Diff line change
Expand Up @@ -421,13 +421,17 @@ async def _intercept_download(
except OSError as exc:
logger.warning("cache read failed %s: %s", cache_path, exc)
else:
return _content_envelope(
"cache",
"Paper already available (returned from cache)",
paper_id,
content,
cache_path,
)
if _arxiv_ar5iv.is_conversion_failure(content):
# An older bridge cached ar5iv's failed-conversion page.
logger.info("ignoring cached conversion failure for %s", paper_id)
Comment thread
coderabbitai[bot] marked this conversation as resolved.
Outdated
else:
return _content_envelope(
"cache",
"Paper already available (returned from cache)",
paper_id,
content,
cache_path,
)

html = await asyncio.to_thread(_arxiv_ar5iv.fetch_html_content, paper_id)
if html is not None:
Expand Down Expand Up @@ -507,6 +511,9 @@ async def _intercept_read_paper(
except OSError as exc:
logger.warning("read_paper cache read failed %s: %s", cache_path, exc)
return None
if _arxiv_ar5iv.is_conversion_failure(content):
# A cached failed-conversion page is not the paper: fetch it again.
return await self._intercept_download(args)
return _content_envelope(
"cache", "Paper read from local cache", paper_id, content, cache_path
)
Expand Down
12 changes: 12 additions & 0 deletions tests/mcp/test_arxiv_ar5iv.py
Original file line number Diff line number Diff line change
Expand Up @@ -122,3 +122,15 @@ def test_fetch_treats_empty_extracted_text_as_miss(
result = _arxiv_ar5iv.fetch_html_content("2401.00001")
assert result is not None
assert "real" in result


def test_fetch_treats_conversion_failure_page_as_miss(monkeypatch: pytest.MonkeyPatch) -> None:
"""ar5iv answers 200 with a failed-conversion page for some papers (for
example hep-th/9711200); that page is not the paper."""
stub = (
b"<html><body><p>No content available</p>"
b"<p>Conversion to HTML had a Fatal error and exited abruptly.</p></body></html>"
)
fake = _make_fake_client([_FakeResponse(200, content=stub, text=stub.decode()), _FakeResponse(404)])
monkeypatch.setattr(_arxiv_ar5iv.httpx, "Client", lambda *a, **k: fake)
assert _arxiv_ar5iv.fetch_html_content("hep-th/9711200") is None
73 changes: 73 additions & 0 deletions tests/mcp/test_arxiv_bridge.py
Original file line number Diff line number Diff line change
Expand Up @@ -1128,3 +1128,76 @@ def test_resolve_backend_rejects_garbage(monkeypatch: pytest.MonkeyPatch) -> Non

monkeypatch.setenv("GPD_ARXIV_BACKEND", "potato")
assert _resolve_backend() == "hybrid"


@pytest.mark.asyncio
async def test_download_paper_ignores_cached_conversion_failure(
monkeypatch: pytest.MonkeyPatch, tmp_path
) -> None:
"""An older bridge cached ar5iv's failed-conversion page as the paper;
download_paper must fetch the paper again instead of serving it."""
import json as _json

from gpd.mcp.servers import _arxiv_ar5iv, _arxiv_gcs, _arxiv_token_bucket
from gpd.mcp.servers.arxiv_bridge import ArxivBridge, ArxivBridgeConfig

_arxiv_token_bucket._reset_for_tests()

async def no_sleep(_seconds: float) -> None:
return None

monkeypatch.setattr(_arxiv_token_bucket.asyncio, "sleep", no_sleep)
monkeypatch.setattr(_arxiv_ar5iv, "fetch_html_content", lambda pid: None)
monkeypatch.setattr(_arxiv_gcs, "fetch_pdf_from_gcs", lambda pid: b"fake-pdf-bytes")
monkeypatch.setattr(_arxiv_gcs, "pdf_bytes_to_markdown", lambda pdf, pid, storage: "# AdS/CFT\n\nbody")
stub = "No content available\nConversion to HTML had a Fatal error and exited abruptly."
(tmp_path / "hep-th_9711200.md").write_text(stub, encoding="utf-8")

fake, log = _make_fake_session()
bridge = ArxivBridge(ArxivBridgeConfig(storage_path=tmp_path, backend="hybrid"))
bridge._session = fake # type: ignore[assignment]
try:
result = await bridge.call_tool("download_paper", {"paper_id": "hep-th/9711200"})
finally:
bridge._session = None

payload = _json.loads(result.content[0].text)
assert payload["source"] == "pdf-gcs"
assert "# AdS/CFT" in payload["content"]
assert "Fatal error" not in (tmp_path / "hep-th_9711200.md").read_text(encoding="utf-8")


@pytest.mark.asyncio
async def test_read_paper_refetches_cached_conversion_failure(
monkeypatch: pytest.MonkeyPatch, tmp_path
) -> None:
import json as _json

from gpd.mcp.servers import _arxiv_ar5iv, _arxiv_gcs, _arxiv_token_bucket
from gpd.mcp.servers.arxiv_bridge import ArxivBridge, ArxivBridgeConfig

_arxiv_token_bucket._reset_for_tests()

async def no_sleep(_seconds: float) -> None:
return None

monkeypatch.setattr(_arxiv_token_bucket.asyncio, "sleep", no_sleep)
monkeypatch.setattr(_arxiv_ar5iv, "fetch_html_content", lambda pid: None)
monkeypatch.setattr(_arxiv_gcs, "fetch_pdf_from_gcs", lambda pid: b"fake-pdf-bytes")
monkeypatch.setattr(_arxiv_gcs, "pdf_bytes_to_markdown", lambda pdf, pid, storage: "# AdS/CFT\n\nbody")
(tmp_path / "hep-th_9711200.md").write_text(
"No content available\nConversion to HTML had a Fatal error and exited abruptly.", encoding="utf-8"
)

fake, log = _make_fake_session()
bridge = ArxivBridge(ArxivBridgeConfig(storage_path=tmp_path, backend="hybrid"))
bridge._session = fake # type: ignore[assignment]
try:
result = await bridge.call_tool("read_paper", {"paper_id": "hep-th/9711200"})
finally:
bridge._session = None

assert log == []
payload = _json.loads(result.content[0].text)
assert payload["source"] == "pdf-gcs"
assert "# AdS/CFT" in payload["content"]
Loading