From b868fbf371fd95c68adcca6f1ce84af67c776b26 Mon Sep 17 00:00:00 2001 From: Cody Mitchell Date: Tue, 29 Sep 2026 21:33:00 -0500 Subject: [PATCH] fix(arxiv): remove a rejected failed-conversion page before falling back When the cached copy of a paper is ar5iv's failed-conversion page and both ar5iv and the PDF mirror miss, the call falls through to the upstream server, which serves any cached Markdown file as it is, so the rejected page could still come back. The bridge now deletes the rejected cache entry before any fallback, and returns an error instead of forwarding when the file cannot be removed. Adds download_paper and read_paper tests with both fetch paths failing; both fail before this change. Same fix as psi-oss/get-physics-done#278 (found in review there). --- src/gpd/mcp/servers/arxiv_bridge.py | 10 ++++++++- tests/mcp/test_arxiv_bridge.py | 35 +++++++++++++++++++++++++++++ 2 files changed, 44 insertions(+), 1 deletion(-) diff --git a/src/gpd/mcp/servers/arxiv_bridge.py b/src/gpd/mcp/servers/arxiv_bridge.py index abd2c26b1..e6b91fe1c 100644 --- a/src/gpd/mcp/servers/arxiv_bridge.py +++ b/src/gpd/mcp/servers/arxiv_bridge.py @@ -689,7 +689,15 @@ async def _intercept_download( else: if _arxiv_ar5iv.is_conversion_failure(content): # An older bridge cached ar5iv's failed-conversion page. - logger.info("ignoring cached conversion failure for %s", paper_id) + # Remove it so no fallback, including the upstream + # server's own cache check, can serve it again. + try: + cache_path.unlink() + except OSError as exc: + return _tool_error( + f"Cached failed-conversion page for {paper_id} could not be removed: {exc}" + ) + logger.info("removed cached conversion failure for %s", paper_id) else: return _content_envelope( "cache", diff --git a/tests/mcp/test_arxiv_bridge.py b/tests/mcp/test_arxiv_bridge.py index 662590d74..ebb2e42a9 100644 --- a/tests/mcp/test_arxiv_bridge.py +++ b/tests/mcp/test_arxiv_bridge.py @@ -1691,3 +1691,38 @@ def test_topic_match_rules(query, paper, expected) -> None: from gpd.mcp.servers.arxiv_bridge import _topic_match assert _topic_match(query, paper) == expected + + +@pytest.mark.asyncio +@pytest.mark.parametrize("tool", ["download_paper", "read_paper"]) +async def test_rejected_cached_conversion_failure_is_removed_before_upstream_fallback( + monkeypatch: pytest.MonkeyPatch, tmp_path, tool +) -> None: + """When ar5iv and the PDF mirror both miss, the call falls through to the + upstream server, which serves any cached Markdown file as it is; the + rejected failure page must be gone by then.""" + from gpd.mcp.servers import _arxiv_ar5iv, _arxiv_gcs, _arxiv_token_bucket + from gpd.mcp.servers.arxiv_bridge import ArxivBridge, ArxivBridgeConfig + + _arxiv_token_bucket._reset_for_tests() + + async def no_sleep(_seconds: float) -> None: + return None + + monkeypatch.setattr(_arxiv_token_bucket.asyncio, "sleep", no_sleep) + monkeypatch.setattr(_arxiv_ar5iv, "fetch_html_content", lambda pid: None) + monkeypatch.setattr(_arxiv_gcs, "fetch_pdf_from_gcs", lambda pid: None) + stub = tmp_path / "hep-th_9711200.md" + stub.write_text("No content available\nConversion to HTML had a Fatal error and exited abruptly.", encoding="utf-8") + + fake, log = _make_fake_session(call_outputs=[("upstream answer", False)]) + bridge = ArxivBridge(ArxivBridgeConfig(storage_path=tmp_path, backend="hybrid")) + bridge._session = fake # type: ignore[assignment] + try: + result = await bridge.call_tool(tool, {"paper_id": "hep-th/9711200"}) + finally: + bridge._session = None + + assert not stub.exists() + assert [call[0] for call in log] == [tool] + assert "Fatal error" not in result.content[0].text