diff --git a/apps/api/app/api/v1/routes/retrieval.py b/apps/api/app/api/v1/routes/retrieval.py
index 087b8639..994f53b2 100644
--- a/apps/api/app/api/v1/routes/retrieval.py
+++ b/apps/api/app/api/v1/routes/retrieval.py
@@ -154,9 +154,13 @@ class RetrievalQueryResponse(BaseModel):
namespace: str
query: str
router_used: str
+ evidence: list[dict] = Field(
+ default_factory=list,
+ description="Composed evidence parts (text and inline images) for downstream agents.",
+ )
evidence_text: str = Field(
default="",
- description="Hierarchical evidence text. Primary output for downstream agents.",
+ description="Text projection of evidence. Tables stay as HTML; images are data URLs.",
)
answer_text: str = Field(
default="",
@@ -166,7 +170,10 @@ class RetrievalQueryResponse(BaseModel):
),
)
referenced_chunks: list[dict] = Field(default_factory=list)
- results: list[dict] = Field(default_factory=list)
+ results: list[dict] = Field(
+ default_factory=list,
+ description="Raw path chunks for debug. Content keeps placeholders; composed parts live on evidence.",
+ )
stop_reason: str | None = None
failure_reason: str | None = None
decision_trace: list[dict] | None = Field(
diff --git a/apps/api/app/mcp/retrieval_server.py b/apps/api/app/mcp/retrieval_server.py
index 89845b64..da806746 100644
--- a/apps/api/app/mcp/retrieval_server.py
+++ b/apps/api/app/mcp/retrieval_server.py
@@ -54,16 +54,20 @@ def resolve_mcp_namespace(*, ctx: Context | None) -> str:
def to_mcp_query_response(response: dict[str, Any]) -> dict[str, Any]:
- """Project the internal retrieval response to the MCP agent contract.
+ """Project the public retrieval response to the MCP agent contract.
- MCP returns exactly 3 PRIMARY fields:
- - evidence_text: hierarchical evidence tree for LLM consumption
+ MCP returns the same Knowhere package the HTTP query emits:
+ - evidence: composed parts (text/HTML and inline images)
+ - evidence_text: text projection of those parts
+ - results: raw path chunks for debug
- referenced_chunks: structured chunk references for citation / follow-up
- decision_trace: navigation decisions including terminal stop/failure
"""
return {
"query": response.get("query"),
+ "evidence": response.get("evidence") or [],
"evidence_text": response.get("evidence_text") or "",
+ "results": response.get("results") or [],
"referenced_chunks": response.get("referenced_chunks") or [],
"decision_trace": response.get("decision_trace") or [],
}
@@ -101,10 +105,12 @@ def create_retrieval_mcp_server(
@server.tool(
name="retrieval.query",
description=(
- "Search published documents. Returns evidence_text (hierarchical "
- "evidence for LLM consumption), referenced_chunks (cited chunk "
- "metadata for follow-up queries), and decision_trace (navigation "
- "decisions including stop/failure reasons). "
+ "Search published documents. Compose already happened in Knowhere. "
+ "Returns evidence (composed parts for LLM consumption), "
+ "results (raw path chunks for debug), "
+ "evidence_text (text projection of evidence), "
+ "referenced_chunks (cited chunk metadata for follow-up queries), "
+ "and decision_trace (navigation decisions including stop/failure reasons). "
"Include navigation intent directly in your query text — the "
"engine will automatically locate the right documents and sections."
),
diff --git a/apps/api/tests/contract/test_evidence_renderer_contract.py b/apps/api/tests/contract/test_evidence_renderer_contract.py
index a714ff54..da2cac7b 100644
--- a/apps/api/tests/contract/test_evidence_renderer_contract.py
+++ b/apps/api/tests/contract/test_evidence_renderer_contract.py
@@ -1,46 +1,91 @@
-from shared.services.retrieval.execution.routes import _render_rows_evidence
+import asyncio
+from app.mcp.retrieval_server import to_mcp_query_response
+from shared.services.retrieval.execution.response_projection import (
+ project_public_retrieval_response,
+)
+from shared.services.retrieval.execution.routes import _evidence_fields
-def test_render_rows_evidence_should_group_siblings_under_parent_path() -> None:
+
+def test_evidence_fields_flatten_composed_parts_in_result_order() -> None:
rows = [
+ {"composed": [{"type": "text", "text": "first"}]},
{
- "chunk_id": "c2",
- "content": "second section content",
- "sort_order": 2,
- "source": {
- "source_file_name": "alpha.pdf",
- "section_path": "Alpha / Two",
- },
- },
- {
- "chunk_id": "c1",
- "content": "first section content\nwith more detail",
- "sort_order": 1,
- "source": {
- "source_file_name": "alpha.pdf",
- "section_path": "Alpha / One",
- },
- },
- {
- "chunk_id": "c3",
- "content": "
",
- "source_file_name": "beta.pdf",
- "section_path": "Beta / Table",
+ "composed": [
+ {"type": "text", "text": "mid "},
+ {"type": "image", "media_type": "image/png", "data": "abc"},
+ ]
},
+ {"composed": [{"type": "text", "text": ""}]},
+ ]
+
+ fields = _evidence_fields(rows)
+
+ assert fields["evidence"] == [
+ {"type": "text", "text": "first"},
+ {"type": "text", "text": "mid "},
+ {"type": "image", "media_type": "image/png", "data": "abc"},
+ {"type": "text", "text": ""},
]
+ assert fields["evidence_text"] == (
+ "firstmid data:image/png;base64,abc"
+ )
+
+
+def test_mcp_query_response_keeps_evidence_and_debug_results() -> None:
+ response = to_mcp_query_response(
+ {
+ "query": "q",
+ "evidence": [{"type": "text", "text": "t"}],
+ "evidence_text": "t",
+ "results": [
+ {
+ "content": "[images/a.png]",
+ }
+ ],
+ "referenced_chunks": [{"chunk_id": "c1"}],
+ "decision_trace": [{"step": 1}],
+ "answer_text": "should drop",
+ }
+ )
+
+ assert response == {
+ "query": "q",
+ "evidence": [{"type": "text", "text": "t"}],
+ "evidence_text": "t",
+ "results": [
+ {
+ "content": "[images/a.png]",
+ }
+ ],
+ "referenced_chunks": [{"chunk_id": "c1"}],
+ "decision_trace": [{"step": 1}],
+ }
+
+
+def test_public_results_keep_placeholders_without_composed() -> None:
+ public = asyncio.run(
+ project_public_retrieval_response(
+ {
+ "namespace": "default",
+ "query": "q",
+ "router_used": "classic",
+ "evidence": [{"type": "text", "text": ""}],
+ "evidence_text": "",
+ "results": [
+ {
+ "chunk_id": "c1",
+ "chunk_type": "text",
+ "content": "见表 [tables/a.html]",
+ "composed": [{"type": "text", "text": "见表 "}],
+ "score": 1,
+ "document_id": "d1",
+ }
+ ],
+ }
+ )
+ )
- evidence_text = _render_rows_evidence(rows)
-
- assert evidence_text.count("[E1]") == 1
- assert evidence_text.count("[E2]") == 1
- assert "[E3]" not in evidence_text
- assert "[§ alpha.pdf / Alpha]" in evidence_text
- assert "[§ beta.pdf / Beta]" in evidence_text
- assert "[§ alpha.pdf / Alpha / One]" not in evidence_text
- assert "[§ alpha.pdf / Alpha / Two]" not in evidence_text
- assert "first section content" in evidence_text
- assert "second section content" in evidence_text
- assert "" in evidence_text
- assert "[Document]" not in evidence_text
- assert "▸" not in evidence_text
- assert "┈" not in evidence_text
+ assert public["evidence"] == [{"type": "text", "text": ""}]
+ assert public["results"][0]["content"] == "见表 [tables/a.html]"
+ assert "composed" not in public["results"][0]
diff --git a/apps/api/tests/contract/test_retrieval_document_scope.py b/apps/api/tests/contract/test_retrieval_document_scope.py
index 3f15c1af..9329414d 100644
--- a/apps/api/tests/contract/test_retrieval_document_scope.py
+++ b/apps/api/tests/contract/test_retrieval_document_scope.py
@@ -387,7 +387,8 @@ async def test_connected_hydration_and_assembly_reject_outside_rows(
)
assert {r["document_id"] for r in assembled} == expected
for row in assembled:
- assert "asset secret" in row["content"]
+ # content stays raw for debug; compose keeps the placeholder intact.
+ assert "[images/" in row["content"]
def test_cache_scope_none_empty_and_set_identity():
diff --git a/apps/worker/tests/contract/test_page_memory_retrieval_contract.py b/apps/worker/tests/contract/test_page_memory_retrieval_contract.py
index e9346c7e..bfebab1d 100644
--- a/apps/worker/tests/contract/test_page_memory_retrieval_contract.py
+++ b/apps/worker/tests/contract/test_page_memory_retrieval_contract.py
@@ -142,13 +142,15 @@ async def test_table_result_assembly_uses_summary_not_html() -> None:
assert len(assembled) == 1
content = assembled[0]["content"]
- assert "[Table: https://assets.example.com/job-1/tables/table-1.html]" in content
- assert "企业入驻信息登记模板" in content
- assert "企业名称;统一社会信用代码" in content
- assert "SHOULD NOT LEAK" not in content
- assert "" in composed_text
+ assert "[tables/" not in composed_text
+ assert composed_text.index("见表") < composed_text.index(" str:
- """Render one evidence block via the shared evidence renderer."""
+ """Render one evidence block as ``[E#]`` + path + bodies."""
bodies = [_chunk_body(child.chunk) for child in selected]
- return render_evidence_blocks(
+ return _render_evidence_blocks(
[(group.parent_title or "", bodies)],
start_index=evidence_index,
)
+def _render_evidence_blocks(
+ groups: Sequence[tuple[str, Sequence[str]]],
+ *,
+ start_index: int = 1,
+) -> str:
+ parts: list[str] = []
+ index = max(1, int(start_index or 1))
+ for path, bodies in groups:
+ texts = [str(t or "").strip() for t in bodies]
+ texts = [t for t in texts if t]
+ if not texts:
+ continue
+ block: list[str] = [f"[E{index + len(parts)}]"]
+ header = str(path or "").strip()
+ if header:
+ block.append(f"[§ {header}]")
+ indent = len(texts) >= 2
+ for text in texts:
+ if indent:
+ block.append(
+ "\n".join(
+ (" " + ln if ln.strip() else ln) for ln in text.splitlines()
+ )
+ )
+ else:
+ block.append(text)
+ parts.append("\n".join(block).strip())
+ return "\n\n".join(parts)
+
+
def _scored_flat(groups: Sequence[_ParentGroup]) -> List[Tuple[Chunk, float]]:
out: List[Tuple[Chunk, float]] = []
for g in groups:
diff --git a/packages/shared-python/shared/services/retrieval/agent_tools/tools/read.py b/packages/shared-python/shared/services/retrieval/agent_tools/tools/read.py
index d219d1a8..ea7f946a 100644
--- a/packages/shared-python/shared/services/retrieval/agent_tools/tools/read.py
+++ b/packages/shared-python/shared/services/retrieval/agent_tools/tools/read.py
@@ -1,8 +1,8 @@
"""``corpus.read`` — full body content for already-located sections/chunks.
Unlike ``hydration.result_assembly.assemble_retrieval_results`` (which
-down-weights ``page`` chunks to their summary — see that module's
-``_page_summary``, a deliberate trade-off for the retrieval-answer surface),
+down-weights ``page`` chunks to their summary — see ``page_summary``, a
+deliberate trade-off for the retrieval-answer surface),
``read`` returns the page chunk's full body content, with ``[SAME-AS
p]`` markers resolved to the owner section's text (§2 of
``CORPUS_SCHEMA.md``) rather than stripped or summarized. ``connect_to``
diff --git a/packages/shared-python/shared/services/retrieval/execution/plan.py b/packages/shared-python/shared/services/retrieval/execution/plan.py
index 73a0f4be..0baad9a4 100644
--- a/packages/shared-python/shared/services/retrieval/execution/plan.py
+++ b/packages/shared-python/shared/services/retrieval/execution/plan.py
@@ -123,6 +123,7 @@ async def _execute_with_overrides(self, request: RetrievalQuery) -> dict[str, An
"namespace": request.namespace,
"query": request.query,
"router_used": "empty_query_filtered",
+ "evidence": [],
"evidence_text": "",
"answer_text": "",
"referenced_chunks": [],
diff --git a/packages/shared-python/shared/services/retrieval/execution/response_projection.py b/packages/shared-python/shared/services/retrieval/execution/response_projection.py
index 1ac29076..8878d82c 100644
--- a/packages/shared-python/shared/services/retrieval/execution/response_projection.py
+++ b/packages/shared-python/shared/services/retrieval/execution/response_projection.py
@@ -25,6 +25,7 @@ async def project_public_retrieval_response(response: dict[str, Any]) -> dict[st
'namespace': response.get('namespace'),
'query': response.get('query'),
'router_used': response.get('router_used'),
+ 'evidence': response.get('evidence') or [],
'evidence_text': response.get('evidence_text') or '',
'answer_text': '',
'referenced_chunks': response.get('referenced_chunks') or [],
diff --git a/packages/shared-python/shared/services/retrieval/execution/routes.py b/packages/shared-python/shared/services/retrieval/execution/routes.py
index 25eaac5a..4f206196 100644
--- a/packages/shared-python/shared/services/retrieval/execution/routes.py
+++ b/packages/shared-python/shared/services/retrieval/execution/routes.py
@@ -13,11 +13,13 @@
RetrievalRouteContext,
RetrievalRouteOutcome,
)
-from shared.services.retrieval.hydration.evidence_text import render_evidence_blocks
+from shared.services.retrieval.hydration.evidence_compose import (
+ collect_evidence,
+ flatten_parts,
+)
from shared.services.retrieval.hydration.result_assembly import (
assemble_retrieval_results,
)
-from shared.services.retrieval.search.lexical_text import split_section_path
from shared.services.retrieval.search.map_unit_discovery import map_unit_discovery
from shared.services.retrieval.search.ranking import rank_retrieval_candidates
from shared.services.retrieval.search.scoped_corpus import (
@@ -47,29 +49,14 @@ def open_agent_explore_database_context() -> AbstractAsyncContextManager[AsyncSe
)
-def _evidence_path_header(row: dict) -> str:
- source = row.get("source")
- if not isinstance(source, dict):
- source = row
- file_name = str(source.get("source_file_name") or "").strip()
- section_path = str(source.get("section_path") or "").strip()
- parts = split_section_path(section_path)
- if len(parts) > 1:
- section_path = " / ".join(parts[:-1])
- if file_name and section_path:
- return f"{file_name} / {section_path}"
- return file_name or section_path
-
-
-def _render_rows_evidence(rows: list[dict]) -> str:
- groups: dict[str, list[str]] = {}
- for row in rows:
- header = _evidence_path_header(row)
- content = str(row.get("content") or "").strip()
- if not content:
- continue
- groups.setdefault(header, []).append(content)
- return render_evidence_blocks(list(groups.items()))
+def _evidence_fields(rows: list[dict]) -> dict:
+ # TODO: 后面用 TypeSafe JEV 补结果重排/筛选。现在没有这一步。
+ # 无论怎么做,发出去的 evidence 和 results 都是已经筛过或重排过的完整列表,不在 Knowhere 外面做。
+ evidence = collect_evidence(rows)
+ return {
+ "evidence": evidence,
+ "evidence_text": flatten_parts(evidence),
+ }
async def run_retrieval_route(
@@ -136,7 +123,7 @@ async def _try_run_small_corpus_route(
"namespace": context.namespace,
"query": context.query,
"router_used": "small_corpus_all",
- "evidence_text": _render_rows_evidence(results),
+ **_evidence_fields(results),
"answer_text": "",
"results": results,
}
@@ -193,7 +180,7 @@ async def _run_classic_topk_route(
"namespace": context.namespace,
"query": context.query,
"router_used": "classic_topk",
- "evidence_text": _render_rows_evidence(results),
+ **_evidence_fields(results),
"answer_text": "",
"results": results,
}
@@ -316,12 +303,12 @@ async def _run_agent_explore_route(
)
decision_trace = [step.to_dict() for step in decision_steps]
- evidence_text = _render_rows_evidence(assembled_rows)
+ evidence_fields = _evidence_fields(assembled_rows)
response = {
"namespace": context.namespace,
"query": context.query,
"router_used": "agent_explore",
- "evidence_text": evidence_text,
+ **evidence_fields,
"answer_text": "",
"referenced_chunks": resolved.refs,
"results": assembled_rows,
@@ -334,6 +321,6 @@ async def _run_agent_explore_route(
completion_label="AGENT EXPLORE RETRIEVAL",
completion_count=len(resolved.refs),
completion_detail=(
- f"chunks | evidence={len(evidence_text)} chars | router=agent_explore"
+ f"chunks | evidence={len(evidence_fields['evidence_text'])} chars | router=agent_explore"
),
)
diff --git a/packages/shared-python/shared/services/retrieval/hydration/asset_inline.py b/packages/shared-python/shared/services/retrieval/hydration/asset_inline.py
index 4c49f00d..7dd7b046 100644
--- a/packages/shared-python/shared/services/retrieval/hydration/asset_inline.py
+++ b/packages/shared-python/shared/services/retrieval/hydration/asset_inline.py
@@ -15,10 +15,13 @@
_SAME_AS_RE = re.compile(r"\[SAME-AS [^\]]+\]")
-def strip_path_placeholders(content: str) -> str:
+def remove_path_placeholders(content: str) -> str:
text = _PATH_REF_RE.sub("", content)
- text = _SAME_AS_RE.sub("", text)
- return text.strip()
+ return _SAME_AS_RE.sub("", text)
+
+
+def strip_path_placeholders(content: str) -> str:
+ return remove_path_placeholders(content).strip()
def inline_assets_at_placeholders(
diff --git a/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py b/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py
new file mode 100644
index 00000000..c5cb9746
--- /dev/null
+++ b/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py
@@ -0,0 +1,359 @@
+"""Compose retrieval evidence parts from raw chunks and connected assets.
+
+Text and standalone image/table chunks place table HTML and image bytes at
+path placeholders. Page chunks are summary plus the page image; they do not
+inline connected charts.
+"""
+
+from __future__ import annotations
+
+import base64
+import tempfile
+from pathlib import Path
+from typing import Any
+
+from loguru import logger
+
+from shared.services.retrieval.hydration.asset_inline import (
+ remove_path_placeholders,
+)
+from shared.services.retrieval.hydration.row_utils import (
+ extract_page_nums,
+ normalize_chunk_type,
+ page_summary,
+)
+from shared.services.retrieval.hydration.table_grid import (
+ TableDownloadError,
+ load_table_html,
+)
+from shared.services.storage.result_storage import get_result_storage
+
+_IMAGE_MEDIA_TYPES = {
+ ".jpg": "image/jpeg",
+ ".jpeg": "image/jpeg",
+ ".png": "image/png",
+ ".gif": "image/gif",
+ ".webp": "image/webp",
+}
+
+
+def compose_evidence_parts(
+ row: dict[str, Any],
+ rows_by_chunk_id: dict[str, dict[str, Any]],
+) -> list[dict[str, Any]]:
+ chunk_type = normalize_chunk_type(row.get("chunk_type"))
+ if chunk_type == "page":
+ return _compose_page_parts(row)
+ if chunk_type == "table":
+ return _compose_standalone_table_parts(row)
+ if chunk_type == "image":
+ return _compose_standalone_image_parts(row)
+ return _compose_text_parts(row, rows_by_chunk_id)
+
+
+def collect_evidence(rows: list[dict[str, Any]]) -> list[dict[str, Any]]:
+ parts: list[dict[str, Any]] = []
+ for row in rows:
+ composed = row.get("composed")
+ if isinstance(composed, list):
+ parts.extend(composed)
+ return parts
+
+
+def flatten_parts(parts: list[dict[str, Any]] | None) -> str:
+ texts: list[str] = []
+ for part in parts or []:
+ if not isinstance(part, dict):
+ continue
+ if part.get("type") == "text":
+ text = str(part.get("text") or "")
+ if text:
+ texts.append(text)
+ continue
+ if part.get("type") != "image":
+ continue
+ media_type = str(part.get("media_type") or "").strip() or "application/octet-stream"
+ data = str(part.get("data") or "").strip()
+ if data:
+ texts.append(f"data:{media_type};base64,{data}")
+ return "".join(texts)
+
+
+def _compose_page_parts(row: dict[str, Any]) -> list[dict[str, Any]]:
+ parts: list[dict[str, Any]] = []
+ summary = page_summary(row)
+ if summary:
+ parts.append(_text_part(summary))
+ image, warning = _try_read_page_image(row)
+ if image is not None:
+ parts.append(image)
+ if warning:
+ parts.append(_text_part(f"Page image unavailable: {warning}"))
+ return parts
+
+
+def _compose_standalone_table_parts(row: dict[str, Any]) -> list[dict[str, Any]]:
+ html = _try_read_table_html(row)
+ if html is None:
+ return []
+ return [_text_part(f"\n{html}\n")]
+
+
+def _compose_standalone_image_parts(row: dict[str, Any]) -> list[dict[str, Any]]:
+ parts: list[dict[str, Any]] = []
+ description = str(row.get("content") or "").strip()
+ if description:
+ parts.append(_text_part(description))
+ image = _try_read_image(row)
+ if image is not None:
+ parts.append(image)
+ return parts
+
+
+def _compose_text_parts(
+ row: dict[str, Any],
+ rows_by_chunk_id: dict[str, dict[str, Any]],
+) -> list[dict[str, Any]]:
+ content = str(row.get("content") or "")
+ tables, images = _embed_targets(row, rows_by_chunk_id)
+ for _target_id, target_row, ref in tables:
+ html = _try_read_table_html(target_row)
+ content, placed = _replace_placeholder(content, ref, "" if html is None else f"\n{html}\n")
+ if html is not None and not placed:
+ _warn_skipped(target_row, "table", "placeholder not found")
+ parts: list[dict[str, Any]] = []
+ remaining = content
+ unused = list(images)
+ while unused:
+ match = _earliest_image_placeholder(remaining, unused)
+ if match is None:
+ _warn_skipped(unused[0][1], "image", "placeholder not found")
+ unused.pop(0)
+ continue
+ before, after, target_row = match
+ if before:
+ parts.append(_text_part(before))
+ image = _try_read_image(target_row)
+ if image is not None:
+ if parts and parts[-1]["type"] == "text":
+ parts[-1]["text"] += "\n"
+ else:
+ parts.append(_text_part("\n"))
+ parts.append(image)
+ parts.append(_text_part("\n"))
+ remaining = after
+ unused = [item for item in unused if item[1] is not target_row]
+ if remaining:
+ parts.append(_text_part(remaining))
+ cleaned: list[dict[str, Any]] = []
+ for part in parts:
+ cleaned_part = _clean_text_part(part)
+ if not _is_empty_text_part(cleaned_part):
+ cleaned.append(cleaned_part)
+ return cleaned
+
+
+def _embed_targets(
+ row: dict[str, Any],
+ rows_by_chunk_id: dict[str, dict[str, Any]],
+) -> tuple[
+ list[tuple[str, dict[str, Any], str]],
+ list[tuple[str, dict[str, Any], str]],
+]:
+ tables: list[tuple[str, dict[str, Any], str]] = []
+ images: list[tuple[str, dict[str, Any], str]] = []
+ for item in _connections(row):
+ if item.get("relation") != "embeds":
+ continue
+ target_id = str(item.get("target") or "").strip()
+ ref = str(item.get("ref") or "").strip()
+ target_row = rows_by_chunk_id.get(target_id)
+ if not target_id or not ref or target_row is None:
+ continue
+ target_type = normalize_chunk_type(target_row.get("chunk_type"))
+ if target_type == "table":
+ tables.append((target_id, target_row, ref))
+ elif target_type == "image":
+ images.append((target_id, target_row, ref))
+ return tables, images
+
+
+def _connections(row: dict[str, Any]) -> list[dict[str, Any]]:
+ metadata = row.get("chunk_metadata") or row.get("metadata") or {}
+ if not isinstance(metadata, dict):
+ return []
+ connections = metadata.get("connect_to") or []
+ if not isinstance(connections, list):
+ return []
+ return [item for item in connections if isinstance(item, dict)]
+
+
+def _replace_placeholder(text: str, ref: str, replacement: str) -> tuple[str, bool]:
+ for candidate in _ref_candidates(ref):
+ if candidate and candidate in text:
+ return text.replace(candidate, replacement, 1), True
+ return text, False
+
+
+def _earliest_image_placeholder(
+ text: str,
+ images: list[tuple[str, dict[str, Any], str]],
+) -> tuple[str, str, dict[str, Any]] | None:
+ best: tuple[int, int, dict[str, Any]] | None = None
+ for _target_id, target_row, ref in images:
+ for candidate in _ref_candidates(ref):
+ if not candidate:
+ continue
+ index = text.find(candidate)
+ if index < 0:
+ continue
+ length = len(candidate)
+ if (
+ best is None
+ or index < best[0]
+ or (index == best[0] and length > best[1])
+ ):
+ best = (index, length, target_row)
+ if best is None:
+ return None
+ index, length, target_row = best
+ return text[:index], text[index + length :], target_row
+
+
+def _ref_candidates(ref: str) -> list[str]:
+ raw = str(ref or "").strip()
+ if not raw:
+ return []
+ out = [raw]
+ if raw.startswith("[") and raw.endswith("]"):
+ inner = raw[1:-1].strip()
+ if inner and inner not in out:
+ out.append(inner)
+ else:
+ bracketed = f"[{raw}]"
+ if bracketed not in out:
+ out.append(bracketed)
+ return out
+
+
+def _try_read_table_html(row: dict[str, Any]) -> str | None:
+ try:
+ html = load_table_html(row).strip()
+ except TableDownloadError as exc:
+ _warn_skipped(row, "table", str(exc))
+ return None
+ if html:
+ return html
+ _warn_skipped(row, "table", "missing table HTML")
+ return None
+
+
+def _try_read_image(row: dict[str, Any]) -> dict[str, Any] | None:
+ artifact = str(row.get("file_path") or "").strip()
+ return _try_read_image_artifact(row, artifact, media_type=_media_type_from_path(artifact))
+
+
+def _try_read_page_image(row: dict[str, Any]) -> tuple[dict[str, Any] | None, str | None]:
+ asset, warning = _select_page_asset(row)
+ if asset is None:
+ reason = warning or "missing page image"
+ _warn_skipped(row, "image", reason)
+ return None, reason
+ artifact = str(asset.get("artifact_ref") or "").strip()
+ media_type = (
+ str(asset.get("content_type") or "").split(";", 1)[0].strip()
+ or _media_type_from_path(artifact)
+ )
+ image = _try_read_image_artifact(row, artifact, media_type=media_type)
+ if image is None:
+ return None, "could not read page image"
+ return image, None
+
+
+def _try_read_image_artifact(
+ row: dict[str, Any],
+ artifact: str,
+ *,
+ media_type: str,
+) -> dict[str, Any] | None:
+ job_id = str(row.get("job_id") or "").strip()
+ storage = get_result_storage()
+ normalized = storage.normalize_artifact_ref(artifact)
+ if not job_id or not normalized:
+ _warn_skipped(row, "image", "missing artifact")
+ return None
+ try:
+ temp_path = storage.download_raw_to_temp(
+ job_id=job_id,
+ relative_path=normalized,
+ suffix=Path(normalized).suffix or ".bin",
+ temp_dir=tempfile.gettempdir(),
+ )
+ body = Path(temp_path).read_bytes()
+ except Exception as exc:
+ _warn_skipped(row, "image", str(exc))
+ return None
+ if not body:
+ _warn_skipped(row, "image", "empty image bytes")
+ return None
+ return {
+ "type": "image",
+ "media_type": media_type,
+ "data": base64.b64encode(body).decode("ascii"),
+ }
+
+
+def _select_page_asset(row: dict[str, Any]) -> tuple[dict[str, Any] | None, str | None]:
+ metadata = row.get("chunk_metadata") or row.get("metadata") or {}
+ if not isinstance(metadata, dict):
+ return None, "missing page image"
+ assets = metadata.get("page_assets") or []
+ if not isinstance(assets, list):
+ return None, "missing page image"
+ candidates = [item for item in assets if isinstance(item, dict)]
+ if not candidates:
+ return None, "missing page image"
+ page_nums = extract_page_nums(row) or []
+ if not page_nums:
+ return None, "missing page number"
+ for item in candidates:
+ raw_page_num = item.get("page_num")
+ if raw_page_num is None:
+ continue
+ try:
+ page_num = int(raw_page_num)
+ except (TypeError, ValueError):
+ continue
+ if page_num in page_nums:
+ return item, None
+ return None, "page image does not match this page"
+
+
+def _media_type_from_path(path: str) -> str:
+ suffix = Path(path).suffix.lower()
+ return _IMAGE_MEDIA_TYPES.get(suffix, "application/octet-stream")
+
+
+def _text_part(text: str) -> dict[str, Any]:
+ return {"type": "text", "text": text}
+
+
+def _clean_text_part(part: dict[str, Any]) -> dict[str, Any]:
+ if part.get("type") != "text":
+ return part
+ return {"type": "text", "text": remove_path_placeholders(str(part.get("text") or ""))}
+
+
+def _is_empty_text_part(part: dict[str, Any]) -> bool:
+ return part.get("type") == "text" and not str(part.get("text") or "")
+
+
+def _warn_skipped(row: dict[str, Any], kind: str, reason: str) -> None:
+ logger.warning(
+ "retrieval: skipped unreachable evidence asset type={} ref={} asset_url={} source_path={} reason={}",
+ kind,
+ row.get("chunk_id"),
+ row.get("asset_url"),
+ row.get("file_path"),
+ reason,
+ )
diff --git a/packages/shared-python/shared/services/retrieval/hydration/evidence_text.py b/packages/shared-python/shared/services/retrieval/hydration/evidence_text.py
deleted file mode 100644
index a868302e..00000000
--- a/packages/shared-python/shared/services/retrieval/hydration/evidence_text.py
+++ /dev/null
@@ -1,45 +0,0 @@
-"""Shared evidence_text rendering: one ``[E#]`` block per group.
-
-Each block is ``[E#]`` + ``[§ path]`` (full traceable path) + body lines.
-Groups are caller-provided; bodies in one group stay in that group.
-"""
-
-from __future__ import annotations
-
-from typing import Sequence
-
-
-def render_evidence_blocks(
- groups: Sequence[tuple[str, Sequence[str]]],
- *,
- start_index: int = 1,
-) -> str:
- """Render (path, bodies) groups into evidence_text.
-
- ``path`` is the full traceable header (e.g. file / section chain).
- Bodies are joined with newlines; multiple bodies in one group are indented.
- ``start_index`` sets the first ``[E#]`` number (default 1).
- """
- parts: list[str] = []
- index = max(1, int(start_index or 1))
- for path, bodies in groups:
- texts = [str(t or "").strip() for t in bodies]
- texts = [t for t in texts if t]
- if not texts:
- continue
- block: list[str] = [f"[E{index + len(parts)}]"]
- header = str(path or "").strip()
- if header:
- block.append(f"[§ {header}]")
- indent = len(texts) >= 2
- for text in texts:
- if indent:
- block.append(
- "\n".join(
- (" " + ln if ln.strip() else ln) for ln in text.splitlines()
- )
- )
- else:
- block.append(text)
- parts.append("\n".join(block).strip())
- return "\n\n".join(parts)
diff --git a/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py b/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py
index 96a38fd6..1cc3219f 100644
--- a/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py
+++ b/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py
@@ -7,16 +7,14 @@
from sqlalchemy.ext.asyncio import AsyncSession
-from shared.services.retrieval.hydration.asset_inline import (
- inline_assets_at_placeholders,
- strip_path_placeholders,
-)
from shared.services.retrieval.hydration.connected import hydrate_connected_target_rows
+from shared.services.retrieval.hydration.evidence_compose import compose_evidence_parts
from shared.services.retrieval.hydration.row_utils import (
extract_page_nums,
filter_excluded_rows,
iter_connected_target_ids,
normalize_chunk_type,
+ page_summary,
)
@@ -69,21 +67,18 @@ async def assemble_retrieval_results(
base_content = str(row.get('content') or '')
chunk_type = normalize_chunk_type(row.get('chunk_type'))
if chunk_type == 'page':
- assembled_row['content'] = _page_summary(row)
+ assembled_row['content'] = page_summary(row)
assembled_row['content_source'] = 'summary'
page_nums = extract_page_nums(row)
if page_nums is not None:
assembled_row['page_nums'] = page_nums
- elif chunk_type == 'table':
- assembled_row['content'] = _compose_table_content(row, rows_by_chunk_id)
- assembled_row['content_source'] = 'summary'
- elif chunk_type == 'text':
- assembled_row['content'] = _compose_text_content(row, rows_by_chunk_id)
- assembled_row['content_source'] = 'content'
else:
assembled_row['content'] = base_content
assembled_row['content_source'] = 'content'
- assembled_row['content'] = strip_path_placeholders(assembled_row['content'])
+ assembled_row['composed'] = compose_evidence_parts(
+ assembled_row,
+ rows_by_chunk_id,
+ )
assembled.append(assembled_row)
return assembled
@@ -109,80 +104,6 @@ def _filter_rows_by_allowed_chunk_types(
]
-def _page_summary(row: dict[str, Any]) -> str:
- metadata = row.get('chunk_metadata') or row.get('metadata') or {}
- if not isinstance(metadata, dict):
- return ''
- return str(metadata.get('summary') or '').strip()
-
-
-def _compose_text_content(
- row: dict[str, Any],
- rows_by_chunk_id: dict[str, dict[str, Any]],
-) -> str:
- base_content = str(row.get('content') or '')
- display_by_target = _connected_display_by_target(row, rows_by_chunk_id)
- if not display_by_target:
- return base_content
- metadata = row.get('chunk_metadata') or row.get('metadata') or {}
- connections = (
- metadata.get('connect_to') if isinstance(metadata, dict) else None
- ) or []
- content, _embedded = inline_assets_at_placeholders(
- base_content,
- connections=connections if isinstance(connections, list) else [],
- display_by_target=display_by_target,
- )
- return content
-
-
-def _compose_table_content(
- row: dict[str, Any],
- rows_by_chunk_id: dict[str, dict[str, Any]],
-) -> str:
- parts = [_table_summary_content(row)]
- parts.extend(_connected_image_parts(row, rows_by_chunk_id))
- return '\n\n'.join(part for part in parts if part)
-
-
-def _connected_display_by_target(
- row: dict[str, Any],
- rows_by_chunk_id: dict[str, dict[str, Any]],
-) -> dict[str, str]:
- display: dict[str, str] = {}
- for target_id in iter_connected_target_ids(row):
- target_row = rows_by_chunk_id.get(target_id)
- if not target_row:
- continue
- target_type = normalize_chunk_type(target_row.get('chunk_type'))
- if target_type == 'table':
- target_content = _compose_table_content(target_row, rows_by_chunk_id)
- elif target_type == 'image':
- target_content = _image_display_content(target_row)
- else:
- continue
- if target_content:
- display[target_id] = target_content
- return display
-
-
-def _connected_image_parts(
- row: dict[str, Any],
- rows_by_chunk_id: dict[str, dict[str, Any]],
-) -> list[str]:
- parts: list[str] = []
- for target_id in iter_connected_target_ids(row):
- target_row = rows_by_chunk_id.get(target_id)
- if not target_row:
- continue
- if normalize_chunk_type(target_row.get('chunk_type')) != 'image':
- continue
- content = _image_display_content(target_row)
- if content:
- parts.append(content)
- return parts
-
-
def _image_display_content(row: dict[str, Any]) -> str:
display_ref = (
str(row.get('asset_url') or '').strip()
@@ -197,44 +118,3 @@ def _image_display_content(row: dict[str, Any]) -> str:
if description:
lines.extend(line for line in description.split('\n') if line.strip())
return '\n'.join(lines)
-
-
-def _table_summary_content(row: dict[str, Any]) -> str:
- metadata = row.get('chunk_metadata') or row.get('metadata') or {}
- if not isinstance(metadata, dict):
- metadata = {}
-
- display_ref = _table_display_ref(row)
- lines = [f"[Table: {display_ref}]" if display_ref else "[Table]"]
-
- summary = str(metadata.get('summary') or row.get('summary') or '').strip()
- if summary:
- lines.extend(line for line in summary.split('\n') if line.strip())
-
- keywords = metadata.get('keywords') or row.get('keywords') or []
- if isinstance(keywords, list):
- keyword_text = ';'.join(
- str(keyword).strip() for keyword in keywords if str(keyword).strip()
- )
- else:
- keyword_text = str(keywords or '').strip()
- if keyword_text:
- lines.append(keyword_text)
-
- caption = str(metadata.get('caption') or row.get('caption') or '').strip()
- if caption:
- lines.append(caption)
-
- return '\n'.join(lines)
-
-
-def _table_display_ref(row: dict[str, Any]) -> str:
- for key in ('asset_url', 'file_path', 'source_chunk_path'):
- value = str(row.get(key) or '').strip()
- if value:
- return value
-
- content = str(row.get('content') or '').strip()
- if content and not content.lstrip().lower().startswith(' list[int] | None:
return page_nums if isinstance(page_nums, list) else None
+def page_summary(row: dict[str, Any]) -> str:
+ metadata = row.get('chunk_metadata') or row.get('metadata') or {}
+ if not isinstance(metadata, dict):
+ return ''
+ return str(metadata.get('summary') or '').strip()
+
+
def build_reference_lookup_key(
*,
document_id: object,
diff --git a/packages/shared-python/shared/services/retrieval/hydration/table_grid.py b/packages/shared-python/shared/services/retrieval/hydration/table_grid.py
index bf3b92b8..28c6159b 100644
--- a/packages/shared-python/shared/services/retrieval/hydration/table_grid.py
+++ b/packages/shared-python/shared/services/retrieval/hydration/table_grid.py
@@ -1,7 +1,7 @@
"""Load table HTML from storage and expand rowspan/colspan into a grid.
-Used by explore-phase ``corpus.read``, grep/recall hit mounting, and
-``corpus.query_table``. Final retrieval assembly does not call this module.
+Used by explore-phase ``corpus.read``, grep/recall hit mounting,
+``corpus.query_table``, and final evidence compose for table HTML.
"""
from __future__ import annotations
diff --git a/packages/shared-python/shared/tests/test_asset_inline.py b/packages/shared-python/shared/tests/test_asset_inline.py
index 75b09c04..888fe2e0 100644
--- a/packages/shared-python/shared/tests/test_asset_inline.py
+++ b/packages/shared-python/shared/tests/test_asset_inline.py
@@ -94,11 +94,17 @@ async def test_assemble_inserts_table_at_placeholder() -> None:
)
assert len(assembled) == 1
content = assembled[0]["content"]
- assert "[tables/" not in content
- assert content.index("见表") < content.index("[Table:")
- assert content.index("[Table:") < content.index("结束")
- assert "企业入驻信息登记模板" in content
- assert "SHOULD NOT LEAK" not in content
+ assert "[tables/table-1.html]" in content
+ assert "[Table:" not in content
+ composed_text = "".join(
+ str(part.get("text") or "")
+ for part in assembled[0]["composed"]
+ if part.get("type") == "text"
+ )
+ assert composed_text.index("见表") < composed_text.index(" list[dict[str, object]]:
)
assert [row["chunk_id"] for row in assembled] == ["text-1"]
- assert display_marker in assembled[0]["content"]
- assert "资产说明" in assembled[0]["content"]
+ assert placeholder in assembled[0]["content"]
+ assert display_marker not in assembled[0]["content"]
def test_node_unit_span_inlines_section_assets() -> None:
diff --git a/packages/shared-python/shared/tests/test_evidence_compose.py b/packages/shared-python/shared/tests/test_evidence_compose.py
new file mode 100644
index 00000000..c8387c68
--- /dev/null
+++ b/packages/shared-python/shared/tests/test_evidence_compose.py
@@ -0,0 +1,358 @@
+from __future__ import annotations
+
+import base64
+
+import pytest
+
+from shared.services.retrieval.hydration.evidence_compose import (
+ compose_evidence_parts,
+ flatten_parts,
+)
+from shared.services.retrieval.hydration.result_assembly import assemble_retrieval_results
+
+
+def test_flatten_parts_keeps_html_and_encodes_images() -> None:
+ text = flatten_parts(
+ [
+ {"type": "text", "text": "before"},
+ {"type": "image", "media_type": "image/png", "data": "abc"},
+ {"type": "text", "text": "after"},
+ ]
+ )
+ assert text == "beforedata:image/png;base64,abcafter"
+
+
+def test_page_parts_are_summary_then_page_image(monkeypatch) -> None:
+ monkeypatch.setattr(
+ "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact",
+ lambda row, artifact, media_type: {
+ "type": "image",
+ "media_type": media_type,
+ "data": base64.b64encode(b"page").decode("ascii"),
+ },
+ )
+ parts = compose_evidence_parts(
+ {
+ "chunk_type": "page",
+ "chunk_metadata": {
+ "summary": "制度摘要",
+ "page_nums": [4],
+ "page_assets": [
+ {
+ "page_num": 4,
+ "artifact_ref": "page_citation_assets/page-4.png",
+ "content_type": "image/png",
+ }
+ ],
+ },
+ },
+ {},
+ )
+ assert parts[0] == {"type": "text", "text": "制度摘要"}
+ assert parts[1]["type"] == "image"
+ assert parts[1]["media_type"] == "image/png"
+
+
+def test_page_parts_do_not_inline_connected_charts() -> None:
+ parts = compose_evidence_parts(
+ {
+ "chunk_type": "page",
+ "content": "RAW",
+ "chunk_metadata": {
+ "summary": "只要摘要",
+ "connect_to": [
+ {
+ "target": "table-1",
+ "relation": "embeds",
+ "ref": "[tables/a.html]",
+ }
+ ],
+ },
+ },
+ {
+ "table-1": {
+ "chunk_type": "table",
+ "content": "",
+ }
+ },
+ )
+ assert parts[0] == {"type": "text", "text": "只要摘要"}
+ assert parts[1] == {
+ "type": "text",
+ "text": "Page image unavailable: missing page image",
+ }
+
+
+def test_standalone_table_uses_html() -> None:
+ parts = compose_evidence_parts(
+ {
+ "chunk_type": "table",
+ "content": "",
+ "file_path": "tables/a.html",
+ },
+ {},
+ )
+ assert parts[0]["type"] == "text"
+ assert "" in parts[0]["text"]
+
+
+def test_standalone_image_is_bytes(monkeypatch) -> None:
+ monkeypatch.setattr(
+ "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact",
+ lambda row, artifact, media_type: {
+ "type": "image",
+ "media_type": "image/jpeg",
+ "data": "Zm9v",
+ },
+ )
+ parts = compose_evidence_parts(
+ {
+ "chunk_type": "image",
+ "content": "chart caption",
+ "file_path": "images/a.jpg",
+ "job_id": "job-1",
+ },
+ {},
+ )
+ assert parts[0] == {"type": "text", "text": "chart caption"}
+ assert parts[1] == {"type": "image", "media_type": "image/jpeg", "data": "Zm9v"}
+
+
+@pytest.mark.asyncio
+async def test_text_result_keeps_placeholders_and_composes_assets(monkeypatch) -> None:
+ monkeypatch.setattr(
+ "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact",
+ lambda row, artifact, media_type: {
+ "type": "image",
+ "media_type": "image/png",
+ "data": "aW1n",
+ },
+ )
+ assembled = await assemble_retrieval_results(
+ rows=[
+ {
+ "chunk_id": "text-1",
+ "chunk_type": "text",
+ "content": "见表 [tables/a.html] 再看 [images/a.png] 结束",
+ "chunk_metadata": {
+ "connect_to": [
+ {
+ "target": "table-1",
+ "relation": "embeds",
+ "ref": "[tables/a.html]",
+ },
+ {
+ "target": "image-1",
+ "relation": "embeds",
+ "ref": "[images/a.png]",
+ },
+ ]
+ },
+ },
+ {
+ "chunk_id": "table-1",
+ "chunk_type": "table",
+ "content": "",
+ "file_path": "tables/a.html",
+ },
+ {
+ "chunk_id": "image-1",
+ "chunk_type": "image",
+ "file_path": "images/a.png",
+ "job_id": "job-1",
+ },
+ ],
+ exclude_document_ids=[],
+ exclude_sections=[],
+ )
+ assert assembled[0]["content"] == "见表 [tables/a.html] 再看 [images/a.png] 结束"
+ types = [part["type"] for part in assembled[0]["composed"]]
+ assert "image" in types
+ composed_text = "".join(
+ part["text"] for part in assembled[0]["composed"] if part["type"] == "text"
+ )
+ assert "" in composed_text
+ assert "[tables/" not in composed_text
+ assert "[images/" not in composed_text
+
+
+def test_text_image_keeps_newlines_around_image(monkeypatch) -> None:
+ monkeypatch.setattr(
+ "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact",
+ lambda row, artifact, media_type: {
+ "type": "image",
+ "media_type": "image/png",
+ "data": "aW1n",
+ },
+ )
+ parts = compose_evidence_parts(
+ {
+ "chunk_type": "text",
+ "content": "前 [images/a.png] 后",
+ "chunk_metadata": {
+ "connect_to": [
+ {
+ "target": "image-1",
+ "relation": "embeds",
+ "ref": "[images/a.png]",
+ }
+ ]
+ },
+ },
+ {
+ "image-1": {
+ "chunk_type": "image",
+ "file_path": "images/a.png",
+ "job_id": "job-1",
+ }
+ },
+ )
+ assert parts[0] == {"type": "text", "text": "前 \n"}
+ assert parts[1] == {"type": "image", "media_type": "image/png", "data": "aW1n"}
+ assert parts[2] == {"type": "text", "text": "\n"}
+ assert parts[3] == {"type": "text", "text": " 后"}
+
+
+def test_unreachable_assets_clear_placeholders(monkeypatch) -> None:
+ monkeypatch.setattr(
+ "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact",
+ lambda row, artifact, media_type: None,
+ )
+ monkeypatch.setattr(
+ "shared.services.retrieval.hydration.evidence_compose._try_read_table_html",
+ lambda row: None,
+ )
+ parts = compose_evidence_parts(
+ {
+ "chunk_type": "text",
+ "content": "见表 [tables/a.html] 再看 [images/a.png] 结束",
+ "chunk_metadata": {
+ "connect_to": [
+ {
+ "target": "table-1",
+ "relation": "embeds",
+ "ref": "[tables/a.html]",
+ },
+ {
+ "target": "image-1",
+ "relation": "embeds",
+ "ref": "[images/a.png]",
+ },
+ ]
+ },
+ },
+ {
+ "table-1": {"chunk_type": "table", "file_path": "tables/a.html"},
+ "image-1": {
+ "chunk_type": "image",
+ "file_path": "images/a.png",
+ "job_id": "job-1",
+ },
+ },
+ )
+ composed_text = "".join(part["text"] for part in parts if part["type"] == "text")
+ assert composed_text == "见表 再看 结束"
+ assert all(part["type"] != "image" for part in parts)
+
+
+def test_missing_placeholder_does_not_append_asset(monkeypatch) -> None:
+ monkeypatch.setattr(
+ "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact",
+ lambda row, artifact, media_type: {
+ "type": "image",
+ "media_type": "image/png",
+ "data": "aW1n",
+ },
+ )
+ parts = compose_evidence_parts(
+ {
+ "chunk_type": "text",
+ "content": "没有占位符",
+ "chunk_metadata": {
+ "connect_to": [
+ {
+ "target": "image-1",
+ "relation": "embeds",
+ "ref": "[images/a.png]",
+ }
+ ]
+ },
+ },
+ {
+ "image-1": {
+ "chunk_type": "image",
+ "file_path": "images/a.png",
+ "job_id": "job-1",
+ }
+ },
+ )
+ assert parts == [{"type": "text", "text": "没有占位符"}]
+
+
+def test_page_image_requires_matching_page_num(monkeypatch) -> None:
+ monkeypatch.setattr(
+ "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact",
+ lambda row, artifact, media_type: {
+ "type": "image",
+ "media_type": media_type,
+ "data": "cGFnZQ==",
+ },
+ )
+ parts = compose_evidence_parts(
+ {
+ "chunk_type": "page",
+ "chunk_metadata": {
+ "summary": "只要摘要",
+ "page_nums": [4],
+ "page_assets": [
+ {
+ "page_num": 9,
+ "artifact_ref": "page_citation_assets/page-9.png",
+ "content_type": "image/png",
+ }
+ ],
+ },
+ },
+ {},
+ )
+ assert parts == [
+ {"type": "text", "text": "只要摘要"},
+ {
+ "type": "text",
+ "text": "Page image unavailable: page image does not match this page",
+ },
+ ]
+
+
+def test_page_image_missing_page_number_does_not_use_first_asset(monkeypatch) -> None:
+ monkeypatch.setattr(
+ "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact",
+ lambda row, artifact, media_type: {
+ "type": "image",
+ "media_type": media_type,
+ "data": "cGFnZQ==",
+ },
+ )
+ parts = compose_evidence_parts(
+ {
+ "chunk_type": "page",
+ "chunk_metadata": {
+ "summary": "只要摘要",
+ "page_assets": [
+ {
+ "page_num": 1,
+ "artifact_ref": "page_citation_assets/page-1.png",
+ "content_type": "image/png",
+ }
+ ],
+ },
+ },
+ {},
+ )
+ assert parts == [
+ {"type": "text", "text": "只要摘要"},
+ {
+ "type": "text",
+ "text": "Page image unavailable: missing page number",
+ },
+ ]