From 57ad57c4392ea0bbd7e404830d016211c3c18019 Mon Sep 17 00:00:00 2001
From: chengke <404835780@qq.com>
Date: Sun, 20 Sep 2026 23:08:41 +0800
Subject: [PATCH 1/4] feat(retrieval): enhance evidence handling and response
structure
- Added a new `evidence` field to the `RetrievalQueryResponse` model to include composed evidence parts (text and inline images).
- Updated the MCP response projection to return the new `evidence` field alongside `evidence_text`.
- Refactored evidence rendering logic to flatten composed parts and improve the assembly of retrieval results.
- Adjusted tests to validate the new evidence structure and ensure proper handling of composed evidence in responses.
---
apps/api/app/api/v1/routes/retrieval.py | 6 +-
apps/api/app/mcp/retrieval_server.py | 14 +-
.../test_evidence_renderer_contract.py | 56 +--
.../test_page_memory_retrieval_contract.py | 16 +-
.../services/retrieval/execution/plan.py | 1 +
.../execution/response_projection.py | 1 +
.../services/retrieval/execution/routes.py | 45 +--
.../retrieval/hydration/asset_inline.py | 9 +-
.../retrieval/hydration/evidence_compose.py | 355 ++++++++++++++++++
.../retrieval/hydration/result_assembly.py | 124 +-----
.../services/retrieval/hydration/row_utils.py | 1 +
.../retrieval/hydration/table_grid.py | 4 +-
.../shared/tests/test_asset_inline.py | 20 +-
.../shared/tests/test_evidence_compose.py | 314 ++++++++++++++++
14 files changed, 753 insertions(+), 213 deletions(-)
create mode 100644 packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py
create mode 100644 packages/shared-python/shared/tests/test_evidence_compose.py
diff --git a/apps/api/app/api/v1/routes/retrieval.py b/apps/api/app/api/v1/routes/retrieval.py
index 087b8639..ecc49e4a 100644
--- a/apps/api/app/api/v1/routes/retrieval.py
+++ b/apps/api/app/api/v1/routes/retrieval.py
@@ -154,9 +154,13 @@ class RetrievalQueryResponse(BaseModel):
namespace: str
query: str
router_used: str
+ evidence: list[dict] = Field(
+ default_factory=list,
+ description="Composed evidence parts (text and inline images) for downstream agents.",
+ )
evidence_text: str = Field(
default="",
- description="Hierarchical evidence text. Primary output for downstream agents.",
+ description="Text projection of evidence. Tables stay as HTML; images are data URLs.",
)
answer_text: str = Field(
default="",
diff --git a/apps/api/app/mcp/retrieval_server.py b/apps/api/app/mcp/retrieval_server.py
index 89845b64..301ba92c 100644
--- a/apps/api/app/mcp/retrieval_server.py
+++ b/apps/api/app/mcp/retrieval_server.py
@@ -56,13 +56,15 @@ def resolve_mcp_namespace(*, ctx: Context | None) -> str:
def to_mcp_query_response(response: dict[str, Any]) -> dict[str, Any]:
"""Project the internal retrieval response to the MCP agent contract.
- MCP returns exactly 3 PRIMARY fields:
- - evidence_text: hierarchical evidence tree for LLM consumption
+ MCP returns composed evidence plus the text projection and citations:
+ - evidence: composed parts (text and inline images)
+ - evidence_text: text projection of those parts
- referenced_chunks: structured chunk references for citation / follow-up
- decision_trace: navigation decisions including terminal stop/failure
"""
return {
"query": response.get("query"),
+ "evidence": response.get("evidence") or [],
"evidence_text": response.get("evidence_text") or "",
"referenced_chunks": response.get("referenced_chunks") or [],
"decision_trace": response.get("decision_trace") or [],
@@ -101,10 +103,10 @@ def create_retrieval_mcp_server(
@server.tool(
name="retrieval.query",
description=(
- "Search published documents. Returns evidence_text (hierarchical "
- "evidence for LLM consumption), referenced_chunks (cited chunk "
- "metadata for follow-up queries), and decision_trace (navigation "
- "decisions including stop/failure reasons). "
+ "Search published documents. Returns evidence (composed parts for "
+ "LLM consumption), evidence_text (text projection of those parts), "
+ "referenced_chunks (cited chunk metadata for follow-up queries), "
+ "and decision_trace (navigation decisions including stop/failure reasons). "
"Include navigation intent directly in your query text — the "
"engine will automatically locate the right documents and sections."
),
diff --git a/apps/api/tests/contract/test_evidence_renderer_contract.py b/apps/api/tests/contract/test_evidence_renderer_contract.py
index a714ff54..b023a8df 100644
--- a/apps/api/tests/contract/test_evidence_renderer_contract.py
+++ b/apps/api/tests/contract/test_evidence_renderer_contract.py
@@ -1,46 +1,26 @@
-from shared.services.retrieval.execution.routes import _render_rows_evidence
+from shared.services.retrieval.execution.routes import _evidence_fields
-def test_render_rows_evidence_should_group_siblings_under_parent_path() -> None:
+def test_evidence_fields_flatten_composed_parts_in_result_order() -> None:
rows = [
+ {"composed": [{"type": "text", "text": "first"}]},
{
- "chunk_id": "c2",
- "content": "second section content",
- "sort_order": 2,
- "source": {
- "source_file_name": "alpha.pdf",
- "section_path": "Alpha / Two",
- },
- },
- {
- "chunk_id": "c1",
- "content": "first section content\nwith more detail",
- "sort_order": 1,
- "source": {
- "source_file_name": "alpha.pdf",
- "section_path": "Alpha / One",
- },
- },
- {
- "chunk_id": "c3",
- "content": "
",
- "source_file_name": "beta.pdf",
- "section_path": "Beta / Table",
+ "composed": [
+ {"type": "text", "text": "mid "},
+ {"type": "image", "media_type": "image/png", "data": "abc"},
+ ]
},
+ {"composed": [{"type": "text", "text": ""}]},
]
- evidence_text = _render_rows_evidence(rows)
+ fields = _evidence_fields(rows)
- assert evidence_text.count("[E1]") == 1
- assert evidence_text.count("[E2]") == 1
- assert "[E3]" not in evidence_text
- assert "[§ alpha.pdf / Alpha]" in evidence_text
- assert "[§ beta.pdf / Beta]" in evidence_text
- assert "[§ alpha.pdf / Alpha / One]" not in evidence_text
- assert "[§ alpha.pdf / Alpha / Two]" not in evidence_text
- assert "first section content" in evidence_text
- assert "second section content" in evidence_text
- assert "" in evidence_text
- assert "[Document]" not in evidence_text
- assert "▸" not in evidence_text
- assert "┈" not in evidence_text
+ assert fields["evidence"] == [
+ {"type": "text", "text": "first"},
+ {"type": "text", "text": "mid "},
+ {"type": "image", "media_type": "image/png", "data": "abc"},
+ {"type": "text", "text": ""},
+ ]
+ assert fields["evidence_text"] == (
+ "firstmid data:image/png;base64,abc"
+ )
diff --git a/apps/worker/tests/contract/test_page_memory_retrieval_contract.py b/apps/worker/tests/contract/test_page_memory_retrieval_contract.py
index e9346c7e..bfebab1d 100644
--- a/apps/worker/tests/contract/test_page_memory_retrieval_contract.py
+++ b/apps/worker/tests/contract/test_page_memory_retrieval_contract.py
@@ -142,13 +142,15 @@ async def test_table_result_assembly_uses_summary_not_html() -> None:
assert len(assembled) == 1
content = assembled[0]["content"]
- assert "[Table: https://assets.example.com/job-1/tables/table-1.html]" in content
- assert "企业入驻信息登记模板" in content
- assert "企业名称;统一社会信用代码" in content
- assert "SHOULD NOT LEAK" not in content
- assert "" in composed_text
+ assert "[tables/" not in composed_text
+ assert composed_text.index("见表") < composed_text.index(" dict[str, An
"namespace": request.namespace,
"query": request.query,
"router_used": "empty_query_filtered",
+ "evidence": [],
"evidence_text": "",
"answer_text": "",
"referenced_chunks": [],
diff --git a/packages/shared-python/shared/services/retrieval/execution/response_projection.py b/packages/shared-python/shared/services/retrieval/execution/response_projection.py
index 1ac29076..8878d82c 100644
--- a/packages/shared-python/shared/services/retrieval/execution/response_projection.py
+++ b/packages/shared-python/shared/services/retrieval/execution/response_projection.py
@@ -25,6 +25,7 @@ async def project_public_retrieval_response(response: dict[str, Any]) -> dict[st
'namespace': response.get('namespace'),
'query': response.get('query'),
'router_used': response.get('router_used'),
+ 'evidence': response.get('evidence') or [],
'evidence_text': response.get('evidence_text') or '',
'answer_text': '',
'referenced_chunks': response.get('referenced_chunks') or [],
diff --git a/packages/shared-python/shared/services/retrieval/execution/routes.py b/packages/shared-python/shared/services/retrieval/execution/routes.py
index 25eaac5a..dde8a9eb 100644
--- a/packages/shared-python/shared/services/retrieval/execution/routes.py
+++ b/packages/shared-python/shared/services/retrieval/execution/routes.py
@@ -13,11 +13,13 @@
RetrievalRouteContext,
RetrievalRouteOutcome,
)
-from shared.services.retrieval.hydration.evidence_text import render_evidence_blocks
+from shared.services.retrieval.hydration.evidence_compose import (
+ collect_evidence,
+ flatten_parts,
+)
from shared.services.retrieval.hydration.result_assembly import (
assemble_retrieval_results,
)
-from shared.services.retrieval.search.lexical_text import split_section_path
from shared.services.retrieval.search.map_unit_discovery import map_unit_discovery
from shared.services.retrieval.search.ranking import rank_retrieval_candidates
from shared.services.retrieval.search.scoped_corpus import (
@@ -47,29 +49,12 @@ def open_agent_explore_database_context() -> AbstractAsyncContextManager[AsyncSe
)
-def _evidence_path_header(row: dict) -> str:
- source = row.get("source")
- if not isinstance(source, dict):
- source = row
- file_name = str(source.get("source_file_name") or "").strip()
- section_path = str(source.get("section_path") or "").strip()
- parts = split_section_path(section_path)
- if len(parts) > 1:
- section_path = " / ".join(parts[:-1])
- if file_name and section_path:
- return f"{file_name} / {section_path}"
- return file_name or section_path
-
-
-def _render_rows_evidence(rows: list[dict]) -> str:
- groups: dict[str, list[str]] = {}
- for row in rows:
- header = _evidence_path_header(row)
- content = str(row.get("content") or "").strip()
- if not content:
- continue
- groups.setdefault(header, []).append(content)
- return render_evidence_blocks(list(groups.items()))
+def _evidence_fields(rows: list[dict]) -> dict:
+ evidence = collect_evidence(rows)
+ return {
+ "evidence": evidence,
+ "evidence_text": flatten_parts(evidence),
+ }
async def run_retrieval_route(
@@ -136,7 +121,7 @@ async def _try_run_small_corpus_route(
"namespace": context.namespace,
"query": context.query,
"router_used": "small_corpus_all",
- "evidence_text": _render_rows_evidence(results),
+ **_evidence_fields(results),
"answer_text": "",
"results": results,
}
@@ -193,7 +178,7 @@ async def _run_classic_topk_route(
"namespace": context.namespace,
"query": context.query,
"router_used": "classic_topk",
- "evidence_text": _render_rows_evidence(results),
+ **_evidence_fields(results),
"answer_text": "",
"results": results,
}
@@ -316,12 +301,12 @@ async def _run_agent_explore_route(
)
decision_trace = [step.to_dict() for step in decision_steps]
- evidence_text = _render_rows_evidence(assembled_rows)
+ evidence_fields = _evidence_fields(assembled_rows)
response = {
"namespace": context.namespace,
"query": context.query,
"router_used": "agent_explore",
- "evidence_text": evidence_text,
+ **evidence_fields,
"answer_text": "",
"referenced_chunks": resolved.refs,
"results": assembled_rows,
@@ -334,6 +319,6 @@ async def _run_agent_explore_route(
completion_label="AGENT EXPLORE RETRIEVAL",
completion_count=len(resolved.refs),
completion_detail=(
- f"chunks | evidence={len(evidence_text)} chars | router=agent_explore"
+ f"chunks | evidence={len(evidence_fields['evidence_text'])} chars | router=agent_explore"
),
)
diff --git a/packages/shared-python/shared/services/retrieval/hydration/asset_inline.py b/packages/shared-python/shared/services/retrieval/hydration/asset_inline.py
index 4c49f00d..7dd7b046 100644
--- a/packages/shared-python/shared/services/retrieval/hydration/asset_inline.py
+++ b/packages/shared-python/shared/services/retrieval/hydration/asset_inline.py
@@ -15,10 +15,13 @@
_SAME_AS_RE = re.compile(r"\[SAME-AS [^\]]+\]")
-def strip_path_placeholders(content: str) -> str:
+def remove_path_placeholders(content: str) -> str:
text = _PATH_REF_RE.sub("", content)
- text = _SAME_AS_RE.sub("", text)
- return text.strip()
+ return _SAME_AS_RE.sub("", text)
+
+
+def strip_path_placeholders(content: str) -> str:
+ return remove_path_placeholders(content).strip()
def inline_assets_at_placeholders(
diff --git a/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py b/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py
new file mode 100644
index 00000000..d6638332
--- /dev/null
+++ b/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py
@@ -0,0 +1,355 @@
+"""Compose retrieval evidence parts from raw chunks and connected assets.
+
+Text and standalone image/table chunks place table HTML and image bytes at
+path placeholders. Page chunks are summary plus the page image; they do not
+inline connected charts.
+"""
+
+from __future__ import annotations
+
+import base64
+import tempfile
+from pathlib import Path
+from typing import Any
+
+from loguru import logger
+
+from shared.services.retrieval.hydration.asset_inline import (
+ remove_path_placeholders,
+)
+from shared.services.retrieval.hydration.row_utils import (
+ extract_page_nums,
+ normalize_chunk_type,
+)
+from shared.services.retrieval.hydration.table_grid import (
+ TableDownloadError,
+ load_table_html,
+)
+from shared.services.storage.result_storage import get_result_storage
+
+_IMAGE_MEDIA_TYPES = {
+ ".jpg": "image/jpeg",
+ ".jpeg": "image/jpeg",
+ ".png": "image/png",
+ ".gif": "image/gif",
+ ".webp": "image/webp",
+}
+
+
+def compose_evidence_parts(
+ row: dict[str, Any],
+ rows_by_chunk_id: dict[str, dict[str, Any]],
+) -> list[dict[str, Any]]:
+ chunk_type = normalize_chunk_type(row.get("chunk_type"))
+ if chunk_type == "page":
+ return _compose_page_parts(row)
+ if chunk_type == "table":
+ return _compose_standalone_table_parts(row)
+ if chunk_type == "image":
+ return _compose_standalone_image_parts(row)
+ return _compose_text_parts(row, rows_by_chunk_id)
+
+
+def collect_evidence(rows: list[dict[str, Any]]) -> list[dict[str, Any]]:
+ parts: list[dict[str, Any]] = []
+ for row in rows:
+ composed = row.get("composed")
+ if isinstance(composed, list):
+ parts.extend(composed)
+ return parts
+
+
+def flatten_parts(parts: list[dict[str, Any]] | None) -> str:
+ texts: list[str] = []
+ for part in parts or []:
+ if not isinstance(part, dict):
+ continue
+ if part.get("type") == "text":
+ text = str(part.get("text") or "")
+ if text:
+ texts.append(text)
+ continue
+ if part.get("type") != "image":
+ continue
+ media_type = str(part.get("media_type") or "").strip() or "application/octet-stream"
+ data = str(part.get("data") or "").strip()
+ if data:
+ texts.append(f"data:{media_type};base64,{data}")
+ return "".join(texts)
+
+
+def _compose_page_parts(row: dict[str, Any]) -> list[dict[str, Any]]:
+ parts: list[dict[str, Any]] = []
+ summary = _page_summary(row)
+ if summary:
+ parts.append(_text_part(summary))
+ image = _try_read_page_image(row)
+ if image is not None:
+ parts.append(image)
+ return parts
+
+
+def _compose_standalone_table_parts(row: dict[str, Any]) -> list[dict[str, Any]]:
+ html = _try_read_table_html(row)
+ if html is None:
+ return []
+ return [_text_part(f"\n{html}\n")]
+
+
+def _compose_standalone_image_parts(row: dict[str, Any]) -> list[dict[str, Any]]:
+ parts: list[dict[str, Any]] = []
+ description = str(row.get("content") or "").strip()
+ if description:
+ parts.append(_text_part(description))
+ image = _try_read_image(row)
+ if image is not None:
+ parts.append(image)
+ return parts
+
+
+def _compose_text_parts(
+ row: dict[str, Any],
+ rows_by_chunk_id: dict[str, dict[str, Any]],
+) -> list[dict[str, Any]]:
+ content = str(row.get("content") or "")
+ tables, images = _embed_targets(row, rows_by_chunk_id)
+ for _target_id, target_row, ref in tables:
+ html = _try_read_table_html(target_row)
+ content, placed = _replace_placeholder(content, ref, "" if html is None else f"\n{html}\n")
+ if html is not None and not placed:
+ _warn_skipped(target_row, "table", "placeholder not found")
+ parts: list[dict[str, Any]] = []
+ remaining = content
+ unused = list(images)
+ while unused:
+ match = _earliest_image_placeholder(remaining, unused)
+ if match is None:
+ _warn_skipped(unused[0][1], "image", "placeholder not found")
+ unused.pop(0)
+ continue
+ before, after, target_row = match
+ if before:
+ parts.append(_text_part(before))
+ image = _try_read_image(target_row)
+ if image is not None:
+ if parts and parts[-1]["type"] == "text":
+ parts[-1]["text"] += "\n"
+ else:
+ parts.append(_text_part("\n"))
+ parts.append(image)
+ parts.append(_text_part("\n"))
+ remaining = after
+ unused = [item for item in unused if item[1] is not target_row]
+ if remaining:
+ parts.append(_text_part(remaining))
+ cleaned: list[dict[str, Any]] = []
+ for part in parts:
+ cleaned_part = _clean_text_part(part)
+ if not _is_empty_text_part(cleaned_part):
+ cleaned.append(cleaned_part)
+ return cleaned
+
+
+def _embed_targets(
+ row: dict[str, Any],
+ rows_by_chunk_id: dict[str, dict[str, Any]],
+) -> tuple[
+ list[tuple[str, dict[str, Any], str]],
+ list[tuple[str, dict[str, Any], str]],
+]:
+ tables: list[tuple[str, dict[str, Any], str]] = []
+ images: list[tuple[str, dict[str, Any], str]] = []
+ for item in _connections(row):
+ if item.get("relation") != "embeds":
+ continue
+ target_id = str(item.get("target") or "").strip()
+ ref = str(item.get("ref") or "").strip()
+ target_row = rows_by_chunk_id.get(target_id)
+ if not target_id or not ref or target_row is None:
+ continue
+ target_type = normalize_chunk_type(target_row.get("chunk_type"))
+ if target_type == "table":
+ tables.append((target_id, target_row, ref))
+ elif target_type == "image":
+ images.append((target_id, target_row, ref))
+ return tables, images
+
+
+def _connections(row: dict[str, Any]) -> list[dict[str, Any]]:
+ metadata = row.get("chunk_metadata") or row.get("metadata") or {}
+ if not isinstance(metadata, dict):
+ return []
+ connections = metadata.get("connect_to") or []
+ if not isinstance(connections, list):
+ return []
+ return [item for item in connections if isinstance(item, dict)]
+
+
+def _replace_placeholder(text: str, ref: str, replacement: str) -> tuple[str, bool]:
+ for candidate in _ref_candidates(ref):
+ if candidate and candidate in text:
+ return text.replace(candidate, replacement, 1), True
+ return text, False
+
+
+def _earliest_image_placeholder(
+ text: str,
+ images: list[tuple[str, dict[str, Any], str]],
+) -> tuple[str, str, dict[str, Any]] | None:
+ best: tuple[int, int, dict[str, Any]] | None = None
+ for _target_id, target_row, ref in images:
+ for candidate in _ref_candidates(ref):
+ if not candidate:
+ continue
+ index = text.find(candidate)
+ if index < 0:
+ continue
+ length = len(candidate)
+ if (
+ best is None
+ or index < best[0]
+ or (index == best[0] and length > best[1])
+ ):
+ best = (index, length, target_row)
+ if best is None:
+ return None
+ index, length, target_row = best
+ return text[:index], text[index + length :], target_row
+
+
+def _ref_candidates(ref: str) -> list[str]:
+ raw = str(ref or "").strip()
+ if not raw:
+ return []
+ out = [raw]
+ if raw.startswith("[") and raw.endswith("]"):
+ inner = raw[1:-1].strip()
+ if inner and inner not in out:
+ out.append(inner)
+ else:
+ bracketed = f"[{raw}]"
+ if bracketed not in out:
+ out.append(bracketed)
+ return out
+
+
+def _try_read_table_html(row: dict[str, Any]) -> str | None:
+ try:
+ html = load_table_html(row).strip()
+ except TableDownloadError as exc:
+ _warn_skipped(row, "table", str(exc))
+ return None
+ if html:
+ return html
+ _warn_skipped(row, "table", "missing table HTML")
+ return None
+
+
+def _try_read_image(row: dict[str, Any]) -> dict[str, Any] | None:
+ artifact = str(row.get("file_path") or "").strip()
+ return _try_read_image_artifact(row, artifact, media_type=_media_type_from_path(artifact))
+
+
+def _try_read_page_image(row: dict[str, Any]) -> dict[str, Any] | None:
+ asset = _select_page_asset(row)
+ if asset is None:
+ return None
+ artifact = str(asset.get("artifact_ref") or "").strip()
+ media_type = (
+ str(asset.get("content_type") or "").split(";", 1)[0].strip()
+ or _media_type_from_path(artifact)
+ )
+ return _try_read_image_artifact(row, artifact, media_type=media_type)
+
+
+def _try_read_image_artifact(
+ row: dict[str, Any],
+ artifact: str,
+ *,
+ media_type: str,
+) -> dict[str, Any] | None:
+ job_id = str(row.get("job_id") or "").strip()
+ storage = get_result_storage()
+ normalized = storage.normalize_artifact_ref(artifact)
+ if not job_id or not normalized:
+ _warn_skipped(row, "image", "missing artifact")
+ return None
+ try:
+ temp_path = storage.download_raw_to_temp(
+ job_id=job_id,
+ relative_path=normalized,
+ suffix=Path(normalized).suffix or ".bin",
+ temp_dir=tempfile.gettempdir(),
+ )
+ body = Path(temp_path).read_bytes()
+ except Exception as exc:
+ _warn_skipped(row, "image", str(exc))
+ return None
+ if not body:
+ _warn_skipped(row, "image", "empty image bytes")
+ return None
+ return {
+ "type": "image",
+ "media_type": media_type,
+ "data": base64.b64encode(body).decode("ascii"),
+ }
+
+
+def _select_page_asset(row: dict[str, Any]) -> dict[str, Any] | None:
+ metadata = row.get("chunk_metadata") or row.get("metadata") or {}
+ if not isinstance(metadata, dict):
+ return None
+ assets = metadata.get("page_assets") or []
+ if not isinstance(assets, list):
+ return None
+ candidates = [item for item in assets if isinstance(item, dict)]
+ if not candidates:
+ return None
+ page_nums = extract_page_nums(row) or []
+ if not page_nums:
+ return candidates[0]
+ for item in candidates:
+ try:
+ page_num = int(item.get("page_num"))
+ except (TypeError, ValueError):
+ continue
+ if page_num in page_nums:
+ return item
+ return None
+
+
+def _page_summary(row: dict[str, Any]) -> str:
+ metadata = row.get("chunk_metadata") or row.get("metadata") or {}
+ if not isinstance(metadata, dict):
+ return ""
+ return str(metadata.get("summary") or "").strip()
+
+
+def _media_type_from_path(path: str) -> str:
+ suffix = Path(path).suffix.lower()
+ return _IMAGE_MEDIA_TYPES.get(suffix, "application/octet-stream")
+
+
+def _text_part(text: str) -> dict[str, Any]:
+ return {"type": "text", "text": text}
+
+
+def _clean_text_part(part: dict[str, Any]) -> dict[str, Any]:
+ if part.get("type") != "text":
+ return part
+ return {"type": "text", "text": remove_path_placeholders(str(part.get("text") or ""))}
+
+
+def _is_empty_text_part(part: dict[str, Any]) -> bool:
+ return part.get("type") == "text" and not str(part.get("text") or "")
+
+
+def _warn_skipped(row: dict[str, Any], kind: str, reason: str) -> None:
+ logger.warning(
+ "retrieval: skipped unreachable evidence asset type={} ref={} asset_url={} source_path={} reason={}",
+ kind,
+ row.get("chunk_id"),
+ row.get("asset_url"),
+ row.get("file_path"),
+ reason,
+ )
diff --git a/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py b/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py
index 96a38fd6..f36dde34 100644
--- a/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py
+++ b/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py
@@ -7,11 +7,8 @@
from sqlalchemy.ext.asyncio import AsyncSession
-from shared.services.retrieval.hydration.asset_inline import (
- inline_assets_at_placeholders,
- strip_path_placeholders,
-)
from shared.services.retrieval.hydration.connected import hydrate_connected_target_rows
+from shared.services.retrieval.hydration.evidence_compose import compose_evidence_parts
from shared.services.retrieval.hydration.row_utils import (
extract_page_nums,
filter_excluded_rows,
@@ -74,16 +71,13 @@ async def assemble_retrieval_results(
page_nums = extract_page_nums(row)
if page_nums is not None:
assembled_row['page_nums'] = page_nums
- elif chunk_type == 'table':
- assembled_row['content'] = _compose_table_content(row, rows_by_chunk_id)
- assembled_row['content_source'] = 'summary'
- elif chunk_type == 'text':
- assembled_row['content'] = _compose_text_content(row, rows_by_chunk_id)
- assembled_row['content_source'] = 'content'
else:
assembled_row['content'] = base_content
assembled_row['content_source'] = 'content'
- assembled_row['content'] = strip_path_placeholders(assembled_row['content'])
+ assembled_row['composed'] = compose_evidence_parts(
+ assembled_row,
+ rows_by_chunk_id,
+ )
assembled.append(assembled_row)
return assembled
@@ -116,73 +110,6 @@ def _page_summary(row: dict[str, Any]) -> str:
return str(metadata.get('summary') or '').strip()
-def _compose_text_content(
- row: dict[str, Any],
- rows_by_chunk_id: dict[str, dict[str, Any]],
-) -> str:
- base_content = str(row.get('content') or '')
- display_by_target = _connected_display_by_target(row, rows_by_chunk_id)
- if not display_by_target:
- return base_content
- metadata = row.get('chunk_metadata') or row.get('metadata') or {}
- connections = (
- metadata.get('connect_to') if isinstance(metadata, dict) else None
- ) or []
- content, _embedded = inline_assets_at_placeholders(
- base_content,
- connections=connections if isinstance(connections, list) else [],
- display_by_target=display_by_target,
- )
- return content
-
-
-def _compose_table_content(
- row: dict[str, Any],
- rows_by_chunk_id: dict[str, dict[str, Any]],
-) -> str:
- parts = [_table_summary_content(row)]
- parts.extend(_connected_image_parts(row, rows_by_chunk_id))
- return '\n\n'.join(part for part in parts if part)
-
-
-def _connected_display_by_target(
- row: dict[str, Any],
- rows_by_chunk_id: dict[str, dict[str, Any]],
-) -> dict[str, str]:
- display: dict[str, str] = {}
- for target_id in iter_connected_target_ids(row):
- target_row = rows_by_chunk_id.get(target_id)
- if not target_row:
- continue
- target_type = normalize_chunk_type(target_row.get('chunk_type'))
- if target_type == 'table':
- target_content = _compose_table_content(target_row, rows_by_chunk_id)
- elif target_type == 'image':
- target_content = _image_display_content(target_row)
- else:
- continue
- if target_content:
- display[target_id] = target_content
- return display
-
-
-def _connected_image_parts(
- row: dict[str, Any],
- rows_by_chunk_id: dict[str, dict[str, Any]],
-) -> list[str]:
- parts: list[str] = []
- for target_id in iter_connected_target_ids(row):
- target_row = rows_by_chunk_id.get(target_id)
- if not target_row:
- continue
- if normalize_chunk_type(target_row.get('chunk_type')) != 'image':
- continue
- content = _image_display_content(target_row)
- if content:
- parts.append(content)
- return parts
-
-
def _image_display_content(row: dict[str, Any]) -> str:
display_ref = (
str(row.get('asset_url') or '').strip()
@@ -197,44 +124,3 @@ def _image_display_content(row: dict[str, Any]) -> str:
if description:
lines.extend(line for line in description.split('\n') if line.strip())
return '\n'.join(lines)
-
-
-def _table_summary_content(row: dict[str, Any]) -> str:
- metadata = row.get('chunk_metadata') or row.get('metadata') or {}
- if not isinstance(metadata, dict):
- metadata = {}
-
- display_ref = _table_display_ref(row)
- lines = [f"[Table: {display_ref}]" if display_ref else "[Table]"]
-
- summary = str(metadata.get('summary') or row.get('summary') or '').strip()
- if summary:
- lines.extend(line for line in summary.split('\n') if line.strip())
-
- keywords = metadata.get('keywords') or row.get('keywords') or []
- if isinstance(keywords, list):
- keyword_text = ';'.join(
- str(keyword).strip() for keyword in keywords if str(keyword).strip()
- )
- else:
- keyword_text = str(keywords or '').strip()
- if keyword_text:
- lines.append(keyword_text)
-
- caption = str(metadata.get('caption') or row.get('caption') or '').strip()
- if caption:
- lines.append(caption)
-
- return '\n'.join(lines)
-
-
-def _table_display_ref(row: dict[str, Any]) -> str:
- for key in ('asset_url', 'file_path', 'source_chunk_path'):
- value = str(row.get(key) or '').strip()
- if value:
- return value
-
- content = str(row.get('content') or '').strip()
- if content and not content.lstrip().lower().startswith(' None:
)
assert len(assembled) == 1
content = assembled[0]["content"]
- assert "[tables/" not in content
- assert content.index("见表") < content.index("[Table:")
- assert content.index("[Table:") < content.index("结束")
- assert "企业入驻信息登记模板" in content
- assert "SHOULD NOT LEAK" not in content
+ assert "[tables/table-1.html]" in content
+ assert "[Table:" not in content
+ composed_text = "".join(
+ str(part.get("text") or "")
+ for part in assembled[0]["composed"]
+ if part.get("type") == "text"
+ )
+ assert composed_text.index("见表") < composed_text.index(" list[dict[str, object]]:
)
assert [row["chunk_id"] for row in assembled] == ["text-1"]
- assert display_marker in assembled[0]["content"]
- assert "资产说明" in assembled[0]["content"]
+ assert placeholder in assembled[0]["content"]
+ assert display_marker not in assembled[0]["content"]
def test_node_unit_span_inlines_section_assets() -> None:
diff --git a/packages/shared-python/shared/tests/test_evidence_compose.py b/packages/shared-python/shared/tests/test_evidence_compose.py
new file mode 100644
index 00000000..f58014b0
--- /dev/null
+++ b/packages/shared-python/shared/tests/test_evidence_compose.py
@@ -0,0 +1,314 @@
+from __future__ import annotations
+
+import base64
+
+import pytest
+
+from shared.services.retrieval.hydration.evidence_compose import (
+ compose_evidence_parts,
+ flatten_parts,
+)
+from shared.services.retrieval.hydration.result_assembly import assemble_retrieval_results
+
+
+def test_flatten_parts_keeps_html_and_encodes_images() -> None:
+ text = flatten_parts(
+ [
+ {"type": "text", "text": "before"},
+ {"type": "image", "media_type": "image/png", "data": "abc"},
+ {"type": "text", "text": "after"},
+ ]
+ )
+ assert text == "beforedata:image/png;base64,abcafter"
+
+
+def test_page_parts_are_summary_then_page_image(monkeypatch) -> None:
+ monkeypatch.setattr(
+ "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact",
+ lambda row, artifact, media_type: {
+ "type": "image",
+ "media_type": media_type,
+ "data": base64.b64encode(b"page").decode("ascii"),
+ },
+ )
+ parts = compose_evidence_parts(
+ {
+ "chunk_type": "page",
+ "chunk_metadata": {
+ "summary": "制度摘要",
+ "page_nums": [4],
+ "page_assets": [
+ {
+ "page_num": 4,
+ "artifact_ref": "page_citation_assets/page-4.png",
+ "content_type": "image/png",
+ }
+ ],
+ },
+ },
+ {},
+ )
+ assert parts[0] == {"type": "text", "text": "制度摘要"}
+ assert parts[1]["type"] == "image"
+ assert parts[1]["media_type"] == "image/png"
+
+
+def test_page_parts_do_not_inline_connected_charts() -> None:
+ parts = compose_evidence_parts(
+ {
+ "chunk_type": "page",
+ "content": "RAW",
+ "chunk_metadata": {
+ "summary": "只要摘要",
+ "connect_to": [
+ {
+ "target": "table-1",
+ "relation": "embeds",
+ "ref": "[tables/a.html]",
+ }
+ ],
+ },
+ },
+ {
+ "table-1": {
+ "chunk_type": "table",
+ "content": "",
+ }
+ },
+ )
+ assert parts == [{"type": "text", "text": "只要摘要"}]
+
+
+def test_standalone_table_uses_html() -> None:
+ parts = compose_evidence_parts(
+ {
+ "chunk_type": "table",
+ "content": "",
+ "file_path": "tables/a.html",
+ },
+ {},
+ )
+ assert parts[0]["type"] == "text"
+ assert "" in parts[0]["text"]
+
+
+def test_standalone_image_is_bytes(monkeypatch) -> None:
+ monkeypatch.setattr(
+ "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact",
+ lambda row, artifact, media_type: {
+ "type": "image",
+ "media_type": "image/jpeg",
+ "data": "Zm9v",
+ },
+ )
+ parts = compose_evidence_parts(
+ {
+ "chunk_type": "image",
+ "content": "chart caption",
+ "file_path": "images/a.jpg",
+ "job_id": "job-1",
+ },
+ {},
+ )
+ assert parts[0] == {"type": "text", "text": "chart caption"}
+ assert parts[1] == {"type": "image", "media_type": "image/jpeg", "data": "Zm9v"}
+
+
+@pytest.mark.asyncio
+async def test_text_result_keeps_placeholders_and_composes_assets(monkeypatch) -> None:
+ monkeypatch.setattr(
+ "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact",
+ lambda row, artifact, media_type: {
+ "type": "image",
+ "media_type": "image/png",
+ "data": "aW1n",
+ },
+ )
+ assembled = await assemble_retrieval_results(
+ rows=[
+ {
+ "chunk_id": "text-1",
+ "chunk_type": "text",
+ "content": "见表 [tables/a.html] 再看 [images/a.png] 结束",
+ "chunk_metadata": {
+ "connect_to": [
+ {
+ "target": "table-1",
+ "relation": "embeds",
+ "ref": "[tables/a.html]",
+ },
+ {
+ "target": "image-1",
+ "relation": "embeds",
+ "ref": "[images/a.png]",
+ },
+ ]
+ },
+ },
+ {
+ "chunk_id": "table-1",
+ "chunk_type": "table",
+ "content": "",
+ "file_path": "tables/a.html",
+ },
+ {
+ "chunk_id": "image-1",
+ "chunk_type": "image",
+ "file_path": "images/a.png",
+ "job_id": "job-1",
+ },
+ ],
+ exclude_document_ids=[],
+ exclude_sections=[],
+ )
+ assert assembled[0]["content"] == "见表 [tables/a.html] 再看 [images/a.png] 结束"
+ types = [part["type"] for part in assembled[0]["composed"]]
+ assert "image" in types
+ composed_text = "".join(
+ part["text"] for part in assembled[0]["composed"] if part["type"] == "text"
+ )
+ assert "" in composed_text
+ assert "[tables/" not in composed_text
+ assert "[images/" not in composed_text
+
+
+def test_text_image_keeps_newlines_around_image(monkeypatch) -> None:
+ monkeypatch.setattr(
+ "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact",
+ lambda row, artifact, media_type: {
+ "type": "image",
+ "media_type": "image/png",
+ "data": "aW1n",
+ },
+ )
+ parts = compose_evidence_parts(
+ {
+ "chunk_type": "text",
+ "content": "前 [images/a.png] 后",
+ "chunk_metadata": {
+ "connect_to": [
+ {
+ "target": "image-1",
+ "relation": "embeds",
+ "ref": "[images/a.png]",
+ }
+ ]
+ },
+ },
+ {
+ "image-1": {
+ "chunk_type": "image",
+ "file_path": "images/a.png",
+ "job_id": "job-1",
+ }
+ },
+ )
+ assert parts[0] == {"type": "text", "text": "前 \n"}
+ assert parts[1] == {"type": "image", "media_type": "image/png", "data": "aW1n"}
+ assert parts[2] == {"type": "text", "text": "\n"}
+ assert parts[3] == {"type": "text", "text": " 后"}
+
+
+def test_unreachable_assets_clear_placeholders(monkeypatch) -> None:
+ monkeypatch.setattr(
+ "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact",
+ lambda row, artifact, media_type: None,
+ )
+ monkeypatch.setattr(
+ "shared.services.retrieval.hydration.evidence_compose._try_read_table_html",
+ lambda row: None,
+ )
+ parts = compose_evidence_parts(
+ {
+ "chunk_type": "text",
+ "content": "见表 [tables/a.html] 再看 [images/a.png] 结束",
+ "chunk_metadata": {
+ "connect_to": [
+ {
+ "target": "table-1",
+ "relation": "embeds",
+ "ref": "[tables/a.html]",
+ },
+ {
+ "target": "image-1",
+ "relation": "embeds",
+ "ref": "[images/a.png]",
+ },
+ ]
+ },
+ },
+ {
+ "table-1": {"chunk_type": "table", "file_path": "tables/a.html"},
+ "image-1": {
+ "chunk_type": "image",
+ "file_path": "images/a.png",
+ "job_id": "job-1",
+ },
+ },
+ )
+ composed_text = "".join(part["text"] for part in parts if part["type"] == "text")
+ assert composed_text == "见表 再看 结束"
+ assert all(part["type"] != "image" for part in parts)
+
+
+def test_missing_placeholder_does_not_append_asset(monkeypatch) -> None:
+ monkeypatch.setattr(
+ "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact",
+ lambda row, artifact, media_type: {
+ "type": "image",
+ "media_type": "image/png",
+ "data": "aW1n",
+ },
+ )
+ parts = compose_evidence_parts(
+ {
+ "chunk_type": "text",
+ "content": "没有占位符",
+ "chunk_metadata": {
+ "connect_to": [
+ {
+ "target": "image-1",
+ "relation": "embeds",
+ "ref": "[images/a.png]",
+ }
+ ]
+ },
+ },
+ {
+ "image-1": {
+ "chunk_type": "image",
+ "file_path": "images/a.png",
+ "job_id": "job-1",
+ }
+ },
+ )
+ assert parts == [{"type": "text", "text": "没有占位符"}]
+
+
+def test_page_image_requires_matching_page_num(monkeypatch) -> None:
+ monkeypatch.setattr(
+ "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact",
+ lambda row, artifact, media_type: {
+ "type": "image",
+ "media_type": media_type,
+ "data": "cGFnZQ==",
+ },
+ )
+ parts = compose_evidence_parts(
+ {
+ "chunk_type": "page",
+ "chunk_metadata": {
+ "summary": "只要摘要",
+ "page_nums": [4],
+ "page_assets": [
+ {
+ "page_num": 9,
+ "artifact_ref": "page_citation_assets/page-9.png",
+ "content_type": "image/png",
+ }
+ ],
+ },
+ },
+ {},
+ )
+ assert parts == [{"type": "text", "text": "只要摘要"}]
From d7002fcc53b95d00b9014337d0e383fb067437df Mon Sep 17 00:00:00 2001
From: chengke <404835780@qq.com>
Date: Mon, 21 Sep 2026 09:46:31 +0800
Subject: [PATCH 2/4] refactor(retrieval): streamline evidence rendering and
enhance image handling
- Removed the deprecated `render_evidence_blocks` function and replaced it with a new internal function `_render_evidence_blocks` for improved clarity and functionality.
- Updated the `_try_read_page_image` and `_select_page_asset` functions to return additional warnings for better error handling when page images are unavailable or mismatched.
- Refactored the `flatten_parts` function to utilize the new `page_summary` utility for summarizing page content.
- Enhanced tests to validate the new image handling logic and ensure proper warnings are displayed when images are missing or do not match expected page numbers.
---
deprecated/mapnav/nav/nav_compose.py | 36 ++++++++++++--
.../retrieval/agent_tools/tools/read.py | 4 +-
.../services/retrieval/execution/routes.py | 2 +
.../retrieval/hydration/evidence_compose.py | 40 ++++++++--------
.../retrieval/hydration/evidence_text.py | 45 -----------------
.../retrieval/hydration/result_assembly.py | 10 +---
.../services/retrieval/hydration/row_utils.py | 7 +++
.../shared/tests/test_evidence_compose.py | 48 ++++++++++++++++++-
8 files changed, 111 insertions(+), 81 deletions(-)
delete mode 100644 packages/shared-python/shared/services/retrieval/hydration/evidence_text.py
diff --git a/deprecated/mapnav/nav/nav_compose.py b/deprecated/mapnav/nav/nav_compose.py
index e23a5f82..80d9f9a9 100644
--- a/deprecated/mapnav/nav/nav_compose.py
+++ b/deprecated/mapnav/nav/nav_compose.py
@@ -12,8 +12,6 @@
from dataclasses import dataclass
from typing import Any, Dict, List, Optional, Sequence, Set, Tuple
-from shared.services.retrieval.hydration.evidence_text import render_evidence_blocks
-
from ._compat import Chunk
from ._compat import line_node_id
from ._compat import ToolSpace
@@ -322,14 +320,44 @@ def _render_group(
*,
evidence_index: int,
) -> str:
- """Render one evidence block via the shared evidence renderer."""
+ """Render one evidence block as ``[E#]`` + path + bodies."""
bodies = [_chunk_body(child.chunk) for child in selected]
- return render_evidence_blocks(
+ return _render_evidence_blocks(
[(group.parent_title or "", bodies)],
start_index=evidence_index,
)
+def _render_evidence_blocks(
+ groups: Sequence[tuple[str, Sequence[str]]],
+ *,
+ start_index: int = 1,
+) -> str:
+ parts: list[str] = []
+ index = max(1, int(start_index or 1))
+ for path, bodies in groups:
+ texts = [str(t or "").strip() for t in bodies]
+ texts = [t for t in texts if t]
+ if not texts:
+ continue
+ block: list[str] = [f"[E{index + len(parts)}]"]
+ header = str(path or "").strip()
+ if header:
+ block.append(f"[§ {header}]")
+ indent = len(texts) >= 2
+ for text in texts:
+ if indent:
+ block.append(
+ "\n".join(
+ (" " + ln if ln.strip() else ln) for ln in text.splitlines()
+ )
+ )
+ else:
+ block.append(text)
+ parts.append("\n".join(block).strip())
+ return "\n\n".join(parts)
+
+
def _scored_flat(groups: Sequence[_ParentGroup]) -> List[Tuple[Chunk, float]]:
out: List[Tuple[Chunk, float]] = []
for g in groups:
diff --git a/packages/shared-python/shared/services/retrieval/agent_tools/tools/read.py b/packages/shared-python/shared/services/retrieval/agent_tools/tools/read.py
index d219d1a8..ea7f946a 100644
--- a/packages/shared-python/shared/services/retrieval/agent_tools/tools/read.py
+++ b/packages/shared-python/shared/services/retrieval/agent_tools/tools/read.py
@@ -1,8 +1,8 @@
"""``corpus.read`` — full body content for already-located sections/chunks.
Unlike ``hydration.result_assembly.assemble_retrieval_results`` (which
-down-weights ``page`` chunks to their summary — see that module's
-``_page_summary``, a deliberate trade-off for the retrieval-answer surface),
+down-weights ``page`` chunks to their summary — see ``page_summary``, a
+deliberate trade-off for the retrieval-answer surface),
``read`` returns the page chunk's full body content, with ``[SAME-AS
p]`` markers resolved to the owner section's text (§2 of
``CORPUS_SCHEMA.md``) rather than stripped or summarized. ``connect_to``
diff --git a/packages/shared-python/shared/services/retrieval/execution/routes.py b/packages/shared-python/shared/services/retrieval/execution/routes.py
index dde8a9eb..4f206196 100644
--- a/packages/shared-python/shared/services/retrieval/execution/routes.py
+++ b/packages/shared-python/shared/services/retrieval/execution/routes.py
@@ -50,6 +50,8 @@ def open_agent_explore_database_context() -> AbstractAsyncContextManager[AsyncSe
def _evidence_fields(rows: list[dict]) -> dict:
+ # TODO: 后面用 TypeSafe JEV 补结果重排/筛选。现在没有这一步。
+ # 无论怎么做,发出去的 evidence 和 results 都是已经筛过或重排过的完整列表,不在 Knowhere 外面做。
evidence = collect_evidence(rows)
return {
"evidence": evidence,
diff --git a/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py b/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py
index d6638332..e5cc50a4 100644
--- a/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py
+++ b/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py
@@ -20,6 +20,7 @@
from shared.services.retrieval.hydration.row_utils import (
extract_page_nums,
normalize_chunk_type,
+ page_summary,
)
from shared.services.retrieval.hydration.table_grid import (
TableDownloadError,
@@ -80,12 +81,14 @@ def flatten_parts(parts: list[dict[str, Any]] | None) -> str:
def _compose_page_parts(row: dict[str, Any]) -> list[dict[str, Any]]:
parts: list[dict[str, Any]] = []
- summary = _page_summary(row)
+ summary = page_summary(row)
if summary:
parts.append(_text_part(summary))
- image = _try_read_page_image(row)
+ image, warning = _try_read_page_image(row)
if image is not None:
parts.append(image)
+ if warning:
+ parts.append(_text_part(f"Page image unavailable: {warning}"))
return parts
@@ -250,16 +253,20 @@ def _try_read_image(row: dict[str, Any]) -> dict[str, Any] | None:
return _try_read_image_artifact(row, artifact, media_type=_media_type_from_path(artifact))
-def _try_read_page_image(row: dict[str, Any]) -> dict[str, Any] | None:
- asset = _select_page_asset(row)
+def _try_read_page_image(row: dict[str, Any]) -> tuple[dict[str, Any] | None, str | None]:
+ asset, warning = _select_page_asset(row)
if asset is None:
- return None
+ _warn_skipped(row, "image", warning)
+ return None, warning
artifact = str(asset.get("artifact_ref") or "").strip()
media_type = (
str(asset.get("content_type") or "").split(";", 1)[0].strip()
or _media_type_from_path(artifact)
)
- return _try_read_image_artifact(row, artifact, media_type=media_type)
+ image = _try_read_image_artifact(row, artifact, media_type=media_type)
+ if image is None:
+ return None, "could not read page image"
+ return image, None
def _try_read_image_artifact(
@@ -295,34 +302,27 @@ def _try_read_image_artifact(
}
-def _select_page_asset(row: dict[str, Any]) -> dict[str, Any] | None:
+def _select_page_asset(row: dict[str, Any]) -> tuple[dict[str, Any] | None, str | None]:
metadata = row.get("chunk_metadata") or row.get("metadata") or {}
if not isinstance(metadata, dict):
- return None
+ return None, "missing page image"
assets = metadata.get("page_assets") or []
if not isinstance(assets, list):
- return None
+ return None, "missing page image"
candidates = [item for item in assets if isinstance(item, dict)]
if not candidates:
- return None
+ return None, "missing page image"
page_nums = extract_page_nums(row) or []
if not page_nums:
- return candidates[0]
+ return None, "missing page number"
for item in candidates:
try:
page_num = int(item.get("page_num"))
except (TypeError, ValueError):
continue
if page_num in page_nums:
- return item
- return None
-
-
-def _page_summary(row: dict[str, Any]) -> str:
- metadata = row.get("chunk_metadata") or row.get("metadata") or {}
- if not isinstance(metadata, dict):
- return ""
- return str(metadata.get("summary") or "").strip()
+ return item, None
+ return None, "page image does not match this page"
def _media_type_from_path(path: str) -> str:
diff --git a/packages/shared-python/shared/services/retrieval/hydration/evidence_text.py b/packages/shared-python/shared/services/retrieval/hydration/evidence_text.py
deleted file mode 100644
index a868302e..00000000
--- a/packages/shared-python/shared/services/retrieval/hydration/evidence_text.py
+++ /dev/null
@@ -1,45 +0,0 @@
-"""Shared evidence_text rendering: one ``[E#]`` block per group.
-
-Each block is ``[E#]`` + ``[§ path]`` (full traceable path) + body lines.
-Groups are caller-provided; bodies in one group stay in that group.
-"""
-
-from __future__ import annotations
-
-from typing import Sequence
-
-
-def render_evidence_blocks(
- groups: Sequence[tuple[str, Sequence[str]]],
- *,
- start_index: int = 1,
-) -> str:
- """Render (path, bodies) groups into evidence_text.
-
- ``path`` is the full traceable header (e.g. file / section chain).
- Bodies are joined with newlines; multiple bodies in one group are indented.
- ``start_index`` sets the first ``[E#]`` number (default 1).
- """
- parts: list[str] = []
- index = max(1, int(start_index or 1))
- for path, bodies in groups:
- texts = [str(t or "").strip() for t in bodies]
- texts = [t for t in texts if t]
- if not texts:
- continue
- block: list[str] = [f"[E{index + len(parts)}]"]
- header = str(path or "").strip()
- if header:
- block.append(f"[§ {header}]")
- indent = len(texts) >= 2
- for text in texts:
- if indent:
- block.append(
- "\n".join(
- (" " + ln if ln.strip() else ln) for ln in text.splitlines()
- )
- )
- else:
- block.append(text)
- parts.append("\n".join(block).strip())
- return "\n\n".join(parts)
diff --git a/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py b/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py
index f36dde34..1cc3219f 100644
--- a/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py
+++ b/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py
@@ -14,6 +14,7 @@
filter_excluded_rows,
iter_connected_target_ids,
normalize_chunk_type,
+ page_summary,
)
@@ -66,7 +67,7 @@ async def assemble_retrieval_results(
base_content = str(row.get('content') or '')
chunk_type = normalize_chunk_type(row.get('chunk_type'))
if chunk_type == 'page':
- assembled_row['content'] = _page_summary(row)
+ assembled_row['content'] = page_summary(row)
assembled_row['content_source'] = 'summary'
page_nums = extract_page_nums(row)
if page_nums is not None:
@@ -103,13 +104,6 @@ def _filter_rows_by_allowed_chunk_types(
]
-def _page_summary(row: dict[str, Any]) -> str:
- metadata = row.get('chunk_metadata') or row.get('metadata') or {}
- if not isinstance(metadata, dict):
- return ''
- return str(metadata.get('summary') or '').strip()
-
-
def _image_display_content(row: dict[str, Any]) -> str:
display_ref = (
str(row.get('asset_url') or '').strip()
diff --git a/packages/shared-python/shared/services/retrieval/hydration/row_utils.py b/packages/shared-python/shared/services/retrieval/hydration/row_utils.py
index 2a83277b..8e676fe7 100644
--- a/packages/shared-python/shared/services/retrieval/hydration/row_utils.py
+++ b/packages/shared-python/shared/services/retrieval/hydration/row_utils.py
@@ -41,6 +41,13 @@ def extract_page_nums(row: dict[str, Any]) -> list[int] | None:
return page_nums if isinstance(page_nums, list) else None
+def page_summary(row: dict[str, Any]) -> str:
+ metadata = row.get('chunk_metadata') or row.get('metadata') or {}
+ if not isinstance(metadata, dict):
+ return ''
+ return str(metadata.get('summary') or '').strip()
+
+
def build_reference_lookup_key(
*,
document_id: object,
diff --git a/packages/shared-python/shared/tests/test_evidence_compose.py b/packages/shared-python/shared/tests/test_evidence_compose.py
index f58014b0..c8387c68 100644
--- a/packages/shared-python/shared/tests/test_evidence_compose.py
+++ b/packages/shared-python/shared/tests/test_evidence_compose.py
@@ -76,7 +76,11 @@ def test_page_parts_do_not_inline_connected_charts() -> None:
}
},
)
- assert parts == [{"type": "text", "text": "只要摘要"}]
+ assert parts[0] == {"type": "text", "text": "只要摘要"}
+ assert parts[1] == {
+ "type": "text",
+ "text": "Page image unavailable: missing page image",
+ }
def test_standalone_table_uses_html() -> None:
@@ -311,4 +315,44 @@ def test_page_image_requires_matching_page_num(monkeypatch) -> None:
},
{},
)
- assert parts == [{"type": "text", "text": "只要摘要"}]
+ assert parts == [
+ {"type": "text", "text": "只要摘要"},
+ {
+ "type": "text",
+ "text": "Page image unavailable: page image does not match this page",
+ },
+ ]
+
+
+def test_page_image_missing_page_number_does_not_use_first_asset(monkeypatch) -> None:
+ monkeypatch.setattr(
+ "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact",
+ lambda row, artifact, media_type: {
+ "type": "image",
+ "media_type": media_type,
+ "data": "cGFnZQ==",
+ },
+ )
+ parts = compose_evidence_parts(
+ {
+ "chunk_type": "page",
+ "chunk_metadata": {
+ "summary": "只要摘要",
+ "page_assets": [
+ {
+ "page_num": 1,
+ "artifact_ref": "page_citation_assets/page-1.png",
+ "content_type": "image/png",
+ }
+ ],
+ },
+ },
+ {},
+ )
+ assert parts == [
+ {"type": "text", "text": "只要摘要"},
+ {
+ "type": "text",
+ "text": "Page image unavailable: missing page number",
+ },
+ ]
From 7f161c734c59f049619d4bc1771e127e86481230 Mon Sep 17 00:00:00 2001
From: chengke <404835780@qq.com>
Date: Mon, 21 Sep 2026 10:29:24 +0800
Subject: [PATCH 3/4] fix(retrieval): keep result placeholders and emit MCP
results
Public results stay raw for debug; composed parts live only on evidence. MCP query now returns the same package.
Co-authored-by: Cursor
---
apps/api/app/api/v1/routes/retrieval.py | 5 +-
apps/api/app/mcp/retrieval_server.py | 14 ++--
.../test_evidence_renderer_contract.py | 65 +++++++++++++++++++
.../contract/test_retrieval_document_scope.py | 3 +-
.../services/retrieval/hydration/row_utils.py | 1 -
5 files changed, 80 insertions(+), 8 deletions(-)
diff --git a/apps/api/app/api/v1/routes/retrieval.py b/apps/api/app/api/v1/routes/retrieval.py
index ecc49e4a..994f53b2 100644
--- a/apps/api/app/api/v1/routes/retrieval.py
+++ b/apps/api/app/api/v1/routes/retrieval.py
@@ -170,7 +170,10 @@ class RetrievalQueryResponse(BaseModel):
),
)
referenced_chunks: list[dict] = Field(default_factory=list)
- results: list[dict] = Field(default_factory=list)
+ results: list[dict] = Field(
+ default_factory=list,
+ description="Raw path chunks for debug. Content keeps placeholders; composed parts live on evidence.",
+ )
stop_reason: str | None = None
failure_reason: str | None = None
decision_trace: list[dict] | None = Field(
diff --git a/apps/api/app/mcp/retrieval_server.py b/apps/api/app/mcp/retrieval_server.py
index 301ba92c..da806746 100644
--- a/apps/api/app/mcp/retrieval_server.py
+++ b/apps/api/app/mcp/retrieval_server.py
@@ -54,11 +54,12 @@ def resolve_mcp_namespace(*, ctx: Context | None) -> str:
def to_mcp_query_response(response: dict[str, Any]) -> dict[str, Any]:
- """Project the internal retrieval response to the MCP agent contract.
+ """Project the public retrieval response to the MCP agent contract.
- MCP returns composed evidence plus the text projection and citations:
- - evidence: composed parts (text and inline images)
+ MCP returns the same Knowhere package the HTTP query emits:
+ - evidence: composed parts (text/HTML and inline images)
- evidence_text: text projection of those parts
+ - results: raw path chunks for debug
- referenced_chunks: structured chunk references for citation / follow-up
- decision_trace: navigation decisions including terminal stop/failure
"""
@@ -66,6 +67,7 @@ def to_mcp_query_response(response: dict[str, Any]) -> dict[str, Any]:
"query": response.get("query"),
"evidence": response.get("evidence") or [],
"evidence_text": response.get("evidence_text") or "",
+ "results": response.get("results") or [],
"referenced_chunks": response.get("referenced_chunks") or [],
"decision_trace": response.get("decision_trace") or [],
}
@@ -103,8 +105,10 @@ def create_retrieval_mcp_server(
@server.tool(
name="retrieval.query",
description=(
- "Search published documents. Returns evidence (composed parts for "
- "LLM consumption), evidence_text (text projection of those parts), "
+ "Search published documents. Compose already happened in Knowhere. "
+ "Returns evidence (composed parts for LLM consumption), "
+ "results (raw path chunks for debug), "
+ "evidence_text (text projection of evidence), "
"referenced_chunks (cited chunk metadata for follow-up queries), "
"and decision_trace (navigation decisions including stop/failure reasons). "
"Include navigation intent directly in your query text — the "
diff --git a/apps/api/tests/contract/test_evidence_renderer_contract.py b/apps/api/tests/contract/test_evidence_renderer_contract.py
index b023a8df..da2cac7b 100644
--- a/apps/api/tests/contract/test_evidence_renderer_contract.py
+++ b/apps/api/tests/contract/test_evidence_renderer_contract.py
@@ -1,3 +1,9 @@
+import asyncio
+
+from app.mcp.retrieval_server import to_mcp_query_response
+from shared.services.retrieval.execution.response_projection import (
+ project_public_retrieval_response,
+)
from shared.services.retrieval.execution.routes import _evidence_fields
@@ -24,3 +30,62 @@ def test_evidence_fields_flatten_composed_parts_in_result_order() -> None:
assert fields["evidence_text"] == (
"firstmid data:image/png;base64,abc"
)
+
+
+def test_mcp_query_response_keeps_evidence_and_debug_results() -> None:
+ response = to_mcp_query_response(
+ {
+ "query": "q",
+ "evidence": [{"type": "text", "text": "t"}],
+ "evidence_text": "t",
+ "results": [
+ {
+ "content": "[images/a.png]",
+ }
+ ],
+ "referenced_chunks": [{"chunk_id": "c1"}],
+ "decision_trace": [{"step": 1}],
+ "answer_text": "should drop",
+ }
+ )
+
+ assert response == {
+ "query": "q",
+ "evidence": [{"type": "text", "text": "t"}],
+ "evidence_text": "t",
+ "results": [
+ {
+ "content": "[images/a.png]",
+ }
+ ],
+ "referenced_chunks": [{"chunk_id": "c1"}],
+ "decision_trace": [{"step": 1}],
+ }
+
+
+def test_public_results_keep_placeholders_without_composed() -> None:
+ public = asyncio.run(
+ project_public_retrieval_response(
+ {
+ "namespace": "default",
+ "query": "q",
+ "router_used": "classic",
+ "evidence": [{"type": "text", "text": ""}],
+ "evidence_text": "",
+ "results": [
+ {
+ "chunk_id": "c1",
+ "chunk_type": "text",
+ "content": "见表 [tables/a.html]",
+ "composed": [{"type": "text", "text": "见表 "}],
+ "score": 1,
+ "document_id": "d1",
+ }
+ ],
+ }
+ )
+ )
+
+ assert public["evidence"] == [{"type": "text", "text": ""}]
+ assert public["results"][0]["content"] == "见表 [tables/a.html]"
+ assert "composed" not in public["results"][0]
diff --git a/apps/api/tests/contract/test_retrieval_document_scope.py b/apps/api/tests/contract/test_retrieval_document_scope.py
index 3f15c1af..9329414d 100644
--- a/apps/api/tests/contract/test_retrieval_document_scope.py
+++ b/apps/api/tests/contract/test_retrieval_document_scope.py
@@ -387,7 +387,8 @@ async def test_connected_hydration_and_assembly_reject_outside_rows(
)
assert {r["document_id"] for r in assembled} == expected
for row in assembled:
- assert "asset secret" in row["content"]
+ # content stays raw for debug; compose keeps the placeholder intact.
+ assert "[images/" in row["content"]
def test_cache_scope_none_empty_and_set_identity():
diff --git a/packages/shared-python/shared/services/retrieval/hydration/row_utils.py b/packages/shared-python/shared/services/retrieval/hydration/row_utils.py
index 8e676fe7..b6bd1b14 100644
--- a/packages/shared-python/shared/services/retrieval/hydration/row_utils.py
+++ b/packages/shared-python/shared/services/retrieval/hydration/row_utils.py
@@ -12,7 +12,6 @@
'chunk_type',
'content',
'content_source',
- 'composed',
'score',
'asset_url',
'source_chunk_path',
From ae653026cf664628b8e7a7fcd101a612032d83ce Mon Sep 17 00:00:00 2001
From: chengke <404835780@qq.com>
Date: Mon, 21 Sep 2026 10:29:49 +0800
Subject: [PATCH 4/4] fix(retrieval): satisfy pyright on page image selection
Co-authored-by: Cursor
---
.../services/retrieval/hydration/evidence_compose.py | 10 +++++++---
1 file changed, 7 insertions(+), 3 deletions(-)
diff --git a/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py b/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py
index e5cc50a4..c5cb9746 100644
--- a/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py
+++ b/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py
@@ -256,8 +256,9 @@ def _try_read_image(row: dict[str, Any]) -> dict[str, Any] | None:
def _try_read_page_image(row: dict[str, Any]) -> tuple[dict[str, Any] | None, str | None]:
asset, warning = _select_page_asset(row)
if asset is None:
- _warn_skipped(row, "image", warning)
- return None, warning
+ reason = warning or "missing page image"
+ _warn_skipped(row, "image", reason)
+ return None, reason
artifact = str(asset.get("artifact_ref") or "").strip()
media_type = (
str(asset.get("content_type") or "").split(";", 1)[0].strip()
@@ -316,8 +317,11 @@ def _select_page_asset(row: dict[str, Any]) -> tuple[dict[str, Any] | None, str
if not page_nums:
return None, "missing page number"
for item in candidates:
+ raw_page_num = item.get("page_num")
+ if raw_page_num is None:
+ continue
try:
- page_num = int(item.get("page_num"))
+ page_num = int(raw_page_num)
except (TypeError, ValueError):
continue
if page_num in page_nums: