From 57ad57c4392ea0bbd7e404830d016211c3c18019 Mon Sep 17 00:00:00 2001 From: chengke <404835780@qq.com> Date: Sun, 20 Sep 2026 23:08:41 +0800 Subject: [PATCH 1/4] feat(retrieval): enhance evidence handling and response structure - Added a new `evidence` field to the `RetrievalQueryResponse` model to include composed evidence parts (text and inline images). - Updated the MCP response projection to return the new `evidence` field alongside `evidence_text`. - Refactored evidence rendering logic to flatten composed parts and improve the assembly of retrieval results. - Adjusted tests to validate the new evidence structure and ensure proper handling of composed evidence in responses. --- apps/api/app/api/v1/routes/retrieval.py | 6 +- apps/api/app/mcp/retrieval_server.py | 14 +- .../test_evidence_renderer_contract.py | 56 +-- .../test_page_memory_retrieval_contract.py | 16 +- .../services/retrieval/execution/plan.py | 1 + .../execution/response_projection.py | 1 + .../services/retrieval/execution/routes.py | 45 +-- .../retrieval/hydration/asset_inline.py | 9 +- .../retrieval/hydration/evidence_compose.py | 355 ++++++++++++++++++ .../retrieval/hydration/result_assembly.py | 124 +----- .../services/retrieval/hydration/row_utils.py | 1 + .../retrieval/hydration/table_grid.py | 4 +- .../shared/tests/test_asset_inline.py | 20 +- .../shared/tests/test_evidence_compose.py | 314 ++++++++++++++++ 14 files changed, 753 insertions(+), 213 deletions(-) create mode 100644 packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py create mode 100644 packages/shared-python/shared/tests/test_evidence_compose.py diff --git a/apps/api/app/api/v1/routes/retrieval.py b/apps/api/app/api/v1/routes/retrieval.py index 087b8639..ecc49e4a 100644 --- a/apps/api/app/api/v1/routes/retrieval.py +++ b/apps/api/app/api/v1/routes/retrieval.py @@ -154,9 +154,13 @@ class RetrievalQueryResponse(BaseModel): namespace: str query: str router_used: str + evidence: list[dict] = Field( + default_factory=list, + description="Composed evidence parts (text and inline images) for downstream agents.", + ) evidence_text: str = Field( default="", - description="Hierarchical evidence text. Primary output for downstream agents.", + description="Text projection of evidence. Tables stay as HTML; images are data URLs.", ) answer_text: str = Field( default="", diff --git a/apps/api/app/mcp/retrieval_server.py b/apps/api/app/mcp/retrieval_server.py index 89845b64..301ba92c 100644 --- a/apps/api/app/mcp/retrieval_server.py +++ b/apps/api/app/mcp/retrieval_server.py @@ -56,13 +56,15 @@ def resolve_mcp_namespace(*, ctx: Context | None) -> str: def to_mcp_query_response(response: dict[str, Any]) -> dict[str, Any]: """Project the internal retrieval response to the MCP agent contract. - MCP returns exactly 3 PRIMARY fields: - - evidence_text: hierarchical evidence tree for LLM consumption + MCP returns composed evidence plus the text projection and citations: + - evidence: composed parts (text and inline images) + - evidence_text: text projection of those parts - referenced_chunks: structured chunk references for citation / follow-up - decision_trace: navigation decisions including terminal stop/failure """ return { "query": response.get("query"), + "evidence": response.get("evidence") or [], "evidence_text": response.get("evidence_text") or "", "referenced_chunks": response.get("referenced_chunks") or [], "decision_trace": response.get("decision_trace") or [], @@ -101,10 +103,10 @@ def create_retrieval_mcp_server( @server.tool( name="retrieval.query", description=( - "Search published documents. Returns evidence_text (hierarchical " - "evidence for LLM consumption), referenced_chunks (cited chunk " - "metadata for follow-up queries), and decision_trace (navigation " - "decisions including stop/failure reasons). " + "Search published documents. Returns evidence (composed parts for " + "LLM consumption), evidence_text (text projection of those parts), " + "referenced_chunks (cited chunk metadata for follow-up queries), " + "and decision_trace (navigation decisions including stop/failure reasons). " "Include navigation intent directly in your query text — the " "engine will automatically locate the right documents and sections." ), diff --git a/apps/api/tests/contract/test_evidence_renderer_contract.py b/apps/api/tests/contract/test_evidence_renderer_contract.py index a714ff54..b023a8df 100644 --- a/apps/api/tests/contract/test_evidence_renderer_contract.py +++ b/apps/api/tests/contract/test_evidence_renderer_contract.py @@ -1,46 +1,26 @@ -from shared.services.retrieval.execution.routes import _render_rows_evidence +from shared.services.retrieval.execution.routes import _evidence_fields -def test_render_rows_evidence_should_group_siblings_under_parent_path() -> None: +def test_evidence_fields_flatten_composed_parts_in_result_order() -> None: rows = [ + {"composed": [{"type": "text", "text": "first"}]}, { - "chunk_id": "c2", - "content": "second section content", - "sort_order": 2, - "source": { - "source_file_name": "alpha.pdf", - "section_path": "Alpha / Two", - }, - }, - { - "chunk_id": "c1", - "content": "first section content\nwith more detail", - "sort_order": 1, - "source": { - "source_file_name": "alpha.pdf", - "section_path": "Alpha / One", - }, - }, - { - "chunk_id": "c3", - "content": "
metric
", - "source_file_name": "beta.pdf", - "section_path": "Beta / Table", + "composed": [ + {"type": "text", "text": "mid "}, + {"type": "image", "media_type": "image/png", "data": "abc"}, + ] }, + {"composed": [{"type": "text", "text": "
metric
"}]}, ] - evidence_text = _render_rows_evidence(rows) + fields = _evidence_fields(rows) - assert evidence_text.count("[E1]") == 1 - assert evidence_text.count("[E2]") == 1 - assert "[E3]" not in evidence_text - assert "[§ alpha.pdf / Alpha]" in evidence_text - assert "[§ beta.pdf / Beta]" in evidence_text - assert "[§ alpha.pdf / Alpha / One]" not in evidence_text - assert "[§ alpha.pdf / Alpha / Two]" not in evidence_text - assert "first section content" in evidence_text - assert "second section content" in evidence_text - assert "
metric
" in evidence_text - assert "[Document]" not in evidence_text - assert "▸" not in evidence_text - assert "┈" not in evidence_text + assert fields["evidence"] == [ + {"type": "text", "text": "first"}, + {"type": "text", "text": "mid "}, + {"type": "image", "media_type": "image/png", "data": "abc"}, + {"type": "text", "text": "
metric
"}, + ] + assert fields["evidence_text"] == ( + "firstmid data:image/png;base64,abc
metric
" + ) diff --git a/apps/worker/tests/contract/test_page_memory_retrieval_contract.py b/apps/worker/tests/contract/test_page_memory_retrieval_contract.py index e9346c7e..bfebab1d 100644 --- a/apps/worker/tests/contract/test_page_memory_retrieval_contract.py +++ b/apps/worker/tests/contract/test_page_memory_retrieval_contract.py @@ -142,13 +142,15 @@ async def test_table_result_assembly_uses_summary_not_html() -> None: assert len(assembled) == 1 content = assembled[0]["content"] - assert "[Table: https://assets.example.com/job-1/tables/table-1.html]" in content - assert "企业入驻信息登记模板" in content - assert "企业名称;统一社会信用代码" in content - assert "SHOULD NOT LEAK" not in content - assert "SHOULD NOT LEAK" in composed_text + assert "[tables/" not in composed_text + assert composed_text.index("见表") < composed_text.index(" dict[str, An "namespace": request.namespace, "query": request.query, "router_used": "empty_query_filtered", + "evidence": [], "evidence_text": "", "answer_text": "", "referenced_chunks": [], diff --git a/packages/shared-python/shared/services/retrieval/execution/response_projection.py b/packages/shared-python/shared/services/retrieval/execution/response_projection.py index 1ac29076..8878d82c 100644 --- a/packages/shared-python/shared/services/retrieval/execution/response_projection.py +++ b/packages/shared-python/shared/services/retrieval/execution/response_projection.py @@ -25,6 +25,7 @@ async def project_public_retrieval_response(response: dict[str, Any]) -> dict[st 'namespace': response.get('namespace'), 'query': response.get('query'), 'router_used': response.get('router_used'), + 'evidence': response.get('evidence') or [], 'evidence_text': response.get('evidence_text') or '', 'answer_text': '', 'referenced_chunks': response.get('referenced_chunks') or [], diff --git a/packages/shared-python/shared/services/retrieval/execution/routes.py b/packages/shared-python/shared/services/retrieval/execution/routes.py index 25eaac5a..dde8a9eb 100644 --- a/packages/shared-python/shared/services/retrieval/execution/routes.py +++ b/packages/shared-python/shared/services/retrieval/execution/routes.py @@ -13,11 +13,13 @@ RetrievalRouteContext, RetrievalRouteOutcome, ) -from shared.services.retrieval.hydration.evidence_text import render_evidence_blocks +from shared.services.retrieval.hydration.evidence_compose import ( + collect_evidence, + flatten_parts, +) from shared.services.retrieval.hydration.result_assembly import ( assemble_retrieval_results, ) -from shared.services.retrieval.search.lexical_text import split_section_path from shared.services.retrieval.search.map_unit_discovery import map_unit_discovery from shared.services.retrieval.search.ranking import rank_retrieval_candidates from shared.services.retrieval.search.scoped_corpus import ( @@ -47,29 +49,12 @@ def open_agent_explore_database_context() -> AbstractAsyncContextManager[AsyncSe ) -def _evidence_path_header(row: dict) -> str: - source = row.get("source") - if not isinstance(source, dict): - source = row - file_name = str(source.get("source_file_name") or "").strip() - section_path = str(source.get("section_path") or "").strip() - parts = split_section_path(section_path) - if len(parts) > 1: - section_path = " / ".join(parts[:-1]) - if file_name and section_path: - return f"{file_name} / {section_path}" - return file_name or section_path - - -def _render_rows_evidence(rows: list[dict]) -> str: - groups: dict[str, list[str]] = {} - for row in rows: - header = _evidence_path_header(row) - content = str(row.get("content") or "").strip() - if not content: - continue - groups.setdefault(header, []).append(content) - return render_evidence_blocks(list(groups.items())) +def _evidence_fields(rows: list[dict]) -> dict: + evidence = collect_evidence(rows) + return { + "evidence": evidence, + "evidence_text": flatten_parts(evidence), + } async def run_retrieval_route( @@ -136,7 +121,7 @@ async def _try_run_small_corpus_route( "namespace": context.namespace, "query": context.query, "router_used": "small_corpus_all", - "evidence_text": _render_rows_evidence(results), + **_evidence_fields(results), "answer_text": "", "results": results, } @@ -193,7 +178,7 @@ async def _run_classic_topk_route( "namespace": context.namespace, "query": context.query, "router_used": "classic_topk", - "evidence_text": _render_rows_evidence(results), + **_evidence_fields(results), "answer_text": "", "results": results, } @@ -316,12 +301,12 @@ async def _run_agent_explore_route( ) decision_trace = [step.to_dict() for step in decision_steps] - evidence_text = _render_rows_evidence(assembled_rows) + evidence_fields = _evidence_fields(assembled_rows) response = { "namespace": context.namespace, "query": context.query, "router_used": "agent_explore", - "evidence_text": evidence_text, + **evidence_fields, "answer_text": "", "referenced_chunks": resolved.refs, "results": assembled_rows, @@ -334,6 +319,6 @@ async def _run_agent_explore_route( completion_label="AGENT EXPLORE RETRIEVAL", completion_count=len(resolved.refs), completion_detail=( - f"chunks | evidence={len(evidence_text)} chars | router=agent_explore" + f"chunks | evidence={len(evidence_fields['evidence_text'])} chars | router=agent_explore" ), ) diff --git a/packages/shared-python/shared/services/retrieval/hydration/asset_inline.py b/packages/shared-python/shared/services/retrieval/hydration/asset_inline.py index 4c49f00d..7dd7b046 100644 --- a/packages/shared-python/shared/services/retrieval/hydration/asset_inline.py +++ b/packages/shared-python/shared/services/retrieval/hydration/asset_inline.py @@ -15,10 +15,13 @@ _SAME_AS_RE = re.compile(r"\[SAME-AS [^\]]+\]") -def strip_path_placeholders(content: str) -> str: +def remove_path_placeholders(content: str) -> str: text = _PATH_REF_RE.sub("", content) - text = _SAME_AS_RE.sub("", text) - return text.strip() + return _SAME_AS_RE.sub("", text) + + +def strip_path_placeholders(content: str) -> str: + return remove_path_placeholders(content).strip() def inline_assets_at_placeholders( diff --git a/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py b/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py new file mode 100644 index 00000000..d6638332 --- /dev/null +++ b/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py @@ -0,0 +1,355 @@ +"""Compose retrieval evidence parts from raw chunks and connected assets. + +Text and standalone image/table chunks place table HTML and image bytes at +path placeholders. Page chunks are summary plus the page image; they do not +inline connected charts. +""" + +from __future__ import annotations + +import base64 +import tempfile +from pathlib import Path +from typing import Any + +from loguru import logger + +from shared.services.retrieval.hydration.asset_inline import ( + remove_path_placeholders, +) +from shared.services.retrieval.hydration.row_utils import ( + extract_page_nums, + normalize_chunk_type, +) +from shared.services.retrieval.hydration.table_grid import ( + TableDownloadError, + load_table_html, +) +from shared.services.storage.result_storage import get_result_storage + +_IMAGE_MEDIA_TYPES = { + ".jpg": "image/jpeg", + ".jpeg": "image/jpeg", + ".png": "image/png", + ".gif": "image/gif", + ".webp": "image/webp", +} + + +def compose_evidence_parts( + row: dict[str, Any], + rows_by_chunk_id: dict[str, dict[str, Any]], +) -> list[dict[str, Any]]: + chunk_type = normalize_chunk_type(row.get("chunk_type")) + if chunk_type == "page": + return _compose_page_parts(row) + if chunk_type == "table": + return _compose_standalone_table_parts(row) + if chunk_type == "image": + return _compose_standalone_image_parts(row) + return _compose_text_parts(row, rows_by_chunk_id) + + +def collect_evidence(rows: list[dict[str, Any]]) -> list[dict[str, Any]]: + parts: list[dict[str, Any]] = [] + for row in rows: + composed = row.get("composed") + if isinstance(composed, list): + parts.extend(composed) + return parts + + +def flatten_parts(parts: list[dict[str, Any]] | None) -> str: + texts: list[str] = [] + for part in parts or []: + if not isinstance(part, dict): + continue + if part.get("type") == "text": + text = str(part.get("text") or "") + if text: + texts.append(text) + continue + if part.get("type") != "image": + continue + media_type = str(part.get("media_type") or "").strip() or "application/octet-stream" + data = str(part.get("data") or "").strip() + if data: + texts.append(f"data:{media_type};base64,{data}") + return "".join(texts) + + +def _compose_page_parts(row: dict[str, Any]) -> list[dict[str, Any]]: + parts: list[dict[str, Any]] = [] + summary = _page_summary(row) + if summary: + parts.append(_text_part(summary)) + image = _try_read_page_image(row) + if image is not None: + parts.append(image) + return parts + + +def _compose_standalone_table_parts(row: dict[str, Any]) -> list[dict[str, Any]]: + html = _try_read_table_html(row) + if html is None: + return [] + return [_text_part(f"\n{html}\n")] + + +def _compose_standalone_image_parts(row: dict[str, Any]) -> list[dict[str, Any]]: + parts: list[dict[str, Any]] = [] + description = str(row.get("content") or "").strip() + if description: + parts.append(_text_part(description)) + image = _try_read_image(row) + if image is not None: + parts.append(image) + return parts + + +def _compose_text_parts( + row: dict[str, Any], + rows_by_chunk_id: dict[str, dict[str, Any]], +) -> list[dict[str, Any]]: + content = str(row.get("content") or "") + tables, images = _embed_targets(row, rows_by_chunk_id) + for _target_id, target_row, ref in tables: + html = _try_read_table_html(target_row) + content, placed = _replace_placeholder(content, ref, "" if html is None else f"\n{html}\n") + if html is not None and not placed: + _warn_skipped(target_row, "table", "placeholder not found") + parts: list[dict[str, Any]] = [] + remaining = content + unused = list(images) + while unused: + match = _earliest_image_placeholder(remaining, unused) + if match is None: + _warn_skipped(unused[0][1], "image", "placeholder not found") + unused.pop(0) + continue + before, after, target_row = match + if before: + parts.append(_text_part(before)) + image = _try_read_image(target_row) + if image is not None: + if parts and parts[-1]["type"] == "text": + parts[-1]["text"] += "\n" + else: + parts.append(_text_part("\n")) + parts.append(image) + parts.append(_text_part("\n")) + remaining = after + unused = [item for item in unused if item[1] is not target_row] + if remaining: + parts.append(_text_part(remaining)) + cleaned: list[dict[str, Any]] = [] + for part in parts: + cleaned_part = _clean_text_part(part) + if not _is_empty_text_part(cleaned_part): + cleaned.append(cleaned_part) + return cleaned + + +def _embed_targets( + row: dict[str, Any], + rows_by_chunk_id: dict[str, dict[str, Any]], +) -> tuple[ + list[tuple[str, dict[str, Any], str]], + list[tuple[str, dict[str, Any], str]], +]: + tables: list[tuple[str, dict[str, Any], str]] = [] + images: list[tuple[str, dict[str, Any], str]] = [] + for item in _connections(row): + if item.get("relation") != "embeds": + continue + target_id = str(item.get("target") or "").strip() + ref = str(item.get("ref") or "").strip() + target_row = rows_by_chunk_id.get(target_id) + if not target_id or not ref or target_row is None: + continue + target_type = normalize_chunk_type(target_row.get("chunk_type")) + if target_type == "table": + tables.append((target_id, target_row, ref)) + elif target_type == "image": + images.append((target_id, target_row, ref)) + return tables, images + + +def _connections(row: dict[str, Any]) -> list[dict[str, Any]]: + metadata = row.get("chunk_metadata") or row.get("metadata") or {} + if not isinstance(metadata, dict): + return [] + connections = metadata.get("connect_to") or [] + if not isinstance(connections, list): + return [] + return [item for item in connections if isinstance(item, dict)] + + +def _replace_placeholder(text: str, ref: str, replacement: str) -> tuple[str, bool]: + for candidate in _ref_candidates(ref): + if candidate and candidate in text: + return text.replace(candidate, replacement, 1), True + return text, False + + +def _earliest_image_placeholder( + text: str, + images: list[tuple[str, dict[str, Any], str]], +) -> tuple[str, str, dict[str, Any]] | None: + best: tuple[int, int, dict[str, Any]] | None = None + for _target_id, target_row, ref in images: + for candidate in _ref_candidates(ref): + if not candidate: + continue + index = text.find(candidate) + if index < 0: + continue + length = len(candidate) + if ( + best is None + or index < best[0] + or (index == best[0] and length > best[1]) + ): + best = (index, length, target_row) + if best is None: + return None + index, length, target_row = best + return text[:index], text[index + length :], target_row + + +def _ref_candidates(ref: str) -> list[str]: + raw = str(ref or "").strip() + if not raw: + return [] + out = [raw] + if raw.startswith("[") and raw.endswith("]"): + inner = raw[1:-1].strip() + if inner and inner not in out: + out.append(inner) + else: + bracketed = f"[{raw}]" + if bracketed not in out: + out.append(bracketed) + return out + + +def _try_read_table_html(row: dict[str, Any]) -> str | None: + try: + html = load_table_html(row).strip() + except TableDownloadError as exc: + _warn_skipped(row, "table", str(exc)) + return None + if html: + return html + _warn_skipped(row, "table", "missing table HTML") + return None + + +def _try_read_image(row: dict[str, Any]) -> dict[str, Any] | None: + artifact = str(row.get("file_path") or "").strip() + return _try_read_image_artifact(row, artifact, media_type=_media_type_from_path(artifact)) + + +def _try_read_page_image(row: dict[str, Any]) -> dict[str, Any] | None: + asset = _select_page_asset(row) + if asset is None: + return None + artifact = str(asset.get("artifact_ref") or "").strip() + media_type = ( + str(asset.get("content_type") or "").split(";", 1)[0].strip() + or _media_type_from_path(artifact) + ) + return _try_read_image_artifact(row, artifact, media_type=media_type) + + +def _try_read_image_artifact( + row: dict[str, Any], + artifact: str, + *, + media_type: str, +) -> dict[str, Any] | None: + job_id = str(row.get("job_id") or "").strip() + storage = get_result_storage() + normalized = storage.normalize_artifact_ref(artifact) + if not job_id or not normalized: + _warn_skipped(row, "image", "missing artifact") + return None + try: + temp_path = storage.download_raw_to_temp( + job_id=job_id, + relative_path=normalized, + suffix=Path(normalized).suffix or ".bin", + temp_dir=tempfile.gettempdir(), + ) + body = Path(temp_path).read_bytes() + except Exception as exc: + _warn_skipped(row, "image", str(exc)) + return None + if not body: + _warn_skipped(row, "image", "empty image bytes") + return None + return { + "type": "image", + "media_type": media_type, + "data": base64.b64encode(body).decode("ascii"), + } + + +def _select_page_asset(row: dict[str, Any]) -> dict[str, Any] | None: + metadata = row.get("chunk_metadata") or row.get("metadata") or {} + if not isinstance(metadata, dict): + return None + assets = metadata.get("page_assets") or [] + if not isinstance(assets, list): + return None + candidates = [item for item in assets if isinstance(item, dict)] + if not candidates: + return None + page_nums = extract_page_nums(row) or [] + if not page_nums: + return candidates[0] + for item in candidates: + try: + page_num = int(item.get("page_num")) + except (TypeError, ValueError): + continue + if page_num in page_nums: + return item + return None + + +def _page_summary(row: dict[str, Any]) -> str: + metadata = row.get("chunk_metadata") or row.get("metadata") or {} + if not isinstance(metadata, dict): + return "" + return str(metadata.get("summary") or "").strip() + + +def _media_type_from_path(path: str) -> str: + suffix = Path(path).suffix.lower() + return _IMAGE_MEDIA_TYPES.get(suffix, "application/octet-stream") + + +def _text_part(text: str) -> dict[str, Any]: + return {"type": "text", "text": text} + + +def _clean_text_part(part: dict[str, Any]) -> dict[str, Any]: + if part.get("type") != "text": + return part + return {"type": "text", "text": remove_path_placeholders(str(part.get("text") or ""))} + + +def _is_empty_text_part(part: dict[str, Any]) -> bool: + return part.get("type") == "text" and not str(part.get("text") or "") + + +def _warn_skipped(row: dict[str, Any], kind: str, reason: str) -> None: + logger.warning( + "retrieval: skipped unreachable evidence asset type={} ref={} asset_url={} source_path={} reason={}", + kind, + row.get("chunk_id"), + row.get("asset_url"), + row.get("file_path"), + reason, + ) diff --git a/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py b/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py index 96a38fd6..f36dde34 100644 --- a/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py +++ b/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py @@ -7,11 +7,8 @@ from sqlalchemy.ext.asyncio import AsyncSession -from shared.services.retrieval.hydration.asset_inline import ( - inline_assets_at_placeholders, - strip_path_placeholders, -) from shared.services.retrieval.hydration.connected import hydrate_connected_target_rows +from shared.services.retrieval.hydration.evidence_compose import compose_evidence_parts from shared.services.retrieval.hydration.row_utils import ( extract_page_nums, filter_excluded_rows, @@ -74,16 +71,13 @@ async def assemble_retrieval_results( page_nums = extract_page_nums(row) if page_nums is not None: assembled_row['page_nums'] = page_nums - elif chunk_type == 'table': - assembled_row['content'] = _compose_table_content(row, rows_by_chunk_id) - assembled_row['content_source'] = 'summary' - elif chunk_type == 'text': - assembled_row['content'] = _compose_text_content(row, rows_by_chunk_id) - assembled_row['content_source'] = 'content' else: assembled_row['content'] = base_content assembled_row['content_source'] = 'content' - assembled_row['content'] = strip_path_placeholders(assembled_row['content']) + assembled_row['composed'] = compose_evidence_parts( + assembled_row, + rows_by_chunk_id, + ) assembled.append(assembled_row) return assembled @@ -116,73 +110,6 @@ def _page_summary(row: dict[str, Any]) -> str: return str(metadata.get('summary') or '').strip() -def _compose_text_content( - row: dict[str, Any], - rows_by_chunk_id: dict[str, dict[str, Any]], -) -> str: - base_content = str(row.get('content') or '') - display_by_target = _connected_display_by_target(row, rows_by_chunk_id) - if not display_by_target: - return base_content - metadata = row.get('chunk_metadata') or row.get('metadata') or {} - connections = ( - metadata.get('connect_to') if isinstance(metadata, dict) else None - ) or [] - content, _embedded = inline_assets_at_placeholders( - base_content, - connections=connections if isinstance(connections, list) else [], - display_by_target=display_by_target, - ) - return content - - -def _compose_table_content( - row: dict[str, Any], - rows_by_chunk_id: dict[str, dict[str, Any]], -) -> str: - parts = [_table_summary_content(row)] - parts.extend(_connected_image_parts(row, rows_by_chunk_id)) - return '\n\n'.join(part for part in parts if part) - - -def _connected_display_by_target( - row: dict[str, Any], - rows_by_chunk_id: dict[str, dict[str, Any]], -) -> dict[str, str]: - display: dict[str, str] = {} - for target_id in iter_connected_target_ids(row): - target_row = rows_by_chunk_id.get(target_id) - if not target_row: - continue - target_type = normalize_chunk_type(target_row.get('chunk_type')) - if target_type == 'table': - target_content = _compose_table_content(target_row, rows_by_chunk_id) - elif target_type == 'image': - target_content = _image_display_content(target_row) - else: - continue - if target_content: - display[target_id] = target_content - return display - - -def _connected_image_parts( - row: dict[str, Any], - rows_by_chunk_id: dict[str, dict[str, Any]], -) -> list[str]: - parts: list[str] = [] - for target_id in iter_connected_target_ids(row): - target_row = rows_by_chunk_id.get(target_id) - if not target_row: - continue - if normalize_chunk_type(target_row.get('chunk_type')) != 'image': - continue - content = _image_display_content(target_row) - if content: - parts.append(content) - return parts - - def _image_display_content(row: dict[str, Any]) -> str: display_ref = ( str(row.get('asset_url') or '').strip() @@ -197,44 +124,3 @@ def _image_display_content(row: dict[str, Any]) -> str: if description: lines.extend(line for line in description.split('\n') if line.strip()) return '\n'.join(lines) - - -def _table_summary_content(row: dict[str, Any]) -> str: - metadata = row.get('chunk_metadata') or row.get('metadata') or {} - if not isinstance(metadata, dict): - metadata = {} - - display_ref = _table_display_ref(row) - lines = [f"[Table: {display_ref}]" if display_ref else "[Table]"] - - summary = str(metadata.get('summary') or row.get('summary') or '').strip() - if summary: - lines.extend(line for line in summary.split('\n') if line.strip()) - - keywords = metadata.get('keywords') or row.get('keywords') or [] - if isinstance(keywords, list): - keyword_text = ';'.join( - str(keyword).strip() for keyword in keywords if str(keyword).strip() - ) - else: - keyword_text = str(keywords or '').strip() - if keyword_text: - lines.append(keyword_text) - - caption = str(metadata.get('caption') or row.get('caption') or '').strip() - if caption: - lines.append(caption) - - return '\n'.join(lines) - - -def _table_display_ref(row: dict[str, Any]) -> str: - for key in ('asset_url', 'file_path', 'source_chunk_path'): - value = str(row.get(key) or '').strip() - if value: - return value - - content = str(row.get('content') or '').strip() - if content and not content.lstrip().lower().startswith(' None: ) assert len(assembled) == 1 content = assembled[0]["content"] - assert "[tables/" not in content - assert content.index("见表") < content.index("[Table:") - assert content.index("[Table:") < content.index("结束") - assert "企业入驻信息登记模板" in content - assert "SHOULD NOT LEAK" not in content + assert "[tables/table-1.html]" in content + assert "[Table:" not in content + composed_text = "".join( + str(part.get("text") or "") + for part in assembled[0]["composed"] + if part.get("type") == "text" + ) + assert composed_text.index("见表") < composed_text.index(" list[dict[str, object]]: ) assert [row["chunk_id"] for row in assembled] == ["text-1"] - assert display_marker in assembled[0]["content"] - assert "资产说明" in assembled[0]["content"] + assert placeholder in assembled[0]["content"] + assert display_marker not in assembled[0]["content"] def test_node_unit_span_inlines_section_assets() -> None: diff --git a/packages/shared-python/shared/tests/test_evidence_compose.py b/packages/shared-python/shared/tests/test_evidence_compose.py new file mode 100644 index 00000000..f58014b0 --- /dev/null +++ b/packages/shared-python/shared/tests/test_evidence_compose.py @@ -0,0 +1,314 @@ +from __future__ import annotations + +import base64 + +import pytest + +from shared.services.retrieval.hydration.evidence_compose import ( + compose_evidence_parts, + flatten_parts, +) +from shared.services.retrieval.hydration.result_assembly import assemble_retrieval_results + + +def test_flatten_parts_keeps_html_and_encodes_images() -> None: + text = flatten_parts( + [ + {"type": "text", "text": "before"}, + {"type": "image", "media_type": "image/png", "data": "abc"}, + {"type": "text", "text": "after"}, + ] + ) + assert text == "beforedata:image/png;base64,abcafter" + + +def test_page_parts_are_summary_then_page_image(monkeypatch) -> None: + monkeypatch.setattr( + "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact", + lambda row, artifact, media_type: { + "type": "image", + "media_type": media_type, + "data": base64.b64encode(b"page").decode("ascii"), + }, + ) + parts = compose_evidence_parts( + { + "chunk_type": "page", + "chunk_metadata": { + "summary": "制度摘要", + "page_nums": [4], + "page_assets": [ + { + "page_num": 4, + "artifact_ref": "page_citation_assets/page-4.png", + "content_type": "image/png", + } + ], + }, + }, + {}, + ) + assert parts[0] == {"type": "text", "text": "制度摘要"} + assert parts[1]["type"] == "image" + assert parts[1]["media_type"] == "image/png" + + +def test_page_parts_do_not_inline_connected_charts() -> None: + parts = compose_evidence_parts( + { + "chunk_type": "page", + "content": "RAW", + "chunk_metadata": { + "summary": "只要摘要", + "connect_to": [ + { + "target": "table-1", + "relation": "embeds", + "ref": "[tables/a.html]", + } + ], + }, + }, + { + "table-1": { + "chunk_type": "table", + "content": "
no
", + } + }, + ) + assert parts == [{"type": "text", "text": "只要摘要"}] + + +def test_standalone_table_uses_html() -> None: + parts = compose_evidence_parts( + { + "chunk_type": "table", + "content": "
Q4
", + "file_path": "tables/a.html", + }, + {}, + ) + assert parts[0]["type"] == "text" + assert "
Q4
" in parts[0]["text"] + + +def test_standalone_image_is_bytes(monkeypatch) -> None: + monkeypatch.setattr( + "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact", + lambda row, artifact, media_type: { + "type": "image", + "media_type": "image/jpeg", + "data": "Zm9v", + }, + ) + parts = compose_evidence_parts( + { + "chunk_type": "image", + "content": "chart caption", + "file_path": "images/a.jpg", + "job_id": "job-1", + }, + {}, + ) + assert parts[0] == {"type": "text", "text": "chart caption"} + assert parts[1] == {"type": "image", "media_type": "image/jpeg", "data": "Zm9v"} + + +@pytest.mark.asyncio +async def test_text_result_keeps_placeholders_and_composes_assets(monkeypatch) -> None: + monkeypatch.setattr( + "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact", + lambda row, artifact, media_type: { + "type": "image", + "media_type": "image/png", + "data": "aW1n", + }, + ) + assembled = await assemble_retrieval_results( + rows=[ + { + "chunk_id": "text-1", + "chunk_type": "text", + "content": "见表 [tables/a.html] 再看 [images/a.png] 结束", + "chunk_metadata": { + "connect_to": [ + { + "target": "table-1", + "relation": "embeds", + "ref": "[tables/a.html]", + }, + { + "target": "image-1", + "relation": "embeds", + "ref": "[images/a.png]", + }, + ] + }, + }, + { + "chunk_id": "table-1", + "chunk_type": "table", + "content": "
Q4
", + "file_path": "tables/a.html", + }, + { + "chunk_id": "image-1", + "chunk_type": "image", + "file_path": "images/a.png", + "job_id": "job-1", + }, + ], + exclude_document_ids=[], + exclude_sections=[], + ) + assert assembled[0]["content"] == "见表 [tables/a.html] 再看 [images/a.png] 结束" + types = [part["type"] for part in assembled[0]["composed"]] + assert "image" in types + composed_text = "".join( + part["text"] for part in assembled[0]["composed"] if part["type"] == "text" + ) + assert "
Q4
" in composed_text + assert "[tables/" not in composed_text + assert "[images/" not in composed_text + + +def test_text_image_keeps_newlines_around_image(monkeypatch) -> None: + monkeypatch.setattr( + "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact", + lambda row, artifact, media_type: { + "type": "image", + "media_type": "image/png", + "data": "aW1n", + }, + ) + parts = compose_evidence_parts( + { + "chunk_type": "text", + "content": "前 [images/a.png] 后", + "chunk_metadata": { + "connect_to": [ + { + "target": "image-1", + "relation": "embeds", + "ref": "[images/a.png]", + } + ] + }, + }, + { + "image-1": { + "chunk_type": "image", + "file_path": "images/a.png", + "job_id": "job-1", + } + }, + ) + assert parts[0] == {"type": "text", "text": "前 \n"} + assert parts[1] == {"type": "image", "media_type": "image/png", "data": "aW1n"} + assert parts[2] == {"type": "text", "text": "\n"} + assert parts[3] == {"type": "text", "text": " 后"} + + +def test_unreachable_assets_clear_placeholders(monkeypatch) -> None: + monkeypatch.setattr( + "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact", + lambda row, artifact, media_type: None, + ) + monkeypatch.setattr( + "shared.services.retrieval.hydration.evidence_compose._try_read_table_html", + lambda row: None, + ) + parts = compose_evidence_parts( + { + "chunk_type": "text", + "content": "见表 [tables/a.html] 再看 [images/a.png] 结束", + "chunk_metadata": { + "connect_to": [ + { + "target": "table-1", + "relation": "embeds", + "ref": "[tables/a.html]", + }, + { + "target": "image-1", + "relation": "embeds", + "ref": "[images/a.png]", + }, + ] + }, + }, + { + "table-1": {"chunk_type": "table", "file_path": "tables/a.html"}, + "image-1": { + "chunk_type": "image", + "file_path": "images/a.png", + "job_id": "job-1", + }, + }, + ) + composed_text = "".join(part["text"] for part in parts if part["type"] == "text") + assert composed_text == "见表 再看 结束" + assert all(part["type"] != "image" for part in parts) + + +def test_missing_placeholder_does_not_append_asset(monkeypatch) -> None: + monkeypatch.setattr( + "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact", + lambda row, artifact, media_type: { + "type": "image", + "media_type": "image/png", + "data": "aW1n", + }, + ) + parts = compose_evidence_parts( + { + "chunk_type": "text", + "content": "没有占位符", + "chunk_metadata": { + "connect_to": [ + { + "target": "image-1", + "relation": "embeds", + "ref": "[images/a.png]", + } + ] + }, + }, + { + "image-1": { + "chunk_type": "image", + "file_path": "images/a.png", + "job_id": "job-1", + } + }, + ) + assert parts == [{"type": "text", "text": "没有占位符"}] + + +def test_page_image_requires_matching_page_num(monkeypatch) -> None: + monkeypatch.setattr( + "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact", + lambda row, artifact, media_type: { + "type": "image", + "media_type": media_type, + "data": "cGFnZQ==", + }, + ) + parts = compose_evidence_parts( + { + "chunk_type": "page", + "chunk_metadata": { + "summary": "只要摘要", + "page_nums": [4], + "page_assets": [ + { + "page_num": 9, + "artifact_ref": "page_citation_assets/page-9.png", + "content_type": "image/png", + } + ], + }, + }, + {}, + ) + assert parts == [{"type": "text", "text": "只要摘要"}] From d7002fcc53b95d00b9014337d0e383fb067437df Mon Sep 17 00:00:00 2001 From: chengke <404835780@qq.com> Date: Mon, 21 Sep 2026 09:46:31 +0800 Subject: [PATCH 2/4] refactor(retrieval): streamline evidence rendering and enhance image handling - Removed the deprecated `render_evidence_blocks` function and replaced it with a new internal function `_render_evidence_blocks` for improved clarity and functionality. - Updated the `_try_read_page_image` and `_select_page_asset` functions to return additional warnings for better error handling when page images are unavailable or mismatched. - Refactored the `flatten_parts` function to utilize the new `page_summary` utility for summarizing page content. - Enhanced tests to validate the new image handling logic and ensure proper warnings are displayed when images are missing or do not match expected page numbers. --- deprecated/mapnav/nav/nav_compose.py | 36 ++++++++++++-- .../retrieval/agent_tools/tools/read.py | 4 +- .../services/retrieval/execution/routes.py | 2 + .../retrieval/hydration/evidence_compose.py | 40 ++++++++-------- .../retrieval/hydration/evidence_text.py | 45 ----------------- .../retrieval/hydration/result_assembly.py | 10 +--- .../services/retrieval/hydration/row_utils.py | 7 +++ .../shared/tests/test_evidence_compose.py | 48 ++++++++++++++++++- 8 files changed, 111 insertions(+), 81 deletions(-) delete mode 100644 packages/shared-python/shared/services/retrieval/hydration/evidence_text.py diff --git a/deprecated/mapnav/nav/nav_compose.py b/deprecated/mapnav/nav/nav_compose.py index e23a5f82..80d9f9a9 100644 --- a/deprecated/mapnav/nav/nav_compose.py +++ b/deprecated/mapnav/nav/nav_compose.py @@ -12,8 +12,6 @@ from dataclasses import dataclass from typing import Any, Dict, List, Optional, Sequence, Set, Tuple -from shared.services.retrieval.hydration.evidence_text import render_evidence_blocks - from ._compat import Chunk from ._compat import line_node_id from ._compat import ToolSpace @@ -322,14 +320,44 @@ def _render_group( *, evidence_index: int, ) -> str: - """Render one evidence block via the shared evidence renderer.""" + """Render one evidence block as ``[E#]`` + path + bodies.""" bodies = [_chunk_body(child.chunk) for child in selected] - return render_evidence_blocks( + return _render_evidence_blocks( [(group.parent_title or "", bodies)], start_index=evidence_index, ) +def _render_evidence_blocks( + groups: Sequence[tuple[str, Sequence[str]]], + *, + start_index: int = 1, +) -> str: + parts: list[str] = [] + index = max(1, int(start_index or 1)) + for path, bodies in groups: + texts = [str(t or "").strip() for t in bodies] + texts = [t for t in texts if t] + if not texts: + continue + block: list[str] = [f"[E{index + len(parts)}]"] + header = str(path or "").strip() + if header: + block.append(f"[§ {header}]") + indent = len(texts) >= 2 + for text in texts: + if indent: + block.append( + "\n".join( + (" " + ln if ln.strip() else ln) for ln in text.splitlines() + ) + ) + else: + block.append(text) + parts.append("\n".join(block).strip()) + return "\n\n".join(parts) + + def _scored_flat(groups: Sequence[_ParentGroup]) -> List[Tuple[Chunk, float]]: out: List[Tuple[Chunk, float]] = [] for g in groups: diff --git a/packages/shared-python/shared/services/retrieval/agent_tools/tools/read.py b/packages/shared-python/shared/services/retrieval/agent_tools/tools/read.py index d219d1a8..ea7f946a 100644 --- a/packages/shared-python/shared/services/retrieval/agent_tools/tools/read.py +++ b/packages/shared-python/shared/services/retrieval/agent_tools/tools/read.py @@ -1,8 +1,8 @@ """``corpus.read`` — full body content for already-located sections/chunks. Unlike ``hydration.result_assembly.assemble_retrieval_results`` (which -down-weights ``page`` chunks to their summary — see that module's -``_page_summary``, a deliberate trade-off for the retrieval-answer surface), +down-weights ``page`` chunks to their summary — see ``page_summary``, a +deliberate trade-off for the retrieval-answer surface), ``read`` returns the page chunk's full body content, with ``[SAME-AS p]`` markers resolved to the owner section's text (§2 of ``CORPUS_SCHEMA.md``) rather than stripped or summarized. ``connect_to`` diff --git a/packages/shared-python/shared/services/retrieval/execution/routes.py b/packages/shared-python/shared/services/retrieval/execution/routes.py index dde8a9eb..4f206196 100644 --- a/packages/shared-python/shared/services/retrieval/execution/routes.py +++ b/packages/shared-python/shared/services/retrieval/execution/routes.py @@ -50,6 +50,8 @@ def open_agent_explore_database_context() -> AbstractAsyncContextManager[AsyncSe def _evidence_fields(rows: list[dict]) -> dict: + # TODO: 后面用 TypeSafe JEV 补结果重排/筛选。现在没有这一步。 + # 无论怎么做,发出去的 evidence 和 results 都是已经筛过或重排过的完整列表,不在 Knowhere 外面做。 evidence = collect_evidence(rows) return { "evidence": evidence, diff --git a/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py b/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py index d6638332..e5cc50a4 100644 --- a/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py +++ b/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py @@ -20,6 +20,7 @@ from shared.services.retrieval.hydration.row_utils import ( extract_page_nums, normalize_chunk_type, + page_summary, ) from shared.services.retrieval.hydration.table_grid import ( TableDownloadError, @@ -80,12 +81,14 @@ def flatten_parts(parts: list[dict[str, Any]] | None) -> str: def _compose_page_parts(row: dict[str, Any]) -> list[dict[str, Any]]: parts: list[dict[str, Any]] = [] - summary = _page_summary(row) + summary = page_summary(row) if summary: parts.append(_text_part(summary)) - image = _try_read_page_image(row) + image, warning = _try_read_page_image(row) if image is not None: parts.append(image) + if warning: + parts.append(_text_part(f"Page image unavailable: {warning}")) return parts @@ -250,16 +253,20 @@ def _try_read_image(row: dict[str, Any]) -> dict[str, Any] | None: return _try_read_image_artifact(row, artifact, media_type=_media_type_from_path(artifact)) -def _try_read_page_image(row: dict[str, Any]) -> dict[str, Any] | None: - asset = _select_page_asset(row) +def _try_read_page_image(row: dict[str, Any]) -> tuple[dict[str, Any] | None, str | None]: + asset, warning = _select_page_asset(row) if asset is None: - return None + _warn_skipped(row, "image", warning) + return None, warning artifact = str(asset.get("artifact_ref") or "").strip() media_type = ( str(asset.get("content_type") or "").split(";", 1)[0].strip() or _media_type_from_path(artifact) ) - return _try_read_image_artifact(row, artifact, media_type=media_type) + image = _try_read_image_artifact(row, artifact, media_type=media_type) + if image is None: + return None, "could not read page image" + return image, None def _try_read_image_artifact( @@ -295,34 +302,27 @@ def _try_read_image_artifact( } -def _select_page_asset(row: dict[str, Any]) -> dict[str, Any] | None: +def _select_page_asset(row: dict[str, Any]) -> tuple[dict[str, Any] | None, str | None]: metadata = row.get("chunk_metadata") or row.get("metadata") or {} if not isinstance(metadata, dict): - return None + return None, "missing page image" assets = metadata.get("page_assets") or [] if not isinstance(assets, list): - return None + return None, "missing page image" candidates = [item for item in assets if isinstance(item, dict)] if not candidates: - return None + return None, "missing page image" page_nums = extract_page_nums(row) or [] if not page_nums: - return candidates[0] + return None, "missing page number" for item in candidates: try: page_num = int(item.get("page_num")) except (TypeError, ValueError): continue if page_num in page_nums: - return item - return None - - -def _page_summary(row: dict[str, Any]) -> str: - metadata = row.get("chunk_metadata") or row.get("metadata") or {} - if not isinstance(metadata, dict): - return "" - return str(metadata.get("summary") or "").strip() + return item, None + return None, "page image does not match this page" def _media_type_from_path(path: str) -> str: diff --git a/packages/shared-python/shared/services/retrieval/hydration/evidence_text.py b/packages/shared-python/shared/services/retrieval/hydration/evidence_text.py deleted file mode 100644 index a868302e..00000000 --- a/packages/shared-python/shared/services/retrieval/hydration/evidence_text.py +++ /dev/null @@ -1,45 +0,0 @@ -"""Shared evidence_text rendering: one ``[E#]`` block per group. - -Each block is ``[E#]`` + ``[§ path]`` (full traceable path) + body lines. -Groups are caller-provided; bodies in one group stay in that group. -""" - -from __future__ import annotations - -from typing import Sequence - - -def render_evidence_blocks( - groups: Sequence[tuple[str, Sequence[str]]], - *, - start_index: int = 1, -) -> str: - """Render (path, bodies) groups into evidence_text. - - ``path`` is the full traceable header (e.g. file / section chain). - Bodies are joined with newlines; multiple bodies in one group are indented. - ``start_index`` sets the first ``[E#]`` number (default 1). - """ - parts: list[str] = [] - index = max(1, int(start_index or 1)) - for path, bodies in groups: - texts = [str(t or "").strip() for t in bodies] - texts = [t for t in texts if t] - if not texts: - continue - block: list[str] = [f"[E{index + len(parts)}]"] - header = str(path or "").strip() - if header: - block.append(f"[§ {header}]") - indent = len(texts) >= 2 - for text in texts: - if indent: - block.append( - "\n".join( - (" " + ln if ln.strip() else ln) for ln in text.splitlines() - ) - ) - else: - block.append(text) - parts.append("\n".join(block).strip()) - return "\n\n".join(parts) diff --git a/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py b/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py index f36dde34..1cc3219f 100644 --- a/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py +++ b/packages/shared-python/shared/services/retrieval/hydration/result_assembly.py @@ -14,6 +14,7 @@ filter_excluded_rows, iter_connected_target_ids, normalize_chunk_type, + page_summary, ) @@ -66,7 +67,7 @@ async def assemble_retrieval_results( base_content = str(row.get('content') or '') chunk_type = normalize_chunk_type(row.get('chunk_type')) if chunk_type == 'page': - assembled_row['content'] = _page_summary(row) + assembled_row['content'] = page_summary(row) assembled_row['content_source'] = 'summary' page_nums = extract_page_nums(row) if page_nums is not None: @@ -103,13 +104,6 @@ def _filter_rows_by_allowed_chunk_types( ] -def _page_summary(row: dict[str, Any]) -> str: - metadata = row.get('chunk_metadata') or row.get('metadata') or {} - if not isinstance(metadata, dict): - return '' - return str(metadata.get('summary') or '').strip() - - def _image_display_content(row: dict[str, Any]) -> str: display_ref = ( str(row.get('asset_url') or '').strip() diff --git a/packages/shared-python/shared/services/retrieval/hydration/row_utils.py b/packages/shared-python/shared/services/retrieval/hydration/row_utils.py index 2a83277b..8e676fe7 100644 --- a/packages/shared-python/shared/services/retrieval/hydration/row_utils.py +++ b/packages/shared-python/shared/services/retrieval/hydration/row_utils.py @@ -41,6 +41,13 @@ def extract_page_nums(row: dict[str, Any]) -> list[int] | None: return page_nums if isinstance(page_nums, list) else None +def page_summary(row: dict[str, Any]) -> str: + metadata = row.get('chunk_metadata') or row.get('metadata') or {} + if not isinstance(metadata, dict): + return '' + return str(metadata.get('summary') or '').strip() + + def build_reference_lookup_key( *, document_id: object, diff --git a/packages/shared-python/shared/tests/test_evidence_compose.py b/packages/shared-python/shared/tests/test_evidence_compose.py index f58014b0..c8387c68 100644 --- a/packages/shared-python/shared/tests/test_evidence_compose.py +++ b/packages/shared-python/shared/tests/test_evidence_compose.py @@ -76,7 +76,11 @@ def test_page_parts_do_not_inline_connected_charts() -> None: } }, ) - assert parts == [{"type": "text", "text": "只要摘要"}] + assert parts[0] == {"type": "text", "text": "只要摘要"} + assert parts[1] == { + "type": "text", + "text": "Page image unavailable: missing page image", + } def test_standalone_table_uses_html() -> None: @@ -311,4 +315,44 @@ def test_page_image_requires_matching_page_num(monkeypatch) -> None: }, {}, ) - assert parts == [{"type": "text", "text": "只要摘要"}] + assert parts == [ + {"type": "text", "text": "只要摘要"}, + { + "type": "text", + "text": "Page image unavailable: page image does not match this page", + }, + ] + + +def test_page_image_missing_page_number_does_not_use_first_asset(monkeypatch) -> None: + monkeypatch.setattr( + "shared.services.retrieval.hydration.evidence_compose._try_read_image_artifact", + lambda row, artifact, media_type: { + "type": "image", + "media_type": media_type, + "data": "cGFnZQ==", + }, + ) + parts = compose_evidence_parts( + { + "chunk_type": "page", + "chunk_metadata": { + "summary": "只要摘要", + "page_assets": [ + { + "page_num": 1, + "artifact_ref": "page_citation_assets/page-1.png", + "content_type": "image/png", + } + ], + }, + }, + {}, + ) + assert parts == [ + {"type": "text", "text": "只要摘要"}, + { + "type": "text", + "text": "Page image unavailable: missing page number", + }, + ] From 7f161c734c59f049619d4bc1771e127e86481230 Mon Sep 17 00:00:00 2001 From: chengke <404835780@qq.com> Date: Mon, 21 Sep 2026 10:29:24 +0800 Subject: [PATCH 3/4] fix(retrieval): keep result placeholders and emit MCP results Public results stay raw for debug; composed parts live only on evidence. MCP query now returns the same package. Co-authored-by: Cursor --- apps/api/app/api/v1/routes/retrieval.py | 5 +- apps/api/app/mcp/retrieval_server.py | 14 ++-- .../test_evidence_renderer_contract.py | 65 +++++++++++++++++++ .../contract/test_retrieval_document_scope.py | 3 +- .../services/retrieval/hydration/row_utils.py | 1 - 5 files changed, 80 insertions(+), 8 deletions(-) diff --git a/apps/api/app/api/v1/routes/retrieval.py b/apps/api/app/api/v1/routes/retrieval.py index ecc49e4a..994f53b2 100644 --- a/apps/api/app/api/v1/routes/retrieval.py +++ b/apps/api/app/api/v1/routes/retrieval.py @@ -170,7 +170,10 @@ class RetrievalQueryResponse(BaseModel): ), ) referenced_chunks: list[dict] = Field(default_factory=list) - results: list[dict] = Field(default_factory=list) + results: list[dict] = Field( + default_factory=list, + description="Raw path chunks for debug. Content keeps placeholders; composed parts live on evidence.", + ) stop_reason: str | None = None failure_reason: str | None = None decision_trace: list[dict] | None = Field( diff --git a/apps/api/app/mcp/retrieval_server.py b/apps/api/app/mcp/retrieval_server.py index 301ba92c..da806746 100644 --- a/apps/api/app/mcp/retrieval_server.py +++ b/apps/api/app/mcp/retrieval_server.py @@ -54,11 +54,12 @@ def resolve_mcp_namespace(*, ctx: Context | None) -> str: def to_mcp_query_response(response: dict[str, Any]) -> dict[str, Any]: - """Project the internal retrieval response to the MCP agent contract. + """Project the public retrieval response to the MCP agent contract. - MCP returns composed evidence plus the text projection and citations: - - evidence: composed parts (text and inline images) + MCP returns the same Knowhere package the HTTP query emits: + - evidence: composed parts (text/HTML and inline images) - evidence_text: text projection of those parts + - results: raw path chunks for debug - referenced_chunks: structured chunk references for citation / follow-up - decision_trace: navigation decisions including terminal stop/failure """ @@ -66,6 +67,7 @@ def to_mcp_query_response(response: dict[str, Any]) -> dict[str, Any]: "query": response.get("query"), "evidence": response.get("evidence") or [], "evidence_text": response.get("evidence_text") or "", + "results": response.get("results") or [], "referenced_chunks": response.get("referenced_chunks") or [], "decision_trace": response.get("decision_trace") or [], } @@ -103,8 +105,10 @@ def create_retrieval_mcp_server( @server.tool( name="retrieval.query", description=( - "Search published documents. Returns evidence (composed parts for " - "LLM consumption), evidence_text (text projection of those parts), " + "Search published documents. Compose already happened in Knowhere. " + "Returns evidence (composed parts for LLM consumption), " + "results (raw path chunks for debug), " + "evidence_text (text projection of evidence), " "referenced_chunks (cited chunk metadata for follow-up queries), " "and decision_trace (navigation decisions including stop/failure reasons). " "Include navigation intent directly in your query text — the " diff --git a/apps/api/tests/contract/test_evidence_renderer_contract.py b/apps/api/tests/contract/test_evidence_renderer_contract.py index b023a8df..da2cac7b 100644 --- a/apps/api/tests/contract/test_evidence_renderer_contract.py +++ b/apps/api/tests/contract/test_evidence_renderer_contract.py @@ -1,3 +1,9 @@ +import asyncio + +from app.mcp.retrieval_server import to_mcp_query_response +from shared.services.retrieval.execution.response_projection import ( + project_public_retrieval_response, +) from shared.services.retrieval.execution.routes import _evidence_fields @@ -24,3 +30,62 @@ def test_evidence_fields_flatten_composed_parts_in_result_order() -> None: assert fields["evidence_text"] == ( "firstmid data:image/png;base64,abc
metric
" ) + + +def test_mcp_query_response_keeps_evidence_and_debug_results() -> None: + response = to_mcp_query_response( + { + "query": "q", + "evidence": [{"type": "text", "text": "t"}], + "evidence_text": "t", + "results": [ + { + "content": "[images/a.png]", + } + ], + "referenced_chunks": [{"chunk_id": "c1"}], + "decision_trace": [{"step": 1}], + "answer_text": "should drop", + } + ) + + assert response == { + "query": "q", + "evidence": [{"type": "text", "text": "t"}], + "evidence_text": "t", + "results": [ + { + "content": "[images/a.png]", + } + ], + "referenced_chunks": [{"chunk_id": "c1"}], + "decision_trace": [{"step": 1}], + } + + +def test_public_results_keep_placeholders_without_composed() -> None: + public = asyncio.run( + project_public_retrieval_response( + { + "namespace": "default", + "query": "q", + "router_used": "classic", + "evidence": [{"type": "text", "text": "Q4
"}], + "evidence_text": "Q4
", + "results": [ + { + "chunk_id": "c1", + "chunk_type": "text", + "content": "见表 [tables/a.html]", + "composed": [{"type": "text", "text": "见表 Q4
"}], + "score": 1, + "document_id": "d1", + } + ], + } + ) + ) + + assert public["evidence"] == [{"type": "text", "text": "Q4
"}] + assert public["results"][0]["content"] == "见表 [tables/a.html]" + assert "composed" not in public["results"][0] diff --git a/apps/api/tests/contract/test_retrieval_document_scope.py b/apps/api/tests/contract/test_retrieval_document_scope.py index 3f15c1af..9329414d 100644 --- a/apps/api/tests/contract/test_retrieval_document_scope.py +++ b/apps/api/tests/contract/test_retrieval_document_scope.py @@ -387,7 +387,8 @@ async def test_connected_hydration_and_assembly_reject_outside_rows( ) assert {r["document_id"] for r in assembled} == expected for row in assembled: - assert "asset secret" in row["content"] + # content stays raw for debug; compose keeps the placeholder intact. + assert "[images/" in row["content"] def test_cache_scope_none_empty_and_set_identity(): diff --git a/packages/shared-python/shared/services/retrieval/hydration/row_utils.py b/packages/shared-python/shared/services/retrieval/hydration/row_utils.py index 8e676fe7..b6bd1b14 100644 --- a/packages/shared-python/shared/services/retrieval/hydration/row_utils.py +++ b/packages/shared-python/shared/services/retrieval/hydration/row_utils.py @@ -12,7 +12,6 @@ 'chunk_type', 'content', 'content_source', - 'composed', 'score', 'asset_url', 'source_chunk_path', From ae653026cf664628b8e7a7fcd101a612032d83ce Mon Sep 17 00:00:00 2001 From: chengke <404835780@qq.com> Date: Mon, 21 Sep 2026 10:29:49 +0800 Subject: [PATCH 4/4] fix(retrieval): satisfy pyright on page image selection Co-authored-by: Cursor --- .../services/retrieval/hydration/evidence_compose.py | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py b/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py index e5cc50a4..c5cb9746 100644 --- a/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py +++ b/packages/shared-python/shared/services/retrieval/hydration/evidence_compose.py @@ -256,8 +256,9 @@ def _try_read_image(row: dict[str, Any]) -> dict[str, Any] | None: def _try_read_page_image(row: dict[str, Any]) -> tuple[dict[str, Any] | None, str | None]: asset, warning = _select_page_asset(row) if asset is None: - _warn_skipped(row, "image", warning) - return None, warning + reason = warning or "missing page image" + _warn_skipped(row, "image", reason) + return None, reason artifact = str(asset.get("artifact_ref") or "").strip() media_type = ( str(asset.get("content_type") or "").split(";", 1)[0].strip() @@ -316,8 +317,11 @@ def _select_page_asset(row: dict[str, Any]) -> tuple[dict[str, Any] | None, str if not page_nums: return None, "missing page number" for item in candidates: + raw_page_num = item.get("page_num") + if raw_page_num is None: + continue try: - page_num = int(item.get("page_num")) + page_num = int(raw_page_num) except (TypeError, ValueError): continue if page_num in page_nums: