Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
17 changes: 7 additions & 10 deletions pyproject.toml
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
[project]
name = "xgen-edit2docs"
version = "0.23.0"
version = "0.24.0"
description = "AI-agent-native document engine: generate and chat-edit DOCX, XLSX and PPTX as a Python library, agent tool set, MCP server or hosted service. English-first with first-class Korean support. Sister project of edit2ppt."
readme = "README.md"
requires-python = ">=3.12"
Expand Down Expand Up @@ -31,24 +31,21 @@ dependencies = [
# Core engine dependencies (from ppt-master)
"python-pptx>=0.6.21",
"python-docx>=1.1.0",
"PyMuPDF>=1.23.0",
# PDF engine: pdfium-based xgen-pdf (BSD/Apache). PyMuPDF (AGPL) is gone.
"xgen-pdf @ https://github.com/PlateerLab/xgen-pdf/releases/download/v0.1.1/xgen_pdf-0.1.1-py3-none-any.whl",
"mammoth>=1.6.0",
"openpyxl>=3.1.0",
# Lossless raw OOXML layer (xgen_contextifier.open_raw): surgical xlsx/docx
# edits and PPTX native-content preservation that keep charts,
# sparklines, custom XML, cached formula values and native
# charts/tables byte-identical through an edit.
# ⚠ 내부 패키지는 **URL 직접참조**로 고정한다. 우리 패키지는 어느 인덱스에도
# 없으므로 이름으로만 요구하면 `pip install -e .` 이 해석에 실패한다 —
# 실제로 이 저장소는 개발 설치조차 되지 않는 상태였다.
"xgen-contextifier @ https://github.com/PlateerLab/xgen-contextifier/releases/download/v0.9.0/xgen_contextifier-0.9.0-py3-none-any.whl",
# Lossless raw OOXML layer: vendored in xgen_edit2docs.raw (from xgen-contextifier 0.9.0),
# so no dependency on that package (which would pull in PyMuPDF).
# Native chart export (data-pptx-native markers) generates the chart's
# embedded XLSX workbook with XlsxWriter (upstream 53c6cad5).
"XlsxWriter>=3.0",
"nbconvert>=7.0.0",
"markdownify>=0.11.6",
"ebooklib>=0.18",
"beautifulsoup4>=4.12.0",
# HWP/OLE containers (was an indirect dependency via xgen-contextifier).
"olefile>=0.47",
"requests>=2.31.0",
"Pillow>=9.0.0",
"numpy>=1.20.0",
Expand Down
2 changes: 1 addition & 1 deletion src/xgen_edit2docs/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -19,7 +19,7 @@
import importlib
from typing import Any

__version__ = "0.18.0"
__version__ = "0.24.0"

_LAZY: dict[str, str] = {
# Unified, extension-dispatched verbs (docx / xlsx / pptx)
Expand Down
2 changes: 1 addition & 1 deletion src/xgen_edit2docs/core/docs/conversion.md
Original file line number Diff line number Diff line change
Expand Up @@ -33,7 +33,7 @@ Prefer MinerU or another OCR/layout tool when:
Dependency:

```bash
pip install PyMuPDF
pip install xgen-pdf
```

## `source_to_md/doc_to_md.py`
Expand Down
2 changes: 1 addition & 1 deletion src/xgen_edit2docs/core/docs/troubleshooting.md
Original file line number Diff line number Diff line change
Expand Up @@ -76,5 +76,5 @@ Important optional packages:
- `edge-tts` for `notes_to_audio.py` recorded narration audio
- `Pillow` for image utilities
- `numpy` for watermark removal
- `PyMuPDF` for PDF conversion
- `xgen-pdf` for PDF conversion
- `google-genai` / `openai` for image generation backends
10 changes: 5 additions & 5 deletions src/xgen_edit2docs/core/source_to_md/pdf_to_md.py
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
#!/usr/bin/env python3
"""
PDF to Markdown Converter
Uses PyMuPDF to extract PDF text content and convert to Markdown format.
Uses xgen-pdf (pdfium) to extract PDF text content and convert to Markdown format.
Supports heading levels, bold, italic, and list detection.
"""

Expand All @@ -14,9 +14,9 @@
from collections import Counter

try:
import fitz # PyMuPDF
import xgen_pdf as fitz # pdfium-based engine (xgen-pdf)
except ImportError:
print("[ERROR] PyMuPDF not installed. Run: pip install PyMuPDF", file=sys.stderr)
print("[ERROR] xgen-pdf not installed. Run: pip install xgen-pdf", file=sys.stderr)
sys.exit(1)

FONT_BODY_SIZE = 12
Expand Down Expand Up @@ -356,7 +356,7 @@ def should_keep_image(
"""Filter out small, decorative, or duplicate images.

Args:
block: Image block extracted from PyMuPDF.
block: Image block from ``page.get_text("dict")``.
page_rect: Current page rectangle.
seen_hashes: Optional set used to deduplicate image payloads.

Expand Down Expand Up @@ -618,7 +618,7 @@ def clean_text(text: str) -> str:
def merge_adjacent_formatting(text: str) -> str:
"""Merge adjacent same-style formatted spans split across PDF tokens.

PyMuPDF often emits a phrase as several spans, so per-span wrapping in
The text engine may emit a phrase as several spans, so per-span wrapping in
``format_span_text`` produces ``**X****Y**`` (bold) or ``***X******Y***``
(bold-italic) where one phrase is intended. Collapse the abutting markers
so the run reads as a single phrase: ``**X Y**`` / ``***X Y***``.
Expand Down
2 changes: 1 addition & 1 deletion src/xgen_edit2docs/documents/arrange.py
Original file line number Diff line number Diff line change
Expand Up @@ -42,7 +42,7 @@ def apply_arrange(
per op ``{op, target, to?, name?, status, message}``; ``warnings`` is
a list of ``{code, message}`` (e.g. a rename that leaves formula
references dangling)."""
from xgen_contextifier import open_raw
from xgen_edit2docs.raw import open_raw

results: list[dict] = []
warnings: list[dict] = []
Expand Down
6 changes: 3 additions & 3 deletions src/xgen_edit2docs/documents/chart_edit.py
Original file line number Diff line number Diff line change
Expand Up @@ -73,7 +73,7 @@ def list_charts(content: bytes, fmt: str) -> list[dict]:
the address source for :func:`apply_chart_edits`.
"""
try:
from xgen_contextifier import open_raw
from xgen_edit2docs.raw import open_raw

raw = open_raw(content, extension=fmt)
out: list[dict] = []
Expand Down Expand Up @@ -110,8 +110,8 @@ def apply_chart_edits(
deterministic editors. The package is only re-serialized when at least
one edit applied; untouched parts stay byte-identical.
"""
from xgen_contextifier import open_raw
from xgen_contextifier.raw.opc import RawUnsupportedError
from xgen_edit2docs.raw import open_raw
from xgen_edit2docs.raw.opc import RawUnsupportedError

raw = open_raw(content, extension=fmt)
charts = _charts_of(raw, fmt)
Expand Down
6 changes: 3 additions & 3 deletions src/xgen_edit2docs/documents/docx_engine.py
Original file line number Diff line number Diff line change
Expand Up @@ -318,7 +318,7 @@ def _chart_outline(content: bytes) -> list[dict]:
"""Read-only chart summaries via xgen_contextifier (best-effort: outline
must never fail because a chart part is exotic)."""
try:
from xgen_contextifier import open_raw
from xgen_edit2docs.raw import open_raw

raw = open_raw(content, extension="docx")
return [
Expand Down Expand Up @@ -379,7 +379,7 @@ def apply_docx_edits(content: bytes, edits: Iterable[DocxEdit]) -> tuple[bytes,
Per-edit soft failures (like the PPTX text editor): ``old_text``
guards replaces with a whitespace-normalized comparison.
"""
from xgen_contextifier import open_raw
from xgen_edit2docs.raw import open_raw

try:
raw = open_raw(content, extension="docx")
Expand Down Expand Up @@ -477,7 +477,7 @@ def _fragment_blocks(fragment: bytes) -> list:
grafting this replaces)."""
import zipfile

from xgen_contextifier.raw import qn
from xgen_edit2docs.raw import qn
from lxml import etree

with zipfile.ZipFile(io.BytesIO(fragment)) as zf:
Expand Down
2 changes: 1 addition & 1 deletion src/xgen_edit2docs/documents/docx_pages.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,7 @@
line wrap (``xgen_edit2docs.render.fonts`` — the same fonts resvg
rasterizes with), lays out tables/images/headers/footers, and emits
one self-contained SVG per page. ``render_doc`` feeds these to the
resvg/PyMuPDF raster layer for PNG/PDF — the piece LibreOffice used to
resvg/xgen-pdf raster layer for PNG/PDF — the piece LibreOffice used to
provide.

Fidelity scope (deliberate): body paragraphs (runs with bold/italic/
Expand Down
6 changes: 3 additions & 3 deletions src/xgen_edit2docs/documents/xlsx_engine.py
Original file line number Diff line number Diff line change
Expand Up @@ -187,7 +187,7 @@ def _chart_outline(content: bytes) -> list[dict]:
"""Read-only chart summaries via xgen_contextifier (best-effort: outline
must never fail because a chart part is exotic)."""
try:
from xgen_contextifier import open_raw
from xgen_edit2docs.raw import open_raw

raw = open_raw(content, extension="xlsx")
return [
Expand Down Expand Up @@ -266,7 +266,7 @@ def apply_xlsx_edits(content: bytes, edits: Iterable[XlsxEdit]) -> tuple[bytes,
formula values all survive byte-identical (the old openpyxl
load→save round-trip destroyed every one of those on EVERY edit).
"""
from xgen_contextifier import open_raw
from xgen_edit2docs.raw import open_raw

try:
raw = open_raw(content, extension="xlsx")
Expand Down Expand Up @@ -369,7 +369,7 @@ def _add_raw_sheet(raw, title: str) -> None:
rels entry, and the ``<sheet>`` element in ``xl/workbook.xml``.
Everything else in the package stays byte-identical.
"""
from xgen_contextifier.raw import qn
from xgen_edit2docs.raw import qn

package = raw.package
n = 1
Expand Down
2 changes: 1 addition & 1 deletion src/xgen_edit2docs/documents/xml_edit.py
Original file line number Diff line number Diff line change
Expand Up @@ -51,7 +51,7 @@ class XmlEditResult:


def _open_package(content: bytes):
from xgen_contextifier.raw.opc import OpcPackage
from xgen_edit2docs.raw.opc import OpcPackage

return OpcPackage.open(content)

Expand Down
93 changes: 93 additions & 0 deletions src/xgen_edit2docs/raw/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,93 @@
# xgen_contextifier/raw
"""
Raw document access — the lossless twin of the extraction pipeline.

Contextifier has two ways to look at a document:

* ``DocumentProcessor.extract_text()`` / ``.process()`` — the existing
pipeline that renders an **AI-friendly** view (clean text, normalized
tables/charts) and throws the rest away.
* ``open_raw()`` (this package) — a **lossless, addressable, writable**
view of the same file. Nothing is discarded: every OPC part stays
available, XML is parsed lazily, and edits are *surgical* — when you
save, untouched parts are written back **byte-identical** (the
byte-preservation contract), so charts, pivot tables, sparklines,
custom XML, styles and anything else the higher-level libraries can't
model all survive.

Usage::

from xgen_edit2docs.raw import open_raw

raw = open_raw("report.xlsx") # XlsxRawDocument
raw.sheets["Sales"].set_cell("B3", 142)
raw.charts[0].set_data(categories=["Q1", "Q2"], series=[("Sales", [1, 2])])
raw.save("report-edited.xlsx") # or raw.to_bytes()

raw = open_raw("deck.pptx") # PptxRawDocument
raw = open_raw("paper.docx") # DocxRawDocument

Every format model also exposes ``.package`` (the raw
:class:`~xgen_edit2docs.raw.opc.OpcPackage`) for part-level work, so the
"easy" interface never locks you out of the full container.

Supported today: the OOXML trio (.xlsx / .docx / .pptx). Other handlers
raise :class:`RawUnsupportedError`.
"""

from __future__ import annotations

from xgen_edit2docs.raw.opc import OpcPackage, OpcPart, RawUnsupportedError
from xgen_edit2docs.raw.xmlpart import NS, XmlPart, qn

__all__ = [
"OpcPackage",
"OpcPart",
"RawUnsupportedError",
"XmlPart",
"NS",
"qn",
"open_raw",
]


def open_raw(source, *, extension: str | None = None):
"""Open a document for lossless, writable access.

Args:
source: path (str/Path), bytes, or a binary file object.
extension: override the format sniff (e.g. ``"xlsx"``); by default
the file extension (for paths) or the package content is used.

Returns:
``XlsxRawDocument`` / ``DocxRawDocument`` / ``PptxRawDocument``.

Raises:
RawUnsupportedError: format has no raw model yet.
"""
from pathlib import Path

ext = (extension or "").lower().lstrip(".")
if not ext and isinstance(source, (str, Path)):
ext = Path(source).suffix.lower().lstrip(".")

package = OpcPackage.open(source)
if not ext:
ext = package.sniff_format() or ""

if ext == "xlsx":
from xgen_edit2docs.raw.xlsx import XlsxRawDocument

return XlsxRawDocument(package)
if ext == "docx":
from xgen_edit2docs.raw.docx import DocxRawDocument

return DocxRawDocument(package)
if ext == "pptx":
from xgen_edit2docs.raw.pptx import PptxRawDocument

return PptxRawDocument(package)
raise RawUnsupportedError(
f"No raw model for {ext or 'unknown format'!r} yet "
"(supported: xlsx, docx, pptx)"
)
Loading
Loading