Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
23 changes: 14 additions & 9 deletions backend/app/enrichment/describe.py
Original file line number Diff line number Diff line change
Expand Up @@ -33,22 +33,27 @@

Progress = Callable[..., None]

# A description is a retrieval aid, not an essay — but see summarize.py for why this is
# also told to the model rather than left as a silent backstop: a limit it does not know
# about is one it writes past, and the truncation lands mid-sentence.
MAX_DESCRIPTION_CHARS = 6000

SYSTEM_PROMPT = (
"You write descriptions for a media library. Someone will search this library "
"later with half-remembered words, and your description is what has to match.\n\n"
"Write one paragraph saying what is actually in this item: who appears or speaks, "
"what is shown, the places, organisations, events and specific terms a person would "
"type. Name things explicitly rather than referring to them generally — 'a man in a "
"lab coat' helps nobody find anything; a name does.\n\n"
"Write what is actually in this item: who appears or speaks, what is shown, the "
"places, organisations, events and specific terms a person would type. Name things "
"explicitly rather than referring to them generally — 'a man in a lab coat' helps "
"nobody find anything; a name does.\n\n"
f"Keep the whole reply under {MAX_DESCRIPTION_CHARS} characters — a hard limit, not "
"a target, so finish the sentence you are on rather than trailing off as you "
"approach it. Most descriptions need nowhere near this much; write one paragraph "
"unless the material genuinely needs more than one.\n\n"
"Describe, do not judge. Do not assess whether anything shown or said is true, and "
"do not summarise the argument — another field does that. Never open with 'This "
"image' or 'The video shows'. No heading, no preamble, nothing after the paragraph."
"image' or 'The video shows'. No heading, no preamble, nothing after the text."
)

# A description is a retrieval aid, not an essay. The model is asked for one paragraph;
# this is the backstop for when it does not listen.
MAX_DESCRIPTION_CHARS = 2000


def _prompt(asset: Asset, material: source.SourceMaterial) -> str:
parts = [f"Filename: {asset.original_name or asset.name}"]
Expand Down
55 changes: 55 additions & 0 deletions backend/app/enrichment/generate_all.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,55 @@
"""The generate-all job: summarise, describe and tag one asset in a single press.

Three separate provider calls, run one after another — not the merged, single-prompt
version `docs/m6-ai-enrichment.md` sketches as a future cost optimisation ("one pass,
not three"). That would mean redesigning all three prompts around one shared
completion; this is the "press each button in turn" version turned into one job, which
is what a single "Generate all" control needs and costs nothing extra to get right.

Order is summarize, describe, autotag — not the order the buttons sit in the UI.
`describe`'s own prompt already knows to skip repeating an existing summary ("A summary
already exists; do not repeat it"), so running summarize first is what lets that check
actually have something to check against. `autotag` reads both fields as context, so it
goes last regardless.
"""

from __future__ import annotations

import logging
from typing import Callable

from sqlmodel import Session

from app.enrichment import autotag, describe, summarize
from app.models.asset import Asset

logger = logging.getLogger(__name__)

Progress = Callable[..., None]

_STEPS = (
("Summarising", summarize.run),
("Describing", describe.run),
("Suggesting tags", autotag.run),
)


def run(session: Session, asset: Asset, progress: Progress) -> str:
"""Run summarize, describe and autotag over one asset, in turn.

Stops at whichever step first raises — the same "one clear error" behaviour a
single-action job already has, rather than swallowing a failure to force the
remaining steps to run against material that step already showed is unusable.
"""
results = []
total = len(_STEPS)
for index, (stage, step) in enumerate(_STEPS):
base = int(index * 100 / total)

def inner(_stage: str, pct: int, detail: str = "", *, base=base) -> None:
progress(stage, base + pct // total, detail)

progress(stage, base, "")
results.append(step(session, asset, inner))

return " · ".join(results)
22 changes: 14 additions & 8 deletions backend/app/enrichment/summarize.py
Original file line number Diff line number Diff line change
Expand Up @@ -26,22 +26,28 @@

Progress = Callable[..., None]

# Enough for several real paragraphs and not enough for an essay. Also told to the
# model in the prompt below, which is what stops the backstop below from being the
# thing that actually decides where a summary ends — a limit a model does not know
# about is a limit it will write past, and the truncation lands mid-sentence.
MAX_SUMMARY_CHARS = 6000

SYSTEM_PROMPT = (
"You write summaries for a media library. The person reading yours is trying to "
"find this file again later, sometimes years afterwards, often remembering only "
"roughly what was in it.\n\n"
"Write one paragraph of plain prose. Lead with what the thing actually is, then "
"what it covers — the specific names, places, claims and terms someone would "
"search for. Prefer the concrete over the general.\n\n"
"Write plain prose. Lead with what the thing actually is, then what it covers — "
"the specific names, places, claims and terms someone would search for. Prefer "
"the concrete over the general.\n\n"
f"Keep the whole reply under {MAX_SUMMARY_CHARS} characters — a hard limit, not a "
"target, so finish the paragraph you are on rather than trailing off mid-sentence "
"as you approach it. Most summaries need nowhere near this much; write one "
"paragraph unless the material genuinely needs more than one.\n\n"
"Never open with a phrase like 'This video' or 'The transcript shows'. Do not "
"editorialise, do not assess whether anything said is true, and do not add a "
"preamble, a heading, or anything after the paragraph."
"preamble, a heading, or anything after the text."
)

# Enough for a real paragraph and not enough for an essay. The model is told to write
# one paragraph; this is the backstop for when it does not listen.
MAX_SUMMARY_CHARS = 2000


def _prompt(asset: Asset, material: source.SourceMaterial) -> str:
parts = [f"Filename: {asset.original_name or asset.name}"]
Expand Down
4 changes: 4 additions & 0 deletions backend/app/jobs/enrichment.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,7 @@
from app.enrichment.describe import run as run_describe
from app.enrichment.extract_text import TextExtractionError
from app.enrichment.extract_text import run as run_extract_text
from app.enrichment.generate_all import run as run_generate_all
from app.enrichment.source import NoSourceMaterial
from app.enrichment.summarize import run as run_summarize
from app.enrichment.transcribe import TranscriptionError
Expand All @@ -45,6 +46,7 @@
KIND_DESCRIBE,
KIND_EMBED,
KIND_EXTRACT_TEXT,
KIND_GENERATE_ALL,
KIND_SUMMARIZE,
KIND_TRANSCRIBE,
LIBRARY_KINDS,
Expand Down Expand Up @@ -230,6 +232,8 @@ def _run_job(job_id: str) -> None:
detail = run_autotag(session, asset, progress)
elif job.kind == KIND_DESCRIBE:
detail = run_describe(session, asset, progress)
elif job.kind == KIND_GENERATE_ALL:
detail = run_generate_all(session, asset, progress)
elif job.kind == KIND_BULK_ENRICH:
bulk = run_bulk(session, job.user_id, job.payload, progress)
detail = f"{bulk.done} done"
Expand Down
5 changes: 5 additions & 0 deletions backend/app/models/job.py
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,10 @@
# the odd one out of the per-asset set: it calls no provider, costs nothing, and its
# output is an input to the other three rather than something a user reads directly.
KIND_EXTRACT_TEXT = "extract_text"
# The "Generate all" button: summarize, describe and autotag run one after another as a
# single job, rather than three activity rows the user has to watch separately. See
# `enrichment/generate_all.py` for why that order and not the button order.
KIND_GENERATE_ALL = "generate_all"
# One action applied across a chosen set of assets. Which action, and which assets,
# live in `EnrichmentJob.payload` — see there for why it is one job and not N.
KIND_BULK_ENRICH = "bulk_enrich"
Expand All @@ -44,6 +48,7 @@
KIND_AUTOTAG,
KIND_EMBED,
KIND_EXTRACT_TEXT,
KIND_GENERATE_ALL,
}
)

Expand Down
36 changes: 36 additions & 0 deletions backend/app/routers/enrichment.py
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,7 @@
KIND_DESCRIBE,
KIND_EMBED,
KIND_EXTRACT_TEXT,
KIND_GENERATE_ALL,
KIND_SUMMARIZE,
)
from app.models.document import DocumentPage
Expand Down Expand Up @@ -182,6 +183,41 @@ def start_describe(
return DataResponse(data=KINDS["enrichment"].to_activity(job))


@router.post(
"/{asset_id}/generate-all", response_model=DataResponse[ActivityJobRead], status_code=202
)
def start_generate_all(
asset_id: str,
user: CurrentUser,
session: Session = Depends(get_session),
) -> DataResponse[ActivityJobRead]:
"""Queue summarize, describe and autotag together — the one-button version of
pressing each in turn."""
asset = _owned_asset(asset_id, user.id, session)
_require_provider(session, user.id)

if not summarisable(asset):
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST,
detail={
"code": "not_summarisable",
"message": "This asset has no content to generate from",
},
)

if enrichment_jobs.active_job(session, asset.id, KIND_GENERATE_ALL) is not None:
raise HTTPException(
status_code=status.HTTP_409_CONFLICT,
detail={
"code": "already_running",
"message": "This asset is already generating metadata",
},
)

job = enrichment_jobs.submit(session, asset, KIND_GENERATE_ALL)
return DataResponse(data=KINDS["enrichment"].to_activity(job))


class DocumentPageRead(BaseModel):
model_config = ConfigDict(from_attributes=True)

Expand Down
42 changes: 42 additions & 0 deletions backend/app/services/assets.py
Original file line number Diff line number Diff line change
Expand Up @@ -89,6 +89,7 @@ async def ingest_upload(

_reindex(session, asset)
_describe(session, storage, asset)
_chain_transcription(session, asset)
return asset


Expand Down Expand Up @@ -141,6 +142,47 @@ def _describe(session: Session, storage: LocalStorage, asset: Asset) -> None:
_reindex(session, asset)


def _chain_transcription(session: Session, asset: Asset) -> None:
"""Transcribe audio and video the moment they land, without being asked.

Mirrors `jobs/enrichment.py::_chain_embedding`, one step earlier in the same
chain: a video that nobody transcribes has nothing for describe, summarise,
autotag or search to read, and asking a user to press Transcribe by hand before
any of that works is a step the pipeline does not need them for.

Guarded on a Deepgram key actually being configured, for the same reason the
embedding chain is guarded on a provider: queueing unconditionally would put a
red "no Deepgram key" row in the activity feed after every single upload, which
trains people to ignore it.

Imports `app.jobs.enrichment` lazily rather than at module load — that module's
import chain leads back through `app.enrichment.describe` to this one, and
importing it at the top of this file would be a cycle.
"""
from app.enrichment.transcribe import can_transcribe
from app.jobs import enrichment as enrichment_jobs
from app.models.job import KIND_TRANSCRIBE
from app.settings_store import load_deepgram_key

if not can_transcribe(asset):
return

try:
if not load_deepgram_key(session, asset.user_id):
logger.info(
"Not transcribing asset %s: no Deepgram key configured for this user",
asset.id,
)
return

if enrichment_jobs.active_job(session, asset.id, KIND_TRANSCRIBE) is not None:
return

enrichment_jobs.submit(session, asset, KIND_TRANSCRIBE)
except Exception: # noqa: BLE001 - an upload that succeeded must stay succeeded
logger.warning("Could not queue transcription for asset %s", asset.id, exc_info=True)


def _reindex(session: Session, asset: Asset) -> None:
"""Keep the keyword index in step with the row.

Expand Down
Loading
Loading