Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
67 changes: 67 additions & 0 deletions backend/alembic/versions/20260920_0730_add_attribution_columns.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,67 @@
"""add attribution columns to asset

M10. Eight columns recording *whose work* an asset is, as opposed to `source`, which
records how the file arrived. The two are conflated constantly and answer different
questions; see docs/m10-attribution.md for the decisions behind the field set.

Two of the columns look wrong at a glance and are not:

- `published_date` is a string, not a DateTime. Publication dates are routinely partial
("1994", "March 2019") and a DateTime cannot hold either without inventing a precision
that then reads as real. ISO 8601 partial dates compare correctly as plain strings, so
the filters that use it need no parsing. It is indexed because the date range filter
orders by it.
- `license` shadows a Python builtin name only in the interactive interpreter's `site`
namespace, never as a model attribute or a SQL identifier. SQLite has no reserved word
here and SQLAlchemy quotes identifiers regardless.

Adding columns and an index are both things SQLite's ALTER supports directly, so batch
mode does not recreate `asset` here and the PRAGMA foreign_keys dance that
`7d4b9c1a6f28` needed does not apply — nothing drops the table, so nothing can trip over
a row referencing it.

The `asset_fts` column that makes these searchable is deliberately *not* here: it has to
drop and rebuild a virtual table, which is the risky half, and it gets its own revision
so a failure there does not strand these columns.

Revision ID: 3f7a21c9d4e5
Revises: 7d4b9c1a6f28
Create Date: 2026-09-20 07:30:00.000000+00:00
"""
from typing import Sequence, Union

from alembic import op
import sqlalchemy as sa
import sqlmodel


revision: str = '3f7a21c9d4e5'
down_revision: Union[str, None] = '7d4b9c1a6f28'
branch_labels: Union[str, Sequence[str], None] = None
depends_on: Union[str, Sequence[str], None] = None


def upgrade() -> None:
with op.batch_alter_table('asset', schema=None) as batch_op:
batch_op.add_column(sa.Column('source_url', sqlmodel.sql.sqltypes.AutoString(), nullable=True))
batch_op.add_column(sa.Column('creator', sqlmodel.sql.sqltypes.AutoString(), nullable=True))
batch_op.add_column(sa.Column('publisher', sqlmodel.sql.sqltypes.AutoString(), nullable=True))
batch_op.add_column(sa.Column('source_title', sqlmodel.sql.sqltypes.AutoString(), nullable=True))
batch_op.add_column(sa.Column('published_date', sqlmodel.sql.sqltypes.AutoString(), nullable=True))
batch_op.add_column(sa.Column('retrieved_at', sa.DateTime(), nullable=True))
batch_op.add_column(sa.Column('license', sqlmodel.sql.sqltypes.AutoString(), nullable=True))
batch_op.add_column(sa.Column('credit_line', sqlmodel.sql.sqltypes.AutoString(), nullable=True))
batch_op.create_index('ix_asset_published_date', ['published_date'], unique=False)


def downgrade() -> None:
with op.batch_alter_table('asset', schema=None) as batch_op:
batch_op.drop_index('ix_asset_published_date')
batch_op.drop_column('credit_line')
batch_op.drop_column('license')
batch_op.drop_column('retrieved_at')
batch_op.drop_column('published_date')
batch_op.drop_column('source_title')
batch_op.drop_column('publisher')
batch_op.drop_column('creator')
batch_op.drop_column('source_url')
Original file line number Diff line number Diff line change
@@ -0,0 +1,130 @@
"""rebuild asset_fts with an attribution column

M10. Finding an asset by its publisher is half the point of recording one, so the
attribution fields have to reach the keyword index.

**SQLite FTS5 has no ALTER TABLE ADD COLUMN.** The only way to add `attribution_text` is
to drop the virtual table and create it again — and because `asset_fts` stores its own
copy of the text rather than using external-content mode (see the reasoning at the top of
app/search/fts.py), dropping it destroys the index for every existing asset. So this
migration repopulates, and that repopulate is not optional decoration: a recreate without
it passes any test that only compares columns, and leaves a populated library with
keyword search silently returning nothing until each asset happens to be edited again.

Split from `3f7a21c9d4e5` precisely because this half can fail and that half cannot. If
this revision dies partway, the columns are still committed and this is re-runnable;
folded together, a failure here would strand them mid-migration.

The repopulated text is composed to match `app/search/fts.py::index_asset`, so the first
ordinary edit of an asset does not silently rewrite its index entry into something
different. Tag order differs from `tags_text_for`'s (group_concat does not promise one)
and that is immaterial — FTS5 tokenises, so order never reaches the index.

`published_date` and `retrieved_at` are deliberately left out of the text: both are
served exactly by the date-range filter on `/api/assets`, and a bare year in a free-text
index mostly collides with titles rather than helping.

Revision ID: 9c2e08b4a1f7
Revises: 3f7a21c9d4e5
Create Date: 2026-09-20 07:31:00.000000+00:00
"""
from typing import Sequence, Union

from alembic import op

revision: str = '9c2e08b4a1f7'
down_revision: Union[str, None] = '3f7a21c9d4e5'
branch_labels: Union[str, Sequence[str], None] = None
depends_on: Union[str, Sequence[str], None] = None


# This migration carries its own literal copy of the DDL rather than importing the
# constants in app/search/fts.py, per the convention that migration set out: a migration
# has to describe the schema as it was when it was written, and one that imports live
# code silently changes meaning when that code changes. test_migrations.py asserts the
# two copies still agree, which is what stops them drifting unnoticed.
ASSET_FTS_DDL = """
CREATE VIRTUAL TABLE asset_fts USING fts5(
asset_id UNINDEXED,
user_id UNINDEXED,
name,
description,
summary,
tags_text,
attribution_text,
tokenize='porter unicode61'
)
"""

REPOPULATE = """
INSERT INTO asset_fts (
asset_id, user_id, name, description, summary, tags_text, attribution_text
)
SELECT
a.id,
a.user_id,
COALESCE(a.name, ''),
COALESCE(a.description, ''),
COALESCE(a.summary, ''),
COALESCE(
(SELECT group_concat(t.name, ' ')
FROM assettag at
JOIN tag t ON t.id = at.tag_id
WHERE at.asset_id = a.id),
''
),
TRIM(
COALESCE(a.creator, '') || ' ' ||
COALESCE(a.publisher, '') || ' ' ||
COALESCE(a.source_title, '') || ' ' ||
COALESCE(a.license, '') || ' ' ||
COALESCE(a.credit_line, '') || ' ' ||
COALESCE(a.source_url, '')
)
FROM asset a
"""

# The pre-M10 shape, for downgrade. Same reasoning: literal, not imported.
ASSET_FTS_DDL_WITHOUT_ATTRIBUTION = """
CREATE VIRTUAL TABLE asset_fts USING fts5(
asset_id UNINDEXED,
user_id UNINDEXED,
name,
description,
summary,
tags_text,
tokenize='porter unicode61'
)
"""

REPOPULATE_WITHOUT_ATTRIBUTION = """
INSERT INTO asset_fts (asset_id, user_id, name, description, summary, tags_text)
SELECT
a.id,
a.user_id,
COALESCE(a.name, ''),
COALESCE(a.description, ''),
COALESCE(a.summary, ''),
COALESCE(
(SELECT group_concat(t.name, ' ')
FROM assettag at
JOIN tag t ON t.id = at.tag_id
WHERE at.asset_id = a.id),
''
)
FROM asset a
"""


def upgrade() -> None:
op.execute("DROP TABLE IF EXISTS asset_fts")
op.execute(ASSET_FTS_DDL)
op.execute(REPOPULATE)


def downgrade() -> None:
# Symmetric, and repopulating for the same reason: a downgrade that left the index
# empty would be a silent data-shaped loss, not a schema change.
op.execute("DROP TABLE IF EXISTS asset_fts")
op.execute(ASSET_FTS_DDL_WITHOUT_ATTRIBUTION)
op.execute(REPOPULATE_WITHOUT_ATTRIBUTION)
192 changes: 192 additions & 0 deletions backend/app/attribution.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,192 @@
"""Whose work an asset is, and how that resolves through a clip to its parent.

`Asset.source` says how a file arrived — uploaded, generated, cut from something else.
This module is about the other question entirely: who made it, who published it, what it
was part of, and what may be done with it. Recording that at ingest costs a form field;
reconstructing it later means opening every file by hand, and for anything gathered from
the open web the answer is frequently gone. The decisions behind the field set, and the
alternatives rejected on the way, are in docs/m10-attribution.md.

Everything here is pure: no session, no I/O. That is what lets the composition and
inheritance rules be tested exhaustively without a database, and it is why the callers
that *do* touch a session (services/assets.py) pass the parent in rather than having this
module go and find one.
"""

from __future__ import annotations

from dataclasses import dataclass
from datetime import datetime
from typing import Any, Optional, Protocol

# Every attribution field, in one place, so the harvester, the suggestion path, the
# schemas and the index builder cannot drift apart. Adding a ninth field means adding it
# here and following the type errors.
ATTRIBUTION_FIELDS: tuple[str, ...] = (
"source_url",
"creator",
"publisher",
"source_title",
"published_date",
"retrieved_at",
"license",
"credit_line",
)

# The subset that reaches the keyword index — the who and the where.
#
# `published_date` and `retrieved_at` are left out on purpose: both are answered exactly
# by the date-range filters on /api/assets, and a bare year in a free-text index mostly
# collides with titles instead of helping. The migration that builds `asset_fts` carries
# its own literal copy of this list; test_attribution.py asserts they still agree.
ATTRIBUTION_TEXT_FIELDS: tuple[str, ...] = (
"creator",
"publisher",
"source_title",
"license",
"credit_line",
"source_url",
)

# What a composed credit line strings together, in reading order.
_CREDIT_ORDER: tuple[str, ...] = (
"creator",
"source_title",
"publisher",
"published_date",
"license",
)

_CREDIT_SEPARATOR = " — "


class HasAttribution(Protocol):
"""Structural type for the parts of `Asset` this module reads.

A Protocol rather than importing Asset: it keeps this module free of the model layer
(so the tests can drive it with a two-field stub) and documents exactly which columns
the attribution rules depend on.
"""

source_url: Optional[str]
creator: Optional[str]
publisher: Optional[str]
source_title: Optional[str]
published_date: Optional[str]
retrieved_at: Optional[datetime]
license: Optional[str]
credit_line: Optional[str]


@dataclass(frozen=True)
class ResolvedAttribution:
"""One asset's effective attribution, after inheritance."""

source_url: Optional[str] = None
creator: Optional[str] = None
publisher: Optional[str] = None
source_title: Optional[str] = None
published_date: Optional[str] = None
retrieved_at: Optional[datetime] = None
license: Optional[str] = None
credit_line: Optional[str] = None
# Which of the above came from the parent rather than from the asset itself. The API
# hands this to the UI so an inherited value can be shown as inherited, rather than
# looking like something typed on the clip and then quietly diverging from it.
inherited: tuple[str, ...] = ()

@property
def credit(self) -> str:
"""The line to display: the override if there is one, else the composition."""
if not _is_blank(self.credit_line):
return str(self.credit_line).strip()
return compose_credit(self)

@property
def is_empty(self) -> bool:
"""True when nothing is recorded at all — what the `unattributed` filter means."""
return all(_is_blank(getattr(self, name)) for name in ATTRIBUTION_FIELDS)


def _is_blank(value: Any) -> bool:
"""Empty for attribution purposes: None, or a string that is only whitespace.

A datetime is never blank. Written as one helper because "did the user actually put
something here" is asked by inheritance, by composition and by the unattributed
filter, and three subtly different answers would be three subtly different bugs.
"""
if value is None:
return True
if isinstance(value, str):
return not value.strip()
return False


def compose_credit(source: Any) -> str:
"""Build a one-line citation from whatever fields are filled in.

Deliberately mechanical: non-empty parts, in a fixed order, joined by an em dash.
A cleverer format ("Doe, J. *Panorama* (BBC, 2019)") needs rules for every
combination of missing fields and gets them wrong for the combination nobody tried.
`credit_line` exists precisely so a user who wants a specific wording can write it,
and this never has to guess on their behalf.

Ignores `credit_line` itself — this is what a credit line is composed *from*. Use
`ResolvedAttribution.credit` to get the override-or-composition.
"""
parts = []
for name in _CREDIT_ORDER:
value = getattr(source, name, None)
if not _is_blank(value):
parts.append(str(value).strip())
return _CREDIT_SEPARATOR.join(parts)


def resolve(
asset: Any, parent: Any = None
) -> ResolvedAttribution:
"""An asset's effective attribution, falling back to its parent field by field.

A clip of a documentary has the documentary's publisher without anyone retyping it,
and correcting the documentary corrects every clip — resolution happens on read, so
there is no copy anywhere to go stale. Override is per field, not all or nothing, so
a clip can carry its own creator for one interviewee while keeping the rest.

**One level, and that is complete rather than a simplification.**
`services/assets.py::create_clip` refuses to clip a clip, and `promote_clip` leaves
`parent_asset_id` pointing at the original, so no chain of length two can exist. Do
not add a recursive walk for a depth the write paths cannot produce.
"""
values: dict[str, Any] = {}
inherited: list[str] = []

for name in ATTRIBUTION_FIELDS:
own = getattr(asset, name, None)
if not _is_blank(own):
values[name] = own
continue

from_parent = getattr(parent, name, None) if parent is not None else None
if not _is_blank(from_parent):
values[name] = from_parent
inherited.append(name)
else:
values[name] = None

return ResolvedAttribution(**values, inherited=tuple(inherited))


def index_text(resolved: Any) -> str:
"""The attribution text that goes into `asset_fts`.

Takes a resolved attribution (or a bare asset) so a clip is findable by the publisher
it inherited, not only by the fields typed on the clip itself. Keeping a clip out of
those results would make "everything from the BBC" quietly incomplete in a way the
user has no way to notice.
"""
parts = []
for name in ATTRIBUTION_TEXT_FIELDS:
value = getattr(resolved, name, None)
if not _is_blank(value):
parts.append(str(value).strip())
return " ".join(parts)
Loading
Loading