From a04f3329301376a8498d9e92c4c41bc5cbff73f1 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 20 Sep 2026 07:44:34 +0000 Subject: [PATCH 1/4] feat(attribution): the schema, the composition rules, and a searchable index MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit M10, steps 1 and 2 of six. Adds the eight attribution columns, the module that composes and inherits them, and the index rebuild that makes them findable. Nothing writes them yet — that is the harvester, next. `published_date` is a string against this repo's own datetime convention, and the model comment says why so it does not get "fixed": publication dates are routinely partial ("1994", "March 2019") and a datetime cannot hold either without inventing a January 1st that then reads as real. ISO partial dates compare correctly as plain strings, so the date filters need no parsing. `credit_line` is an override, not the composed value. Storing the composition would leave it stale the moment `publisher` is corrected — the same rot that made copy-on-create wrong for a clip's inherited attribution one level down. Two migrations, not one. The second drops and recreates `asset_fts`, because FTS5 has no ALTER TABLE ADD COLUMN — and since that table stores its own copy of the text rather than using external-content mode, the recreate destroys the index for every existing asset. The repopulate is what puts it back, and it is invisible to any test that only compares columns: without it the schema is perfect and keyword search silently returns nothing until each asset happens to be edited again. Splitting it out means a failure in that half cannot strand the columns from the first. Verified by removing the repopulate and watching the new tests go red. Inheritance resolves through `parent_asset_id` on read, one level — which is complete, not a simplification, since `create_clip` refuses to clip a clip and `promote_clip` keeps the pointer on the original. The keyword index is the one place that cannot resolve on read, because it stores a snapshot, so `_apply` re-indexes an asset's children when an attribution field changes. Without that, correcting a publisher would leave every clip of it indexed under the old one. Also folds the two existing metadata writers into a shared `_apply` that takes the provenance stamp as an argument, and adds a third value, "embedded", beside "human" and "ai". A tag read out of a file is a fact about the file but not a claim anyone checked — often the camera owner or a studio default — and keeping it distinct is what will let a later pass propose over it while never proposing over something the user typed. 858 backend tests pass. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_019RWM7S6z1UPsZqso4p2HAj --- .../20260920_0730_add_attribution_columns.py | 67 +++++ ...0731_rebuild_asset_fts_with_attribution.py | 130 ++++++++++ backend/app/attribution.py | 192 +++++++++++++++ backend/app/models/asset.py | 44 +++- backend/app/search/fts.py | 34 ++- backend/app/services/assets.py | 101 ++++++-- backend/tests/test_attribution.py | 228 ++++++++++++++++++ backend/tests/test_migrations.py | 106 ++++++++ 8 files changed, 872 insertions(+), 30 deletions(-) create mode 100644 backend/alembic/versions/20260920_0730_add_attribution_columns.py create mode 100644 backend/alembic/versions/20260920_0731_rebuild_asset_fts_with_attribution.py create mode 100644 backend/app/attribution.py create mode 100644 backend/tests/test_attribution.py diff --git a/backend/alembic/versions/20260920_0730_add_attribution_columns.py b/backend/alembic/versions/20260920_0730_add_attribution_columns.py new file mode 100644 index 0000000..e23b150 --- /dev/null +++ b/backend/alembic/versions/20260920_0730_add_attribution_columns.py @@ -0,0 +1,67 @@ +"""add attribution columns to asset + +M10. Eight columns recording *whose work* an asset is, as opposed to `source`, which +records how the file arrived. The two are conflated constantly and answer different +questions; see docs/m10-attribution.md for the decisions behind the field set. + +Two of the columns look wrong at a glance and are not: + +- `published_date` is a string, not a DateTime. Publication dates are routinely partial + ("1994", "March 2019") and a DateTime cannot hold either without inventing a precision + that then reads as real. ISO 8601 partial dates compare correctly as plain strings, so + the filters that use it need no parsing. It is indexed because the date range filter + orders by it. +- `license` shadows a Python builtin name only in the interactive interpreter's `site` + namespace, never as a model attribute or a SQL identifier. SQLite has no reserved word + here and SQLAlchemy quotes identifiers regardless. + +Adding columns and an index are both things SQLite's ALTER supports directly, so batch +mode does not recreate `asset` here and the PRAGMA foreign_keys dance that +`7d4b9c1a6f28` needed does not apply — nothing drops the table, so nothing can trip over +a row referencing it. + +The `asset_fts` column that makes these searchable is deliberately *not* here: it has to +drop and rebuild a virtual table, which is the risky half, and it gets its own revision +so a failure there does not strand these columns. + +Revision ID: 3f7a21c9d4e5 +Revises: 7d4b9c1a6f28 +Create Date: 2026-09-20 07:30:00.000000+00:00 +""" +from typing import Sequence, Union + +from alembic import op +import sqlalchemy as sa +import sqlmodel + + +revision: str = '3f7a21c9d4e5' +down_revision: Union[str, None] = '7d4b9c1a6f28' +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +def upgrade() -> None: + with op.batch_alter_table('asset', schema=None) as batch_op: + batch_op.add_column(sa.Column('source_url', sqlmodel.sql.sqltypes.AutoString(), nullable=True)) + batch_op.add_column(sa.Column('creator', sqlmodel.sql.sqltypes.AutoString(), nullable=True)) + batch_op.add_column(sa.Column('publisher', sqlmodel.sql.sqltypes.AutoString(), nullable=True)) + batch_op.add_column(sa.Column('source_title', sqlmodel.sql.sqltypes.AutoString(), nullable=True)) + batch_op.add_column(sa.Column('published_date', sqlmodel.sql.sqltypes.AutoString(), nullable=True)) + batch_op.add_column(sa.Column('retrieved_at', sa.DateTime(), nullable=True)) + batch_op.add_column(sa.Column('license', sqlmodel.sql.sqltypes.AutoString(), nullable=True)) + batch_op.add_column(sa.Column('credit_line', sqlmodel.sql.sqltypes.AutoString(), nullable=True)) + batch_op.create_index('ix_asset_published_date', ['published_date'], unique=False) + + +def downgrade() -> None: + with op.batch_alter_table('asset', schema=None) as batch_op: + batch_op.drop_index('ix_asset_published_date') + batch_op.drop_column('credit_line') + batch_op.drop_column('license') + batch_op.drop_column('retrieved_at') + batch_op.drop_column('published_date') + batch_op.drop_column('source_title') + batch_op.drop_column('publisher') + batch_op.drop_column('creator') + batch_op.drop_column('source_url') diff --git a/backend/alembic/versions/20260920_0731_rebuild_asset_fts_with_attribution.py b/backend/alembic/versions/20260920_0731_rebuild_asset_fts_with_attribution.py new file mode 100644 index 0000000..8af3461 --- /dev/null +++ b/backend/alembic/versions/20260920_0731_rebuild_asset_fts_with_attribution.py @@ -0,0 +1,130 @@ +"""rebuild asset_fts with an attribution column + +M10. Finding an asset by its publisher is half the point of recording one, so the +attribution fields have to reach the keyword index. + +**SQLite FTS5 has no ALTER TABLE ADD COLUMN.** The only way to add `attribution_text` is +to drop the virtual table and create it again — and because `asset_fts` stores its own +copy of the text rather than using external-content mode (see the reasoning at the top of +app/search/fts.py), dropping it destroys the index for every existing asset. So this +migration repopulates, and that repopulate is not optional decoration: a recreate without +it passes any test that only compares columns, and leaves a populated library with +keyword search silently returning nothing until each asset happens to be edited again. + +Split from `3f7a21c9d4e5` precisely because this half can fail and that half cannot. If +this revision dies partway, the columns are still committed and this is re-runnable; +folded together, a failure here would strand them mid-migration. + +The repopulated text is composed to match `app/search/fts.py::index_asset`, so the first +ordinary edit of an asset does not silently rewrite its index entry into something +different. Tag order differs from `tags_text_for`'s (group_concat does not promise one) +and that is immaterial — FTS5 tokenises, so order never reaches the index. + +`published_date` and `retrieved_at` are deliberately left out of the text: both are +served exactly by the date-range filter on `/api/assets`, and a bare year in a free-text +index mostly collides with titles rather than helping. + +Revision ID: 9c2e08b4a1f7 +Revises: 3f7a21c9d4e5 +Create Date: 2026-09-20 07:31:00.000000+00:00 +""" +from typing import Sequence, Union + +from alembic import op + +revision: str = '9c2e08b4a1f7' +down_revision: Union[str, None] = '3f7a21c9d4e5' +branch_labels: Union[str, Sequence[str], None] = None +depends_on: Union[str, Sequence[str], None] = None + + +# This migration carries its own literal copy of the DDL rather than importing the +# constants in app/search/fts.py, per the convention that migration set out: a migration +# has to describe the schema as it was when it was written, and one that imports live +# code silently changes meaning when that code changes. test_migrations.py asserts the +# two copies still agree, which is what stops them drifting unnoticed. +ASSET_FTS_DDL = """ +CREATE VIRTUAL TABLE asset_fts USING fts5( + asset_id UNINDEXED, + user_id UNINDEXED, + name, + description, + summary, + tags_text, + attribution_text, + tokenize='porter unicode61' +) +""" + +REPOPULATE = """ +INSERT INTO asset_fts ( + asset_id, user_id, name, description, summary, tags_text, attribution_text +) +SELECT + a.id, + a.user_id, + COALESCE(a.name, ''), + COALESCE(a.description, ''), + COALESCE(a.summary, ''), + COALESCE( + (SELECT group_concat(t.name, ' ') + FROM assettag at + JOIN tag t ON t.id = at.tag_id + WHERE at.asset_id = a.id), + '' + ), + TRIM( + COALESCE(a.creator, '') || ' ' || + COALESCE(a.publisher, '') || ' ' || + COALESCE(a.source_title, '') || ' ' || + COALESCE(a.license, '') || ' ' || + COALESCE(a.credit_line, '') || ' ' || + COALESCE(a.source_url, '') + ) +FROM asset a +""" + +# The pre-M10 shape, for downgrade. Same reasoning: literal, not imported. +ASSET_FTS_DDL_WITHOUT_ATTRIBUTION = """ +CREATE VIRTUAL TABLE asset_fts USING fts5( + asset_id UNINDEXED, + user_id UNINDEXED, + name, + description, + summary, + tags_text, + tokenize='porter unicode61' +) +""" + +REPOPULATE_WITHOUT_ATTRIBUTION = """ +INSERT INTO asset_fts (asset_id, user_id, name, description, summary, tags_text) +SELECT + a.id, + a.user_id, + COALESCE(a.name, ''), + COALESCE(a.description, ''), + COALESCE(a.summary, ''), + COALESCE( + (SELECT group_concat(t.name, ' ') + FROM assettag at + JOIN tag t ON t.id = at.tag_id + WHERE at.asset_id = a.id), + '' + ) +FROM asset a +""" + + +def upgrade() -> None: + op.execute("DROP TABLE IF EXISTS asset_fts") + op.execute(ASSET_FTS_DDL) + op.execute(REPOPULATE) + + +def downgrade() -> None: + # Symmetric, and repopulating for the same reason: a downgrade that left the index + # empty would be a silent data-shaped loss, not a schema change. + op.execute("DROP TABLE IF EXISTS asset_fts") + op.execute(ASSET_FTS_DDL_WITHOUT_ATTRIBUTION) + op.execute(REPOPULATE_WITHOUT_ATTRIBUTION) diff --git a/backend/app/attribution.py b/backend/app/attribution.py new file mode 100644 index 0000000..934bc2c --- /dev/null +++ b/backend/app/attribution.py @@ -0,0 +1,192 @@ +"""Whose work an asset is, and how that resolves through a clip to its parent. + +`Asset.source` says how a file arrived — uploaded, generated, cut from something else. +This module is about the other question entirely: who made it, who published it, what it +was part of, and what may be done with it. Recording that at ingest costs a form field; +reconstructing it later means opening every file by hand, and for anything gathered from +the open web the answer is frequently gone. The decisions behind the field set, and the +alternatives rejected on the way, are in docs/m10-attribution.md. + +Everything here is pure: no session, no I/O. That is what lets the composition and +inheritance rules be tested exhaustively without a database, and it is why the callers +that *do* touch a session (services/assets.py) pass the parent in rather than having this +module go and find one. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from datetime import datetime +from typing import Any, Optional, Protocol + +# Every attribution field, in one place, so the harvester, the suggestion path, the +# schemas and the index builder cannot drift apart. Adding a ninth field means adding it +# here and following the type errors. +ATTRIBUTION_FIELDS: tuple[str, ...] = ( + "source_url", + "creator", + "publisher", + "source_title", + "published_date", + "retrieved_at", + "license", + "credit_line", +) + +# The subset that reaches the keyword index — the who and the where. +# +# `published_date` and `retrieved_at` are left out on purpose: both are answered exactly +# by the date-range filters on /api/assets, and a bare year in a free-text index mostly +# collides with titles instead of helping. The migration that builds `asset_fts` carries +# its own literal copy of this list; test_attribution.py asserts they still agree. +ATTRIBUTION_TEXT_FIELDS: tuple[str, ...] = ( + "creator", + "publisher", + "source_title", + "license", + "credit_line", + "source_url", +) + +# What a composed credit line strings together, in reading order. +_CREDIT_ORDER: tuple[str, ...] = ( + "creator", + "source_title", + "publisher", + "published_date", + "license", +) + +_CREDIT_SEPARATOR = " — " + + +class HasAttribution(Protocol): + """Structural type for the parts of `Asset` this module reads. + + A Protocol rather than importing Asset: it keeps this module free of the model layer + (so the tests can drive it with a two-field stub) and documents exactly which columns + the attribution rules depend on. + """ + + source_url: Optional[str] + creator: Optional[str] + publisher: Optional[str] + source_title: Optional[str] + published_date: Optional[str] + retrieved_at: Optional[datetime] + license: Optional[str] + credit_line: Optional[str] + + +@dataclass(frozen=True) +class ResolvedAttribution: + """One asset's effective attribution, after inheritance.""" + + source_url: Optional[str] = None + creator: Optional[str] = None + publisher: Optional[str] = None + source_title: Optional[str] = None + published_date: Optional[str] = None + retrieved_at: Optional[datetime] = None + license: Optional[str] = None + credit_line: Optional[str] = None + # Which of the above came from the parent rather than from the asset itself. The API + # hands this to the UI so an inherited value can be shown as inherited, rather than + # looking like something typed on the clip and then quietly diverging from it. + inherited: tuple[str, ...] = () + + @property + def credit(self) -> str: + """The line to display: the override if there is one, else the composition.""" + if not _is_blank(self.credit_line): + return str(self.credit_line).strip() + return compose_credit(self) + + @property + def is_empty(self) -> bool: + """True when nothing is recorded at all — what the `unattributed` filter means.""" + return all(_is_blank(getattr(self, name)) for name in ATTRIBUTION_FIELDS) + + +def _is_blank(value: Any) -> bool: + """Empty for attribution purposes: None, or a string that is only whitespace. + + A datetime is never blank. Written as one helper because "did the user actually put + something here" is asked by inheritance, by composition and by the unattributed + filter, and three subtly different answers would be three subtly different bugs. + """ + if value is None: + return True + if isinstance(value, str): + return not value.strip() + return False + + +def compose_credit(source: Any) -> str: + """Build a one-line citation from whatever fields are filled in. + + Deliberately mechanical: non-empty parts, in a fixed order, joined by an em dash. + A cleverer format ("Doe, J. *Panorama* (BBC, 2019)") needs rules for every + combination of missing fields and gets them wrong for the combination nobody tried. + `credit_line` exists precisely so a user who wants a specific wording can write it, + and this never has to guess on their behalf. + + Ignores `credit_line` itself — this is what a credit line is composed *from*. Use + `ResolvedAttribution.credit` to get the override-or-composition. + """ + parts = [] + for name in _CREDIT_ORDER: + value = getattr(source, name, None) + if not _is_blank(value): + parts.append(str(value).strip()) + return _CREDIT_SEPARATOR.join(parts) + + +def resolve( + asset: Any, parent: Any = None +) -> ResolvedAttribution: + """An asset's effective attribution, falling back to its parent field by field. + + A clip of a documentary has the documentary's publisher without anyone retyping it, + and correcting the documentary corrects every clip — resolution happens on read, so + there is no copy anywhere to go stale. Override is per field, not all or nothing, so + a clip can carry its own creator for one interviewee while keeping the rest. + + **One level, and that is complete rather than a simplification.** + `services/assets.py::create_clip` refuses to clip a clip, and `promote_clip` leaves + `parent_asset_id` pointing at the original, so no chain of length two can exist. Do + not add a recursive walk for a depth the write paths cannot produce. + """ + values: dict[str, Any] = {} + inherited: list[str] = [] + + for name in ATTRIBUTION_FIELDS: + own = getattr(asset, name, None) + if not _is_blank(own): + values[name] = own + continue + + from_parent = getattr(parent, name, None) if parent is not None else None + if not _is_blank(from_parent): + values[name] = from_parent + inherited.append(name) + else: + values[name] = None + + return ResolvedAttribution(**values, inherited=tuple(inherited)) + + +def index_text(resolved: Any) -> str: + """The attribution text that goes into `asset_fts`. + + Takes a resolved attribution (or a bare asset) so a clip is findable by the publisher + it inherited, not only by the fields typed on the clip itself. Keeping a clip out of + those results would make "everything from the BBC" quietly incomplete in a way the + user has no way to notice. + """ + parts = [] + for name in ATTRIBUTION_TEXT_FIELDS: + value = getattr(resolved, name, None) + if not _is_blank(value): + parts.append(str(value).strip()) + return " ".join(parts) diff --git a/backend/app/models/asset.py b/backend/app/models/asset.py index 751c06b..68d5bd6 100644 --- a/backend/app/models/asset.py +++ b/backend/app/models/asset.py @@ -8,6 +8,20 @@ from app.clock import utcnow +# Who last wrote a field, as recorded in `Asset.field_provenance`. +# +# "embedded" is a third value beside the original two, and the distinction it draws is +# load-bearing: a tag read out of a file (EXIF Artist, an ID3 frame, a PDF Author) is a +# fact about the file but not a claim anybody checked — it is frequently the camera +# owner, a studio default, or boilerplate. Keeping it separate from "human" is what lets +# a later pass propose over a camera-supplied name while never proposing over something +# the user typed. Collapse the two and that distinction is gone for good, because +# nothing else records it. +PROVENANCE_HUMAN = "human" +PROVENANCE_AI = "ai" +PROVENANCE_EMBEDDED = "embedded" + + def new_asset_id() -> str: return str(uuid.uuid4()) @@ -86,8 +100,36 @@ class Asset(SQLModel, table=True): transcript_model: Optional[str] = None transcript_language: Optional[str] = None + # ─── attribution (M10) ─────────────────────────────────────────────────── + # Whose work this is, as opposed to `source` above, which is how the file got here. + # The two get conflated constantly; they answer different questions and neither + # substitutes for the other. Specified in docs/m10-attribution.md. + source_url: Optional[str] = None + creator: Optional[str] = None # author, photographer, speaker, director + publisher: Optional[str] = None # outlet, channel, studio, imprint + source_title: Optional[str] = None # the programme, film, article or book + # A string, not a datetime, against this file's own convention two blocks down — + # and deliberately. Publication dates are routinely partial: a book is from 1994, a + # magazine piece from March 2019. A datetime cannot hold either without inventing a + # January 1st that then reads as a real one, which in a citation record is exactly + # the quiet falsehood this milestone exists to prevent. Stores ISO 8601 `YYYY`, + # `YYYY-MM` or `YYYY-MM-DD`, validated on write in schemas_assets.AssetUpdate. + # ISO partial dates compare correctly as plain strings ("2018-12-31" < "2019" < + # "2019-03-01"), so the date filters need no parsing and no special cases. + published_date: Optional[str] = Field(default=None, index=True) + # A real datetime, because a download happened at an instant — there is no + # partial-precision case here to serve. + retrieved_at: Optional[datetime] = None + license: Optional[str] = None + # The displayed citation, and an *override* only — null until somebody types one. + # The value shown is composed from the fields above on read (app/attribution.py). + # Storing the composition instead would leave it stale the moment `publisher` is + # corrected, which is the same rot that made copy-on-create the wrong answer for a + # clip's inherited attribution one level down. + credit_line: Optional[str] = None + # ─── provenance of the metadata, not the file ─────────────────────────── - # JSON, {"description": "ai"|"human", ...}. FR 8.1.3 requires that a later AI run + # JSON, {"description": "ai"|"human"|"embedded", ...}. FR 8.1.3 requires that a later AI run # never silently overwrites something a person wrote; without recording who last # wrote each field, that rule has nothing to check against. Written from M6, read # never before — but the column exists now so the first enrichment run has diff --git a/backend/app/search/fts.py b/backend/app/search/fts.py index d6f8713..54640d3 100644 --- a/backend/app/search/fts.py +++ b/backend/app/search/fts.py @@ -46,6 +46,11 @@ # UNINDEXED columns ride along so a hit resolves to an asset and a timestamp without a # second query, while staying out of the tokeniser — an asset id must not match a text # search. +# +# `attribution_text` is appended rather than slotted in beside `name`, and that position +# is load-bearing: `search_assets` passes a *column index* to snippet() (2, for `name`), +# so inserting a column anywhere before it would silently start excerpting the wrong +# field. Appending keeps every existing index valid. ASSET_FTS_DDL = """ CREATE VIRTUAL TABLE asset_fts USING fts5( asset_id UNINDEXED, @@ -54,6 +59,7 @@ description, summary, tags_text, + attribution_text, tokenize='porter unicode61' ) """ @@ -69,7 +75,15 @@ ) """ -ASSET_FTS_COLUMNS = ("asset_id", "user_id", "name", "description", "summary", "tags_text") +ASSET_FTS_COLUMNS = ( + "asset_id", + "user_id", + "name", + "description", + "summary", + "tags_text", + "attribution_text", +) SEGMENT_FTS_COLUMNS = ("segment_id", "asset_id", "user_id", "body", "start_time") # The virtual tables, plus the shadow tables SQLite creates behind each one @@ -119,13 +133,22 @@ class FtsHit: # ─── writing ───────────────────────────────────────────────────────────────── -def index_asset(session: Session, asset: Asset, tags_text: str = "") -> None: - """Re-index one asset's own metadata. Safe to call repeatedly.""" +def index_asset( + session: Session, asset: Asset, tags_text: str = "", attribution_text: str = "" +) -> None: + """Re-index one asset's own metadata. Safe to call repeatedly. + + `attribution_text` is passed in rather than composed here for the same reason + `tags_text` is: this module knows about the index, not about what the application + considers worth indexing. `services/assets._reindex` builds both. + """ remove_asset(session, asset.id) session.execute( text( - f"INSERT INTO {ASSET_FTS} (asset_id, user_id, name, description, summary, tags_text)" - " VALUES (:asset_id, :user_id, :name, :description, :summary, :tags_text)" + f"INSERT INTO {ASSET_FTS}" + " (asset_id, user_id, name, description, summary, tags_text, attribution_text)" + " VALUES" + " (:asset_id, :user_id, :name, :description, :summary, :tags_text, :attribution_text)" ), { "asset_id": asset.id, @@ -134,6 +157,7 @@ def index_asset(session: Session, asset: Asset, tags_text: str = "") -> None: "description": asset.description or "", "summary": asset.summary or "", "tags_text": tags_text, + "attribution_text": attribution_text, }, ) session.commit() diff --git a/backend/app/services/assets.py b/backend/app/services/assets.py index 2d3791c..a6bd05d 100644 --- a/backend/app/services/assets.py +++ b/backend/app/services/assets.py @@ -14,6 +14,7 @@ from sqlmodel import Session, col, delete, select, update +from app import attribution from app.auth import sign_media_key from app.clock import utcnow from app.config import settings @@ -31,7 +32,12 @@ sanitize_original_name, ) from app.ingest.probe import ProbeResult, probe -from app.models.asset import Asset +from app.models.asset import ( + PROVENANCE_AI, + PROVENANCE_EMBEDDED, + PROVENANCE_HUMAN, + Asset, +) from app.models.suggestion import Suggestion from app.models.document import DocumentPage from app.models.transcript import TranscriptSegment @@ -201,7 +207,20 @@ def _reindex(session: Session, asset: Asset) -> None: whether the write succeeded. """ try: - fts.index_asset(session, asset, tags_text=tags.tags_text_for(session, asset.id)) + # Attribution is resolved through the parent before indexing, so a clip is + # findable by the publisher it inherited and "everything from the BBC" is not + # quietly missing every clip. The parent lookup only happens for rows that have + # one, and `_reindex_children_of` below keeps those entries current when the + # parent's own attribution later changes. + parent = ( + session.get(Asset, asset.parent_asset_id) if asset.parent_asset_id else None + ) + fts.index_asset( + session, + asset, + tags_text=tags.tags_text_for(session, asset.id), + attribution_text=attribution.index_text(attribution.resolve(asset, parent)), + ) except Exception: # noqa: BLE001 - see above logger.warning("Could not index asset %s for search", asset.id, exc_info=True) @@ -298,25 +317,7 @@ def apply_metadata(session: Session, asset: Asset, changes: dict) -> Asset: `summary` the stamp is now informational only — `apply_ai_metadata` no longer reads it before overwriting either field. """ - if not changes: - return asset - - try: - provenance = json.loads(asset.field_provenance or "{}") - except ValueError: - provenance = {} - - for field, value in changes.items(): - setattr(asset, field, value) - provenance[field] = "human" - - asset.field_provenance = json.dumps(provenance, sort_keys=True) - asset.metadata_modified_date = utcnow() - - session.add(asset) - session.commit() - session.refresh(asset) - _reindex(session, asset) + _apply(session, asset, changes, PROVENANCE_HUMAN) return asset @@ -331,17 +332,45 @@ def apply_ai_metadata(session: Session, asset: Asset, changes: dict) -> list[str Returns the fields written (always every key in `changes`, once any are given). """ + _apply(session, asset, changes, PROVENANCE_AI) + return list(changes.keys()) + + +def apply_embedded_metadata(session: Session, asset: Asset, changes: dict) -> list[str]: + """Write fields read out of the file's own container metadata. + + The third writer, and the reason `_apply` takes the stamp as an argument rather than + hard-coding one. An EXIF `Artist` or an ID3 frame is a fact about the file, so it is + written without asking — but it is not something a person verified, and + `PROVENANCE_EMBEDDED` is what preserves that difference for whatever reads it later. + + Callers are responsible for having filtered to fields that are actually empty: this + overwrites like the other two, because a writer that silently skipped would hide the + same class of bug `apply_ai_metadata`'s docstring describes. + """ + _apply(session, asset, changes, PROVENANCE_EMBEDDED) + return list(changes.keys()) + + +def _apply(session: Session, asset: Asset, changes: dict, provenance_value: str) -> None: + """The shared body of the three writers above: set the fields, stamp who set them. + + A malformed `field_provenance` is rebuilt rather than raising. It is a record *about* + the metadata, and losing that record must never cost the write it describes. + """ if not changes: - return [] + return try: provenance = json.loads(asset.field_provenance or "{}") except ValueError: provenance = {} + if not isinstance(provenance, dict): + provenance = {} for field_name, value in changes.items(): setattr(asset, field_name, value) - provenance[field_name] = "ai" + provenance[field_name] = provenance_value asset.field_provenance = json.dumps(provenance, sort_keys=True) asset.metadata_modified_date = utcnow() @@ -350,7 +379,31 @@ def apply_ai_metadata(session: Session, asset: Asset, changes: dict) -> list[str session.commit() session.refresh(asset) _reindex(session, asset) - return list(changes.keys()) + + # A clip's index entry carries the attribution it inherited, so correcting this + # asset's publisher leaves every clip of it indexed under the old one until they are + # rebuilt. Only on an attribution change, and only for rows that actually have + # children — the common edit (a description, a summary) touches neither. + if any(name in changes for name in attribution.ATTRIBUTION_FIELDS): + _reindex_children_of(session, asset) + + +def _reindex_children_of(session: Session, parent: Asset) -> None: + """Rebuild the index entries of everything derived from this asset. + + Inheritance is resolved at read time everywhere *except* the keyword index, which by + its nature stores a snapshot. This is the one place that snapshot has to be caught + up, and it is why `resolve`'s one-level rule matters: there is no grandchild to + recurse into. + """ + try: + children = list_children(session, parent.id, parent.user_id) + except Exception: # noqa: BLE001 - a stale child index must not fail the parent's write + logger.warning("Could not list children of %s to re-index", parent.id, exc_info=True) + return + + for child in children: + _reindex(session, child) # ─── clips and sub-videos (M7) ──────────────────────────────────────────────── diff --git a/backend/tests/test_attribution.py b/backend/tests/test_attribution.py new file mode 100644 index 0000000..c6e1c81 --- /dev/null +++ b/backend/tests/test_attribution.py @@ -0,0 +1,228 @@ +"""Composition and inheritance rules for attribution (M10). + +`app/attribution.py` is pure on purpose — no session, no I/O — so every rule in it can +be driven directly here with plain objects, rather than through an upload and a clip. +""" + +from datetime import datetime +from pathlib import Path +from types import SimpleNamespace + +from app.attribution import ( + ATTRIBUTION_FIELDS, + ATTRIBUTION_TEXT_FIELDS, + compose_credit, + index_text, + resolve, +) + +BACKEND_ROOT = Path(__file__).resolve().parents[1] + + +def attributed(**fields): + """An object shaped like an Asset for the fields this module reads.""" + values = {name: None for name in ATTRIBUTION_FIELDS} + values.update(fields) + return SimpleNamespace(**values) + + +# ─── composition ───────────────────────────────────────────────────────────── + + +def test_a_full_credit_reads_in_order(): + asset = attributed( + creator="Jane Doe", + source_title="Panorama", + publisher="BBC", + published_date="2019-03", + license="CC BY 4.0", + ) + assert compose_credit(asset) == "Jane Doe — Panorama — BBC — 2019-03 — CC BY 4.0" + + +def test_a_single_field_gets_no_separator(): + """The failure mode a naive join produces: " — BBC" or "BBC — ".""" + assert compose_credit(attributed(publisher="BBC")) == "BBC" + + +def test_missing_fields_are_skipped_not_padded(): + asset = attributed(creator="Jane Doe", publisher="BBC") + assert compose_credit(asset) == "Jane Doe — BBC" + + +def test_nothing_recorded_composes_to_empty_string(): + """Not " — — — — ", and not None: the UI renders this straight into a field.""" + assert compose_credit(attributed()) == "" + + +def test_whitespace_only_fields_count_as_missing(): + asset = attributed(creator=" ", publisher="BBC") + assert compose_credit(asset) == "BBC" + + +def test_values_are_stripped(): + assert compose_credit(attributed(publisher=" BBC ")) == "BBC" + + +def test_credit_line_is_not_part_of_the_composition(): + """It is what the composition is an alternative *to*, not an input to it.""" + asset = attributed(publisher="BBC", credit_line="Something else entirely") + assert compose_credit(asset) == "BBC" + + +# ─── the override ──────────────────────────────────────────────────────────── + + +def test_the_override_wins_over_the_composition(): + resolved = resolve(attributed(publisher="BBC", credit_line="Courtesy of the BBC")) + assert resolved.credit == "Courtesy of the BBC" + + +def test_without_an_override_the_composition_is_used(): + resolved = resolve(attributed(creator="Jane Doe", publisher="BBC")) + assert resolved.credit == "Jane Doe — BBC" + + +def test_a_blank_override_falls_back_rather_than_blanking_the_credit(): + resolved = resolve(attributed(publisher="BBC", credit_line=" ")) + assert resolved.credit == "BBC" + + +# ─── inheritance ───────────────────────────────────────────────────────────── + + +def test_a_clip_with_nothing_of_its_own_inherits_every_field(): + parent = attributed(creator="Jane Doe", publisher="BBC", source_title="Panorama") + resolved = resolve(attributed(), parent) + + assert resolved.creator == "Jane Doe" + assert resolved.publisher == "BBC" + assert set(resolved.inherited) == {"creator", "publisher", "source_title"} + + +def test_an_override_is_per_field_not_all_or_nothing(): + """The whole reason inheritance coalesces field by field.""" + parent = attributed(creator="Jane Doe", publisher="BBC", source_title="Panorama") + clip = attributed(creator="Someone Else") + resolved = resolve(clip, parent) + + assert resolved.creator == "Someone Else" + assert resolved.publisher == "BBC" + assert "creator" not in resolved.inherited + assert "publisher" in resolved.inherited + + +def test_correcting_the_parent_changes_what_the_clip_resolves_to(): + """Resolution happens on read, so there is no copy anywhere to go stale. + + This is the property that made copy-on-create the wrong design: a correction has to + reach the clips already cut from it. + """ + parent = attributed(publisher="BBC Two") + clip = attributed() + assert resolve(clip, parent).publisher == "BBC Two" + + parent.publisher = "BBC Four" + assert resolve(clip, parent).publisher == "BBC Four" + + +def test_a_blank_field_on_the_clip_still_inherits(): + parent = attributed(publisher="BBC") + assert resolve(attributed(publisher=" "), parent).publisher == "BBC" + + +def test_no_parent_resolves_to_the_asset_alone(): + resolved = resolve(attributed(publisher="BBC")) + assert resolved.publisher == "BBC" + assert resolved.inherited == () + + +def test_a_parent_with_nothing_recorded_inherits_nothing(): + resolved = resolve(attributed(), attributed()) + assert resolved.inherited == () + assert resolved.is_empty + + +def test_retrieved_at_inherits_as_a_datetime(): + """The one non-string field: blankness is not emptiness for a datetime.""" + when = datetime(2026, 1, 1, 12, 0, 0) + resolved = resolve(attributed(), attributed(retrieved_at=when)) + assert resolved.retrieved_at == when + assert "retrieved_at" in resolved.inherited + + +# ─── is_empty, which is what the unattributed filter means ─────────────────── + + +def test_is_empty_is_true_only_when_nothing_at_all_is_recorded(): + assert resolve(attributed()).is_empty + assert not resolve(attributed(source_url="https://example.org")).is_empty + assert not resolve(attributed(retrieved_at=datetime(2026, 1, 1))).is_empty + + +def test_an_inheriting_clip_is_not_empty(): + """It has attribution — it just did not type it itself.""" + assert not resolve(attributed(), attributed(publisher="BBC")).is_empty + + +# ─── the search text ───────────────────────────────────────────────────────── + + +def test_index_text_carries_the_who_and_the_where(): + resolved = resolve( + attributed( + creator="Jane Doe", + publisher="BBC", + source_title="Panorama", + license="CC BY 4.0", + source_url="https://example.org/x", + ) + ) + text = index_text(resolved) + for term in ("Jane Doe", "BBC", "Panorama", "CC BY 4.0", "https://example.org/x"): + assert term in text + + +def test_index_text_leaves_dates_out(): + """Both are served exactly by the date-range filters; a bare year in a text index + mostly collides with titles instead of helping.""" + resolved = resolve( + attributed(published_date="1994", retrieved_at=datetime(2026, 1, 1), publisher="BBC") + ) + assert index_text(resolved) == "BBC" + + +def test_index_text_of_nothing_is_empty(): + assert index_text(resolve(attributed())) == "" + + +def test_a_clip_is_indexed_under_what_it_inherited(): + """Otherwise "everything from the BBC" is quietly missing every clip.""" + resolved = resolve(attributed(), attributed(publisher="BBC")) + assert "BBC" in index_text(resolved) + + +# ─── the two copies of the field list ──────────────────────────────────────── + + +def test_the_migration_indexes_the_same_fields_this_module_does(): + """`9c2e08b4a1f7` carries a literal copy of the attribution_text expression. + + That duplication is deliberate — a migration has to describe the schema as it was + when it was written, so it cannot import this module. This is what stops the two + drifting: if a field is added here and not there, an existing library is repopulated + without it and stays unsearchable by that field until every asset is edited again. + """ + migration = ( + BACKEND_ROOT + / "alembic" + / "versions" + / "20260920_0731_rebuild_asset_fts_with_attribution.py" + ).read_text() + expression = migration.split("TRIM(")[1].split("FROM asset")[0] + + for name in ATTRIBUTION_TEXT_FIELDS: + assert f"a.{name}" in expression, f"the migration does not index {name}" + + for name in ("published_date", "retrieved_at"): + assert f"a.{name}" not in expression, f"the migration indexes {name}, this module does not" diff --git a/backend/tests/test_migrations.py b/backend/tests/test_migrations.py index bcba722..27b9c84 100644 --- a/backend/tests/test_migrations.py +++ b/backend/tests/test_migrations.py @@ -213,3 +213,109 @@ def test_add_clip_columns_survives_real_foreign_key_references(alembic_config): ).fetchone() is not None conn.execute("PRAGMA foreign_keys=ON") assert conn.execute("PRAGMA foreign_key_check").fetchall() == [] + + +def test_asset_fts_rebuild_preserves_the_existing_index(alembic_config): + """`9c2e08b4a1f7` drops and recreates asset_fts, which throws away its contents. + + FTS5 has no ALTER TABLE ADD COLUMN, so adding `attribution_text` means recreating + the virtual table — and because asset_fts stores its own copy of the text rather + than using external-content mode, the recreate destroys the index for every asset + already in the library. The repopulate is what puts it back, and it is invisible to + any test that only compares columns: a migration missing it passes the drift check, + leaves the schema perfect, and silently returns nothing for every keyword search + until each asset happens to be edited again. + + So this seeds a real index entry at the revision before, and asserts it is still + searchable after — which is the only assertion that can tell the two apart. + """ + config, db_path = alembic_config + command.upgrade(config, "3f7a21c9d4e5") + + with sqlite3.connect(db_path) as conn: + conn.execute( + "INSERT INTO asset (id, user_id, name, asset_type, source, storage_key," + " size_bytes, field_provenance, upload_date, modified_date, metadata_modified_date," + " description, summary, publisher, creator)" + " VALUES ('a1', 'u', 'Giordano interview', 'video', 'local_upload', 'k1'," + " 100, '{}', '2026-01-01', '2026-01-01', '2026-01-01'," + " 'A long conversation', 'Nano weapons', 'Modern Wisdom', 'James Giordano')" + ) + conn.execute( + "INSERT INTO tag (id, user_id, name, created_at)" + " VALUES ('t1', 'u', 'neuroscience', '2026-01-01')" + ) + conn.execute( + "INSERT INTO assettag (asset_id, tag_id, created_at)" + " VALUES ('a1', 't1', '2026-01-01')" + ) + # The pre-M10 six-column shape, as the old migration created it. + conn.execute( + "INSERT INTO asset_fts (asset_id, user_id, name, description, summary, tags_text)" + " VALUES ('a1', 'u', 'Giordano interview', 'A long conversation'," + " 'Nano weapons', 'neuroscience')" + ) + conn.commit() + + command.upgrade(config, "head") + + with sqlite3.connect(db_path) as conn: + assert conn.execute("SELECT count(*) FROM asset_fts").fetchone()[0] == 1 + + # The pre-existing content still matches, which is the regression that a + # missing repopulate would cause. + for term in ("Giordano", "conversation", "neuroscience"): + hit = conn.execute( + "SELECT asset_id FROM asset_fts WHERE asset_fts MATCH ?", (term,) + ).fetchone() + assert hit is not None and hit[0] == "a1", f"lost the index entry for {term!r}" + + +def test_asset_fts_rebuild_indexes_attribution(alembic_config): + """The point of the rebuild: an asset becomes findable by who published it.""" + config, db_path = alembic_config + command.upgrade(config, "3f7a21c9d4e5") + + with sqlite3.connect(db_path) as conn: + conn.execute( + "INSERT INTO asset (id, user_id, name, asset_type, source, storage_key," + " size_bytes, field_provenance, upload_date, modified_date, metadata_modified_date," + " creator, publisher, source_title, license, source_url)" + " VALUES ('a1', 'u', 'Clip', 'video', 'local_upload', 'k1'," + " 100, '{}', '2026-01-01', '2026-01-01', '2026-01-01'," + " 'Jane Doe', 'BBC', 'Panorama', 'CC BY 4.0', 'https://example.org/x')" + ) + conn.commit() + + command.upgrade(config, "head") + + with sqlite3.connect(db_path) as conn: + for term in ("BBC", "Panorama", "Jane"): + hit = conn.execute( + "SELECT asset_id FROM asset_fts WHERE asset_fts MATCH ?", (term,) + ).fetchone() + assert hit is not None and hit[0] == "a1", f"not findable by {term!r}" + + +def test_asset_fts_downgrade_also_repopulates(alembic_config): + """A downgrade that emptied the index would be a data-shaped loss, not a schema one.""" + config, db_path = alembic_config + command.upgrade(config, "3f7a21c9d4e5") + + with sqlite3.connect(db_path) as conn: + conn.execute( + "INSERT INTO asset (id, user_id, name, asset_type, source, storage_key," + " size_bytes, field_provenance, upload_date, modified_date, metadata_modified_date)" + " VALUES ('a1', 'u', 'Giordano interview', 'video', 'local_upload', 'k1'," + " 100, '{}', '2026-01-01', '2026-01-01', '2026-01-01')" + ) + conn.commit() + + command.upgrade(config, "head") + command.downgrade(config, "3f7a21c9d4e5") + + with sqlite3.connect(db_path) as conn: + hit = conn.execute( + "SELECT asset_id FROM asset_fts WHERE asset_fts MATCH 'Giordano'" + ).fetchone() + assert hit is not None and hit[0] == "a1" From 46a1d7ed2cc6d4be2011250472de29ba1deddbb9 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 20 Sep 2026 07:57:14 +0000 Subject: [PATCH 2/4] feat(attribution): harvest what files already carry, and expose it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit M10, steps 3 and 4 of six. Attribution is now written at ingest, editable by hand, resolved through a clip's parent on read, filterable, and searchable. The harvest reads no new data off the wire. `ingest/probe.py` has always run ffprobe with `-show_format`, which returns the container's tag block, and GAM parsed out the duration and discarded the rest. `ProbeResult` carries it now. EXIF/IPTC, PDF Info and Office core properties join it. Four things the harvester deliberately does not do, each of which was a choice rather than an omission: - It never maps a media container's `title`. There it names the file, not a containing work, and it is very often an encoder's boilerplate; the video fixture sets one precisely so a test can assert it goes nowhere. `album`/`show` do name a work, so those map. Documents go the other way — a document's title is the work's own — and the asymmetry is commented where it would otherwise look arbitrary. - It reads only an allowlist of tag keys. Every MP4 carries `encoder`, `handler_name` and `major_brand`; a mapping that took whatever it recognised would file "Lavf60.16.100" as somebody's creator. - It never harvests `retrieved_at`. When you fetched something is not a fact the file can know. - It only fills blanks. That is what makes it safe to run unasked, and safe for the library-wide re-harvest to run twice. Two bugs caught by testing against real files rather than stubbed tag dictionaries, both of which would have passed against a stub: DateTimeOriginal lives in the Exif sub-IFD rather than IFD0, so reading only the top level works on hand-built fixtures and fails on every actual photograph; and PDF writes its date as "D:20190315101112Z", prefixed and separator-less, which the normaliser did not match. The fixture generator carries the same trap — Pillow serialises the sub-IFD from the value stored under 0x8769, so mutating what get_ifd() returns is silently dropped on save. Filters resolve through the parent the same way the read model does. A filter that only looked at the row would answer "everything from the BBC" with the documentary and none of the clips cut from it, which reads as a bug and is the kind of incompleteness a user cannot notice. `unattributed` follows the same rule: a clip showing its parent's credit is not missing one, and counting it would put rows in the backlog there is nothing to do about. `published_date` accepts only YYYY, YYYY-MM or YYYY-MM-DD. The string column is what allows a partial date to exist; the validator is what stops it becoming free text, where "summer 1994" would sort meaninglessly and break range filters that compare lexicographically precisely because the format is guaranteed. 916 backend tests pass. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_019RWM7S6z1UPsZqso4p2HAj --- backend/app/enrichment/harvest_attribution.py | 98 +++++ backend/app/ingest/embedded_metadata.py | 402 ++++++++++++++++++ backend/app/ingest/probe.py | 33 +- backend/app/jobs/enrichment.py | 10 + backend/app/models/job.py | 9 +- backend/app/routers/assets.py | 77 ++++ backend/app/schemas_assets.py | 59 +++ backend/app/services/assets.py | 121 +++++- backend/tests/fixtures/attributed_audio.mp3 | Bin 0 -> 8657 bytes .../tests/fixtures/attributed_document.docx | Bin 0 -> 36620 bytes .../tests/fixtures/attributed_document.pdf | Bin 0 -> 718 bytes backend/tests/fixtures/attributed_image.jpg | Bin 0 -> 1951 bytes backend/tests/fixtures/attributed_video.mp4 | Bin 0 -> 5646 bytes backend/tests/fixtures/sample_document.docx | Bin 36780 -> 36780 bytes backend/tests/fixtures/sample_document.pptx | Bin 33818 -> 33818 bytes backend/tests/fixtures/sample_document.xlsx | Bin 5355 -> 5356 bytes backend/tests/make_fixtures.py | 111 +++++ backend/tests/test_attribution_api.py | 388 +++++++++++++++++ backend/tests/test_embedded_metadata.py | 219 ++++++++++ 19 files changed, 1522 insertions(+), 5 deletions(-) create mode 100644 backend/app/enrichment/harvest_attribution.py create mode 100644 backend/app/ingest/embedded_metadata.py create mode 100644 backend/tests/fixtures/attributed_audio.mp3 create mode 100644 backend/tests/fixtures/attributed_document.docx create mode 100644 backend/tests/fixtures/attributed_document.pdf create mode 100644 backend/tests/fixtures/attributed_image.jpg create mode 100644 backend/tests/fixtures/attributed_video.mp4 create mode 100644 backend/tests/test_attribution_api.py create mode 100644 backend/tests/test_embedded_metadata.py diff --git a/backend/app/enrichment/harvest_attribution.py b/backend/app/enrichment/harvest_attribution.py new file mode 100644 index 0000000..f0874cd --- /dev/null +++ b/backend/app/enrichment/harvest_attribution.py @@ -0,0 +1,98 @@ +"""Re-reading embedded attribution across a whole library. + +Ingest harvests every new upload, so anything added from M10 onward takes care of +itself. This is for everything added before — which, on an instance that has been +running a while, is the entire library. Those files still carry their EXIF and ID3 and +PDF Author on disk; nothing has ever looked. + +One job walks the lot, for the same reasons `backfill.py` gives: one cancellable row, +one progress bar, one line in the activity feed rather than a wall of near-identical +ones. + +Safe to run repeatedly. It only ever fills fields that are blank, so a second run is a +no-op over everything the first one filled and over everything the user has since +corrected by hand — there is no "already harvested" flag to keep, and no way for this to +walk back over an answer. +""" + +from __future__ import annotations + +import logging +from dataclasses import dataclass +from typing import Callable + +from sqlmodel import Session, col, select + +from app.ingest import embedded_metadata +from app.ingest.probe import probe +from app.models.asset import Asset +from app.services import assets as asset_service +from app.storage import StorageError, build_storage + +logger = logging.getLogger(__name__) + +Progress = Callable[..., None] + + +@dataclass +class HarvestResult: + scanned: int + attributed: int + failed: int + + +def run(session: Session, user_id: str, progress: Progress) -> HarvestResult: + """Harvest every asset of this user's that owns a file.""" + storage = build_storage() + + # Clips are excluded by `storage_key IS NOT NULL`: a clip owns no bytes, and it + # inherits its parent's attribution on read anyway, so there is nothing here for it. + pending = list( + session.exec( + select(Asset) + .where(Asset.user_id == user_id, col(Asset.storage_key).is_not(None)) + .order_by(col(Asset.upload_date).desc()) + ).all() + ) + total = len(pending) + + scanned = attributed = failed = 0 + + for index, asset in enumerate(pending): + pct = int(index * 100 / total) if total else 100 + progress("Reading file metadata", pct, f"{index + 1} of {total} · {asset.name}") + + try: + found = _harvest_one(storage, asset) + except StorageError: + # The file went missing underneath the row. Not this job's problem to fix, + # and not a reason to stop. + failed += 1 + continue + except Exception: # noqa: BLE001 - one unreadable file must not end the run + failed += 1 + logger.warning("Could not harvest attribution for %s", asset.id, exc_info=True) + continue + + scanned += 1 + if not found: + continue + + before = {name: getattr(asset, name, None) for name in found} + asset_service.apply_embedded_attribution(session, asset, found) + session.refresh(asset) + if any(getattr(asset, name, None) != before[name] for name in found): + attributed += 1 + + return HarvestResult(scanned=scanned, attributed=attributed, failed=failed) + + +def _harvest_one(storage, asset: Asset) -> dict: + if not asset.storage_key: + return {} + + with storage.materialise(asset.storage_key) as path: + tags = probe(path).tags if asset.asset_type in ("video", "audio") else {} + return embedded_metadata.harvest( + path, asset.asset_type, asset.original_name or "", probe_tags=tags + ) diff --git a/backend/app/ingest/embedded_metadata.py b/backend/app/ingest/embedded_metadata.py new file mode 100644 index 0000000..7ae4b99 --- /dev/null +++ b/backend/app/ingest/embedded_metadata.py @@ -0,0 +1,402 @@ +"""Attribution the file already carries. + +A JPEG's EXIF `Artist`, an MP3's ID3 frames, an MP4's tag block, a PDF's `Author`, a +.docx's core properties — all written by whoever produced the file. Reading them is not +inference, which is why this writes directly while an AI proposal has to go through the +suggestion queue. The asymmetry is the point; see docs/m10-attribution.md. + +None of it is new data on the wire: `ingest/probe.py` has always run ffprobe with +`-show_format`, which returns the tag block, and GAM parsed out the duration and threw +the rest away. + +Everything here degrades rather than fails, and for the same reason the rest of the +ingest pipeline does: this runs after the row is committed, and a file with unreadable +metadata is still a perfectly good asset. + +What is deliberately *not* harvested: + +- `retrieved_at` — when *you* fetched something is not a fact the file can know. +- `name` — set from the filename at ingest and the user's to change. A container's + `title` is frequently boilerplate from an encoder, and silently renaming somebody's + asset on upload is a worse failure than leaving a field blank. +- `description` / `summary` — not attribution, and owned by the enrichment path. +""" + +from __future__ import annotations + +import logging +import re +from datetime import datetime +from pathlib import Path +from typing import Any, Mapping, Optional + +from app.ingest.filetypes import TYPE_AUDIO, TYPE_DOCUMENT, TYPE_IMAGE, TYPE_VIDEO, extension_of + +logger = logging.getLogger(__name__) + +# EXIF tag numbers, which Pillow returns as an int-keyed mapping. +_EXIF_ARTIST = 0x013B +_EXIF_COPYRIGHT = 0x8298 +_EXIF_DATETIME_ORIGINAL = 0x9003 +_EXIF_DATETIME = 0x0132 + +# IPTC records, as Pillow's IptcImagePlugin keys them: (record, dataset). +_IPTC_BYLINE = (2, 80) +_IPTC_CREDIT = (2, 110) +_IPTC_SOURCE = (2, 115) + +# Container tag names, in preference order, per attribution field. +# +# `title` is absent on purpose. In a media container it names *this file*, not a +# containing work, and it is very often an encoder's boilerplate — so it maps to neither +# `name` (see the module docstring) nor `source_title`. `album` and `show` genuinely do +# name a containing work, so they are here. Documents are the other way round and are +# handled separately below: a document's `title` is the work's own title, and there is no +# album to stand in for it. +_TAG_SOURCES: dict[str, tuple[str, ...]] = { + "creator": ("artist", "author", "album_artist", "composer", "director"), + "publisher": ("publisher", "label", "network", "studio"), + "source_title": ("album", "show"), + "license": ("copyright", "license", "rights"), + "published_date": ("date", "originaldate", "year", "creation_time"), + "source_url": ("purl", "url", "wxxx", "comment"), +} + + +def harvest( + path: Path, + asset_type: str, + original_name: str = "", + probe_tags: Optional[Mapping[str, str]] = None, +) -> dict[str, Any]: + """Attribution fields read out of this file. Never raises. + + Returns only fields it actually found — the caller decides what to do with them, and + `services/assets.py` writes only into fields that are still empty, so a harvest can + never clobber something a person typed. + """ + try: + if asset_type in (TYPE_VIDEO, TYPE_AUDIO): + return _from_container_tags(probe_tags or {}) + if asset_type == TYPE_IMAGE: + return _from_image(path) + if asset_type == TYPE_DOCUMENT: + return _from_document(path, original_name or path.name) + except Exception as exc: # noqa: BLE001 - metadata must never fail an upload + logger.warning("Could not read embedded metadata from %s: %s", path.name, exc) + return {} + + +# ─── video and audio ───────────────────────────────────────────────────────── + + +def _from_container_tags(tags: Mapping[str, str]) -> dict[str, Any]: + """Map an ffprobe tag block onto attribution fields. + + Only the keys in `_TAG_SOURCES` are read. That allowlist is deliberate: a container + tag block is mostly technical noise (`encoder`, `handler_name`, `major_brand`, + `compatible_brands`) and a mapping that took whatever it recognised would file + "Lavf60.16.100" as somebody's creator. + """ + found: dict[str, Any] = {} + + for field_name, candidates in _TAG_SOURCES.items(): + for key in candidates: + value = (tags.get(key) or "").strip() + if not value: + continue + + if field_name == "published_date": + normalised = normalise_partial_date(value) + if normalised: + found[field_name] = normalised + break + continue + + if field_name == "source_url": + # `comment` is in the candidate list because downloaders routinely put + # the source URL there — and also because they put everything else + # there. Only take it when it is actually a URL. + if _looks_like_url(value): + found[field_name] = value + break + continue + + found[field_name] = value + break + + return found + + +# ─── images ────────────────────────────────────────────────────────────────── + + +def _from_image(path: Path) -> dict[str, Any]: + from PIL import Image + + found: dict[str, Any] = {} + with Image.open(path) as image: + exif = image.getexif() + if exif: + artist = _clean(exif.get(_EXIF_ARTIST)) + if artist: + found["creator"] = artist + + rights = _clean(exif.get(_EXIF_COPYRIGHT)) + if rights: + found["license"] = rights + + # DateTimeOriginal is when the shutter fired; DateTime is when the file was + # last written, which an edit updates. Prefer the former. + # + # DateTimeOriginal lives in the Exif sub-IFD (0x8769), not IFD0, so a + # top-level lookup alone finds it in hand-built files and misses it in every + # real photograph — which is the wrong way round for a test to pass. + sub_ifd = {} + try: + sub_ifd = exif.get_ifd(0x8769) or {} + except Exception: # noqa: BLE001 - a malformed sub-IFD is not worth failing over + sub_ifd = {} + + shot = ( + _clean(sub_ifd.get(_EXIF_DATETIME_ORIGINAL)) + or _clean(exif.get(_EXIF_DATETIME_ORIGINAL)) + or _clean(exif.get(_EXIF_DATETIME)) + ) + taken = normalise_partial_date(shot) if shot else None + if taken: + found["published_date"] = taken + + found.update(_from_iptc(image)) + + return found + + +def _from_iptc(image: Any) -> dict[str, Any]: + """IPTC, which is where a press photo's real credit lives. + + EXIF `Artist` is usually the camera owner; a wire photo carries By-line, Credit and + Source instead, and those are the fields a picture desk actually fills in. They win + over EXIF for that reason. + """ + from PIL import IptcImagePlugin + + try: + info = IptcImagePlugin.getiptcinfo(image) + except Exception: # noqa: BLE001 - malformed IPTC is common and not our problem + return {} + if not info: + return {} + + found: dict[str, Any] = {} + byline = _clean(_decode_iptc(info.get(_IPTC_BYLINE))) + if byline: + found["creator"] = byline + + source = _clean(_decode_iptc(info.get(_IPTC_SOURCE))) + if source: + found["publisher"] = source + + credit = _clean(_decode_iptc(info.get(_IPTC_CREDIT))) + if credit: + found["credit_line"] = credit + + return found + + +def _decode_iptc(value: Any) -> Optional[str]: + """IPTC values are bytes, and repeat-able fields come back as a list of them.""" + if isinstance(value, (list, tuple)): + value = value[0] if value else None + if isinstance(value, bytes): + return value.decode("utf-8", errors="replace") + return value if isinstance(value, str) else None + + +# ─── documents ─────────────────────────────────────────────────────────────── + + +def _from_document(path: Path, original_name: str) -> dict[str, Any]: + extension = extension_of(original_name) or extension_of(path.name) + + if extension == ".pdf": + return _from_pdf(path) + if extension == ".docx": + return _from_docx(path) + if extension == ".pptx": + return _from_pptx(path) + if extension == ".xlsx": + return _from_xlsx(path) + # .txt/.md/.csv carry no metadata, and the legacy formats extract_text already + # refuses by name are not readable here either. + return {} + + +def _from_pdf(path: Path) -> dict[str, Any]: + import pypdfium2 + + document = pypdfium2.PdfDocument(path) + try: + return _from_core_properties( + author=document.get_metadata_value("Author"), + title=document.get_metadata_value("Title"), + created=document.get_metadata_value("CreationDate"), + ) + finally: + document.close() + + +def _from_docx(path: Path) -> dict[str, Any]: + import docx + + properties = docx.Document(str(path)).core_properties + return _from_core_properties( + author=properties.author, + title=properties.title, + created=properties.created, + publisher=properties.company if hasattr(properties, "company") else None, + ) + + +def _from_pptx(path: Path) -> dict[str, Any]: + from pptx import Presentation + + properties = Presentation(str(path)).core_properties + return _from_core_properties( + author=properties.author, + title=properties.title, + created=properties.created, + ) + + +def _from_xlsx(path: Path) -> dict[str, Any]: + import openpyxl + + workbook = openpyxl.load_workbook(path, read_only=True, data_only=True) + try: + properties = workbook.properties + return _from_core_properties( + author=properties.creator, + title=properties.title, + created=properties.created, + ) + finally: + workbook.close() + + +def _from_core_properties( + *, + author: Any = None, + title: Any = None, + created: Any = None, + publisher: Any = None, +) -> dict[str, Any]: + """The shape every document format reduces to. + + Unlike a media container, a document's `title` *is* the work's own title, so it maps + to `source_title`. See the note on `_TAG_SOURCES` for why media goes the other way. + """ + found: dict[str, Any] = {} + + cleaned_author = _clean(author) + # Office writes "python-docx"-style generator names into `author` when nobody set + # one, and Word defaults it to the machine's registered owner. Neither is worth + # rejecting heuristically — the user can clear it, and a blank field teaches them + # nothing — but an empty-ish one is not worth writing either. + if cleaned_author: + found["creator"] = cleaned_author + + cleaned_title = _clean(title) + if cleaned_title: + found["source_title"] = cleaned_title + + cleaned_publisher = _clean(publisher) + if cleaned_publisher: + found["publisher"] = cleaned_publisher + + when = normalise_partial_date(created) + if when: + found["published_date"] = when + + return found + + +# ─── shared helpers ────────────────────────────────────────────────────────── + +# Leading year, then optionally month and day, separated by - or : (EXIF uses colons). +_DATE_HEAD = re.compile(r"^\s*(\d{4})(?:[-:/](\d{1,2}))?(?:[-:/](\d{1,2}))?") + +# PDF writes a separator-less date with a marker prefix: "D:20190315101112Z". Matched +# separately because the run-together digits are ambiguous under the pattern above — +# it would read the year and then stop, losing the month and day. +_DATE_COMPACT = re.compile(r"^(?:D:)?(\d{4})(\d{2})(\d{2})(?:\d{2})*[Zz+\-]?") + + +def normalise_partial_date(value: Any) -> Optional[str]: + """Reduce whatever a file claims as a date to `YYYY`, `YYYY-MM` or `YYYY-MM-DD`. + + Containers are wildly inconsistent here: MP4 writes an ISO instant + ("2019-03-15T10:00:00.000000Z"), ID3 often writes a bare year, EXIF uses colons + ("2019:03:15 10:00:00"), and Office hands back a real datetime object. All of them + reduce to the partial-date string `Asset.published_date` stores. + + Precision is never invented: a file that says only "2019" yields "2019", not + "2019-01-01". That is the entire reason the column is a string — see the comment on + the model field. + """ + if value is None: + return None + + if isinstance(value, datetime): + return value.strftime("%Y-%m-%d") + + raw = str(value).strip() + + compact = _DATE_COMPACT.match(raw) + if compact: + year, month, day = compact.groups() + return _assemble_date(year, month, day) + + match = _DATE_HEAD.match(raw.removeprefix("D:")) + if not match: + return None + + year, month, day = match.groups() + return _assemble_date(year, month, day) + + +def _assemble_date( + year: str, month: Optional[str], day: Optional[str] +) -> Optional[str]: + """Build the widest valid partial date these parts support. + + Degrades rather than rejects: a file claiming month 19 still knows its year, and + keeping the year is better than discarding the whole field over one bad component. + """ + if not 1000 <= int(year) <= 9999: + return None + if month is None: + return year + if not 1 <= int(month) <= 12: + return year + if day is None: + return f"{year}-{int(month):02d}" + if not 1 <= int(day) <= 31: + return f"{year}-{int(month):02d}" + return f"{year}-{int(month):02d}-{int(day):02d}" + + +def _looks_like_url(value: str) -> bool: + return value.lower().startswith(("http://", "https://")) + + +def _clean(value: Any) -> Optional[str]: + """Trim, and treat whitespace-only and Pillow's trailing NULs as absent.""" + if value is None: + return None + if isinstance(value, bytes): + value = value.decode("utf-8", errors="replace") + if not isinstance(value, str): + return None + cleaned = value.replace("\x00", "").strip() + return cleaned or None diff --git a/backend/app/ingest/probe.py b/backend/app/ingest/probe.py index cfc1e70..20230c6 100644 --- a/backend/app/ingest/probe.py +++ b/backend/app/ingest/probe.py @@ -11,9 +11,9 @@ import json import logging import subprocess -from dataclasses import dataclass +from dataclasses import dataclass, field from pathlib import Path -from typing import Any, Optional +from typing import Any, Mapping, Optional from app.media_tools import ffmpeg_available @@ -32,6 +32,14 @@ class ProbeResult: codec: Optional[str] = None has_audio: bool = False has_video: bool = False + # The container's own tag block — artist, copyright, date and friends. ffprobe has + # been returning these all along (`-show_format` includes them) and this class threw + # them away; M10's attribution harvester reads them. Keys are lowercased here + # because containers disagree about case and callers should not have to. + # + # `frozen=True` is for immutability, not hashability — a mutable default would be + # shared across instances, and nothing hashes a ProbeResult. + tags: Mapping[str, str] = field(default_factory=dict) def probe(path: Path) -> ProbeResult: @@ -97,9 +105,30 @@ def _interpret(payload: dict[str, Any]) -> ProbeResult: codec=codec, has_audio=audio is not None, has_video=video is not None, + tags=_container_tags(payload, video, audio), ) +def _container_tags( + payload: dict[str, Any], + video: Optional[dict[str, Any]], + audio: Optional[dict[str, Any]], +) -> Mapping[str, str]: + """The container's tag block, lowercased, format first and streams as a fallback. + + Where a tag lives depends on the container: MP4 and MP3 put artist/copyright/date on + the format, while some MKV and transport-stream files carry them only on a stream. + Reading both means the caller does not have to know which it was handed. Format wins, + because it describes the file rather than one track of it. + """ + merged: dict[str, str] = {} + for source in (audio, video, payload.get("format")): + for key, value in ((source or {}).get("tags") or {}).items(): + if isinstance(value, str) and value.strip(): + merged[str(key).strip().lower()] = value.strip() + return merged + + def _display_dimensions(stream: dict[str, Any]) -> tuple[Optional[int], Optional[int]]: """Width and height as the video will actually be shown. diff --git a/backend/app/jobs/enrichment.py b/backend/app/jobs/enrichment.py index ac83cb2..aa577ab 100644 --- a/backend/app/jobs/enrichment.py +++ b/backend/app/jobs/enrichment.py @@ -19,6 +19,7 @@ from app.embeddings import EmbeddingError, build_embedder from app.enrichment.embed import EmbeddingUnavailable from app.enrichment.backfill import run as run_backfill +from app.enrichment.harvest_attribution import run as run_harvest_attribution from app.enrichment.embed import run as run_embed from app.enrichment.autotag import run as run_autotag from app.enrichment.bulk import run as run_bulk @@ -45,6 +46,7 @@ EnrichmentJob, KIND_AUTOTAG, KIND_BACKFILL_EMBEDDINGS, + KIND_HARVEST_ATTRIBUTION, KIND_BULK_ENRICH, KIND_DESCRIBE, KIND_EMBED, @@ -272,6 +274,14 @@ def _run_job(job_id: str) -> None: # Surfaced rather than swallowed: a run that quietly skipped three # assets looks identical to one that embedded everything. detail += f", {result.failed} failed" + elif job.kind == KIND_HARVEST_ATTRIBUTION: + harvested = run_harvest_attribution(session, job.user_id, progress) + detail = ( + f"{harvested.attributed} attributed of {harvested.scanned} scanned" + ) + if harvested.failed: + # Surfaced rather than swallowed, same as the two runs above. + detail += f", {harvested.failed} unreadable" else: raise TranscriptionError(f"Unknown enrichment kind: {job.kind}") diff --git a/backend/app/models/job.py b/backend/app/models/job.py index 7baace3..b3db783 100644 --- a/backend/app/models/job.py +++ b/backend/app/models/job.py @@ -45,6 +45,11 @@ # costs nothing, the same reason `KIND_EXTRACT_TEXT` sits apart from the LLM jobs # despite also being per-asset. KIND_EXTRACT_SUBVIDEO = "extract_subvideo" +# M10. Re-reading embedded attribution (EXIF, ID3, PDF Author) across a whole library, +# for the files that were already there when M10 landed. Library-wide, and outside +# `ENRICHMENT_KINDS` for the same reason as the two above: it calls no provider and +# costs nothing. +KIND_HARVEST_ATTRIBUTION = "harvest_attribution" # Per-asset actions. Everything in here requires an `asset_id`. ENRICHMENT_KINDS = frozenset( @@ -63,7 +68,9 @@ # branches on this set rather than on a hardcoded kind, so adding another one needs no # change to the dispatch. A bulk run over a selection is here too: it has many assets, # which for the purposes of `asset_id` is the same as having none. -LIBRARY_KINDS = frozenset({KIND_BACKFILL_EMBEDDINGS, KIND_BULK_ENRICH}) +LIBRARY_KINDS = frozenset( + {KIND_BACKFILL_EMBEDDINGS, KIND_BULK_ENRICH, KIND_HARVEST_ATTRIBUTION} +) class EnrichmentJob(SQLModel, table=True): diff --git a/backend/app/routers/assets.py b/backend/app/routers/assets.py index c9864c7..06e6e20 100644 --- a/backend/app/routers/assets.py +++ b/backend/app/routers/assets.py @@ -12,10 +12,14 @@ from app.auth import CurrentUser from app.database import get_session from app.ingest.filetypes import ASSET_TYPES +from app.jobs import enrichment as enrichment_jobs +from app.jobs.registry import KINDS from app.models.asset import Asset +from app.models.job import KIND_HARVEST_ATTRIBUTION from app.models.tag import Tag from app.schemas import DataResponse, ListResponse from app.schemas_assets import AssetRead, AssetUpdate, UploadRejection, UploadResult +from app.schemas_jobs import ActivityJobRead from app.schemas_tags import AssetTagsWrite, BulkTagsResult, BulkTagsWrite, TagRead from app.services import assets as service from app.services import tags as tag_service @@ -136,6 +140,12 @@ def list_assets( max_duration: Optional[float] = Query(default=None, ge=0), uploaded_after: Optional[datetime] = Query(default=None), uploaded_before: Optional[datetime] = Query(default=None), + creator: Optional[str] = Query(default=None, max_length=200), + publisher: Optional[str] = Query(default=None, max_length=200), + source_title: Optional[str] = Query(default=None, max_length=200), + published_after: Optional[str] = Query(default=None, max_length=10), + published_before: Optional[str] = Query(default=None, max_length=10), + unattributed: Optional[bool] = Query(default=None), limit: int = Query(default=50, ge=1, le=MAX_PAGE_SIZE), offset: int = Query(default=0, ge=0), session: Session = Depends(get_session), @@ -186,6 +196,41 @@ def list_assets( | col(Asset.original_name).ilike(term) ) + # Attribution (M10). Each resolves through the parent the same way the read model + # does, so a clip is returned by the publisher it inherited — see + # `services/assets.own_or_inherited` for why a row-only filter would be wrong. + for field_name, value in ( + ("creator", creator), + ("publisher", publisher), + ("source_title", source_title), + ): + if value and value.strip(): + needle = f"%{value.strip()}%" + filters.append( + service.own_or_inherited( + field_name, lambda column, n=needle: column.ilike(n) + ) + ) + + # Lexicographic, which is chronological here because AssetUpdate guarantees the + # format. "2019" as a lower bound therefore includes all of 2019, and as an upper + # bound excludes it — the same half-open behaviour a date picker implies. + if published_after: + filters.append( + service.own_or_inherited( + "published_date", lambda column, v=published_after: column >= v + ) + ) + if published_before: + filters.append( + service.own_or_inherited( + "published_date", lambda column, v=published_before: column <= v + ) + ) + + if unattributed is not None: + filters.append(service.unattributed_clause(unattributed)) + # Tag and category filters resolve to a set of ids first. Both are questions about # the join table rather than about the asset row, and an id set keeps them from # turning the main query into a pile of correlated subqueries — one per tag, in the @@ -425,3 +470,35 @@ def bulk_tag_assets( return DataResponse[BulkTagsResult]( data=BulkTagsResult(updated=len(owned), tags_added=[_tag_read(t) for t in added]) ) + + +@router.post( + "/harvest-attribution", + response_model=DataResponse[ActivityJobRead], + status_code=status.HTTP_202_ACCEPTED, +) +def start_attribution_harvest( + user: CurrentUser, session: Session = Depends(get_session) +) -> DataResponse[ActivityJobRead]: + """Re-read embedded attribution for every file already in this library. + + Ingest harvests each new upload, so this exists for everything uploaded before M10 — + on an instance that has been running a while, the whole library. Those files have + been carrying their EXIF and ID3 and PDF Author on disk the entire time; nothing has + ever looked. + + Safe to run repeatedly, because the harvest only ever fills blanks: a second run is a + no-op over what the first one filled and over anything since corrected by hand. That + is also why there is no "already harvested" flag to keep. + """ + if enrichment_jobs.active_library_job(session, user.id, KIND_HARVEST_ATTRIBUTION): + raise HTTPException( + status_code=status.HTTP_409_CONFLICT, + detail={ + "code": "already_running", + "message": "Your library is already being scanned for embedded metadata", + }, + ) + + job = enrichment_jobs.submit_library(session, user.id, KIND_HARVEST_ATTRIBUTION) + return DataResponse(data=KINDS["enrichment"].to_activity(job)) diff --git a/backend/app/schemas_assets.py b/backend/app/schemas_assets.py index 22abd36..999097c 100644 --- a/backend/app/schemas_assets.py +++ b/backend/app/schemas_assets.py @@ -2,11 +2,17 @@ from __future__ import annotations +import re from datetime import datetime from typing import List, Optional from pydantic import BaseModel, Field, field_validator +# A whole year, a year and month, or a full date. Months 01-12 and days 01-31 are +# enforced here so the string column cannot hold "2019-13" and sort between "2019-12" +# and "2020-01" — the range filters rely on lexicographic order being chronological. +_ISO_PARTIAL_DATE = re.compile(r"\d{4}(?:-(?:0[1-9]|1[0-2])(?:-(?:0[1-9]|[12]\d|3[01]))?)?") + class AssetRead(BaseModel): id: str @@ -42,6 +48,29 @@ class AssetRead(BaseModel): # vanished underneath the database shows as missing instead of as a broken image. missing: bool = False + # ─── attribution (M10) ─────────────────────────────────────────────────── + # These are the *resolved* values: a clip with nothing of its own carries what it + # inherited from its parent, so the panel shows a real credit rather than eight + # blanks beside a video that is plainly attributed. + source_url: Optional[str] = None + creator: Optional[str] = None + publisher: Optional[str] = None + source_title: Optional[str] = None + published_date: Optional[str] = None + retrieved_at: Optional[datetime] = None + license: Optional[str] = None + credit_line: Optional[str] = None + + # The line to display: `credit_line` when one was typed, otherwise composed from the + # fields above. Read-only and computed per response, for the same reason `file_url` + # is: a stored copy would be stale the moment a component field changed. + credit: str = "" + # Which of the fields above came from the parent rather than from this row. Sent so + # the UI can mark an inherited value as inherited instead of letting it look like + # something typed on the clip — which is what would make a user "correct" it here + # and quietly break the link to the source. + attribution_inherited: List[str] = Field(default_factory=list) + # Batch-loaded for a listing, never per row: sixty assets a page each asking for # their own tags is sixty queries that grow with the page. tags: List["AssetTagRead"] = Field(default_factory=list) @@ -74,6 +103,36 @@ class AssetUpdate(BaseModel): description: Optional[str] = Field(default=None, max_length=20_000) summary: Optional[str] = Field(default=None, max_length=20_000) + # Attribution (M10). All editable by hand — the harvester only ever fills blanks, + # and an AI may only propose, so this is the sole path that can correct a value. + source_url: Optional[str] = Field(default=None, max_length=2_000) + creator: Optional[str] = Field(default=None, max_length=500) + publisher: Optional[str] = Field(default=None, max_length=500) + source_title: Optional[str] = Field(default=None, max_length=500) + published_date: Optional[str] = Field(default=None, max_length=10) + retrieved_at: Optional[datetime] = None + license: Optional[str] = Field(default=None, max_length=500) + credit_line: Optional[str] = Field(default=None, max_length=2_000) + + @field_validator("published_date") + @classmethod + def _published_date_is_an_iso_partial(cls, value: Optional[str]) -> Optional[str]: + """`YYYY`, `YYYY-MM` or `YYYY-MM-DD`, and nothing else. + + The looseness of a string column is what lets a partial date exist at all; the + validator is what stops it becoming a free-text field where "summer 1994" and + "15/03/19" would sort meaninglessly and break the range filters, which compare + lexicographically precisely because the format is guaranteed here. + """ + if value is None: + return None + cleaned = value.strip() + if not cleaned: + return None + if not _ISO_PARTIAL_DATE.fullmatch(cleaned): + raise ValueError("published_date must be YYYY, YYYY-MM or YYYY-MM-DD") + return cleaned + @field_validator("name") @classmethod def _name_is_not_blank(cls, value: Optional[str]) -> Optional[str]: diff --git a/backend/app/services/assets.py b/backend/app/services/assets.py index a6bd05d..ffebaa8 100644 --- a/backend/app/services/assets.py +++ b/backend/app/services/assets.py @@ -10,15 +10,17 @@ import json import logging import mimetypes -from typing import AsyncIterator, Iterable, Optional +from typing import Any, AsyncIterator, Iterable, Optional +from sqlalchemy import and_, func, not_, or_ +from sqlalchemy.orm import aliased from sqlmodel import Session, col, delete, select, update from app import attribution from app.auth import sign_media_key from app.clock import utcnow from app.config import settings -from app.ingest import thumbnails +from app.ingest import embedded_metadata, thumbnails from app.ingest.filetypes import ( SOURCE_CLIP, SOURCE_SUBVIDEO, @@ -112,6 +114,7 @@ def _describe(session: Session, storage: LocalStorage, asset: Asset) -> None: if not asset.storage_key: return + embedded: dict = {} try: with storage.materialise(asset.storage_key) as path: result = probe(path) @@ -135,6 +138,15 @@ def _describe(session: Session, storage: LocalStorage, asset: Asset) -> None: asset.original_name or "", duration_seconds=result.duration_seconds, ) + + # Read inside the `with`, applied after: the file is only on disk for the + # duration of this block, but writing needs a committed row. + embedded = embedded_metadata.harvest( + path, + asset.asset_type, + asset.original_name or "", + probe_tags=result.tags, + ) except (StorageError, OSError) as exc: logger.warning("Could not describe asset %s: %s", asset.id, exc) return @@ -150,6 +162,37 @@ def _describe(session: Session, storage: LocalStorage, asset: Asset) -> None: session.commit() session.refresh(asset) _reindex(session, asset) + apply_embedded_attribution(session, asset, embedded) + + +def apply_embedded_attribution(session: Session, asset: Asset, embedded: dict) -> None: + """Write harvested attribution into the fields that are still empty. + + Filtering to empty fields is what makes this safe to run without asking, and what + makes the library-wide re-harvest safe to run twice: an EXIF `Artist` is a fact about + the file, but it is frequently the camera's registered owner rather than the + photographer, so it gets to fill a blank and never to overwrite an answer. The + `PROVENANCE_EMBEDDED` stamp is what lets a later pass tell the two apart. + """ + if not embedded: + return + + fillable = { + name: value + for name, value in embedded.items() + if name in attribution.ATTRIBUTION_FIELDS and _is_unset(getattr(asset, name, None)) + } + if not fillable: + return + + try: + apply_embedded_metadata(session, asset, fillable) + except Exception: # noqa: BLE001 - an upload that succeeded must stay succeeded + logger.warning("Could not store embedded attribution for %s", asset.id, exc_info=True) + + +def _is_unset(value) -> bool: + return value is None or (isinstance(value, str) and not value.strip()) def _chain_transcription(session: Session, asset: Asset) -> None: @@ -590,6 +633,64 @@ def parents_for_many(session: Session, assets: Iterable[Asset]) -> dict[str, Ass return {row.id: row for row in rows} +def _blank(column): + """SQL for "nothing recorded here" — NULL, or an empty string. + + `retrieved_at` is a DateTime, where `!= ''` is not a meaningful comparison, so the + empty-string half is only applied to the text columns. + """ + if column.key == "retrieved_at": + return column.is_(None) + return or_(column.is_(None), func.trim(column) == "") + + +def _records_attribution(model) -> Any: + """SQL for "this row has at least one attribution field filled in".""" + return or_(*[not_(_blank(getattr(model, name))) for name in attribution.ATTRIBUTION_FIELDS]) + + +def own_or_inherited(field_name: str, build) -> Any: + """Lift a condition on an asset's own column to "or the parent it inherits from". + + Inheritance is resolved on read everywhere else, so a filter that only looked at the + row would answer "everything from the BBC" with the documentary and none of the + clips cut from it — which reads as a bug and is the sort of quiet incompleteness a + user has no way to notice. + + `build` turns a column into a condition. It is applied to the asset's own column and + to the parent's; the parent only counts where the asset's own value is blank, which + is exactly the coalesce `attribution.resolve` performs on read. + """ + parent = aliased(Asset) + own_column = getattr(Asset, field_name) + return or_( + build(own_column), + and_( + _blank(own_column), + select(1) + .where(parent.id == Asset.parent_asset_id, build(getattr(parent, field_name))) + .exists(), + ), + ) + + +def unattributed_clause(unattributed: bool) -> Any: + """Rows with no attribution at all, their parent's included. + + This is how a backlog gets worked through — "what still has no source" — so it has to + agree with what the panel shows. A clip that displays its parent's credit is not + missing one. + """ + parent = aliased(Asset) + inherits_attribution = ( + select(1) + .where(parent.id == Asset.parent_asset_id, _records_attribution(parent)) + .exists() + ) + anything = or_(_records_attribution(Asset), inherits_attribution) + return not_(anything) if unattributed else anything + + def to_read_model( asset: Asset, storage: LocalStorage, @@ -617,6 +718,8 @@ def to_read_model( if playable_key: missing = not storage.stat(playable_key).exists + resolved_attribution = attribution.resolve(asset, parent) + return AssetRead( id=asset.id, name=asset.name, @@ -638,6 +741,20 @@ def to_read_model( file_url=file_url, thumb_url=thumb_url, missing=missing, + # Resolved, not raw: a clip shows what it inherited, and `attribution_inherited` + # tells the UI which of those to mark as coming from the parent. `parent` is the + # same batch-loaded row the file_url resolution above already uses, so this adds + # no query to a listing. + source_url=resolved_attribution.source_url, + creator=resolved_attribution.creator, + publisher=resolved_attribution.publisher, + source_title=resolved_attribution.source_title, + published_date=resolved_attribution.published_date, + retrieved_at=resolved_attribution.retrieved_at, + license=resolved_attribution.license, + credit_line=resolved_attribution.credit_line, + credit=resolved_attribution.credit, + attribution_inherited=list(resolved_attribution.inherited), tags=[ AssetTagRead(id=t.id, name=t.name, category_id=t.category_id) for t in (asset_tags or []) diff --git a/backend/tests/fixtures/attributed_audio.mp3 b/backend/tests/fixtures/attributed_audio.mp3 new file mode 100644 index 0000000000000000000000000000000000000000..5b62de6c3f1cac9f5afa1158b83df1b7a0a06f20 GIT binary patch literal 8657 zcmc(k`9GA=+xTZQ#u$t&jUvWQ;U4>vGPcAh`<7*peGMU$F_sWTmT0kW*@Y+x*&>A^ zTee7+tl1*wJJaX;eExywhv)n-=UnGC_kGU$T<2QOxi4$Vz(BO(ZepN?qt-|e2&!xE z?Zl_);{-Cn6V(5IpMkx%kDtAlJ;=nsl-5T_{hzwJ2FOIyNP{ZEsiHK1Qv#V78*5Qz z7OG6J4|J9XBysYRH~^sb{ZFABsKz$A*3IbtmK#Yv6tQZXD%o#yJ z!Lw&^IGlols;a7{rmn88fq|KsnT?I3qobRfkB?7aU|3ifnH(D%``|%ZTH2E*d3kvy zC6$$xb#+ZmO>J%6-Q6EQ4h;>BkI&4^EG(?9uKxc0=g%Jsg*G^Kl8)3#%Fx*VTLkIC z{?o?V`2Izf|7%kY4LYdHb529jG?fI7;BXdD9stpow%?bD{&IaYw++EtsO^Xki~xaQAdI^*CiDgZq`^La-4$2YE7uqeis(9|*A4Us6!V;gqPke4?uAx; zZQ}udnNiZrkg$s`2}}@8r`(^7Iq_XHgfbtO9Q?+?hO)Ipt>Xfk{%Ug2X*^JQFgM+{cH6wr>bOYdSZ3LvjK0uHrNk&pYQ3liD z40wTQKrE3TWX+K*zqH#Dk+b0bAsrQYlYBv(g4|qsk37_%&H0IB?Xdpaq_U;=`L3D2 z>q!cIL9ue7Uy{tH;z!eKA3pFp7%}T!U!45tg^!GY>j^%#KkpavJlYd$pZ^oQnUBCm zM8`1h^B_U+0$XswFwu>{fXku;thTAf>Nk5^KDx3{&)+?g72E8w>yWY8A73<2p@OzP%4i~@9TU~k`0*4c0 zBMSZjK^Xc3&>t!4cpKzw+EMJN=c7!cOI(@L8{QxIJegcxOcd@dZk_DhdSE2`>D8I6 z<%{8mDpDz~J>=bj3#Khs$nT(_R)!3CD<_7TsF1-83c+aP#%8uNu4Fda$7g7g?(tSE zHFZ9b*{s>h>{$xr{iCHyzEGacU$v5$clc%Kc-x6*XId~tXCkrLcF^iVc;!pJ<%a3M zlOVOr_ut1{kj``>xb?b`5BO$NpZxfZ#@7yLJ|8ic4<#I-ECuwZ zR)Qb5jK7w8J^eP4^78G%+Lb2`!*8aScFk?-LL}BfAbv)iXk$l5%IWNc9y&lxLmk(X zS&eJfkOmYAUQj(STIR z3@`y!=5Td99dypA%9T)Lba3&i-Ql74aWG z`-ASEZn?Bmwp6@Y#Bsn%>5T_-ip!89eUwz=pnQgNstIdsUy@g)VT}1>G@zSMgrE>J zO+zPO_MI)BIw^7uwHE?qY-&#`86dn3t+DEU?dt`Urm<_yDm;`$O6fhzVlT=!Escrk z+xss-An?(jvYOuNo;(3G7Z)kk5EFZ@?VWKKlDX)MnO=@BoozqYJ<|#A26yw_qw$M- zqhIYP=KU(G1DP#X%hovF3MxDE3BMr(RV*mB!-DA}=!2jb$NMLK25h+L!yS6Bk91+I zt2dE5G8glmlqV+&)&E=^4sCgUr|;Q#$O#QeJ-9G`lIkbL>;ASX1&$aLH_Jr3|^sc5JnnnP6p3V-<60AxEx4z$no0YzZ?Y zNJhy#$W)dZj; zPC*b1gfZ49?q)|#_4mSE`iq>HDtH68c#rJDDaT=m-lCic8c-pGeuhV^{E&?0iOS=k z0ws0I7@bfyN#Vkxl^f#oi}BpJWccqaqD-_SuQ^WF5hU#77r)Q8p1U`H{*Zlx>>d=7 ztI0mLH(cfTt7T#`?84mDd`F(5?vNA^xh)yxtfYt8+;!$nj+g;1Cw>UDOthAYi*RPX zkvjDxBXbOuRl3=EmkCqjs(%LK!!jL2p=^(zS6~3WQ~st?d$rkFVDT9-$37uXa?mLG zuB{Z0QN~1fg*f-rCCTH?LhOj@e$o?jv;w3hNKDr+%{*CE!Vmn+-DM1Pv&(KvHYFQXId#;&n~+S!Ko4Gc(G%$l7Lz z#FCiTaf^S(#MJ!*4gjMT1F(TU0&I8#fSa5KXm&5+wwp1y`R-moitHv`fEgmsg~Z%H z?<2Qi?OO0uD&X;cCZofZ20p`Il}BAF0;bItv7ZMUCq}LmXk_c9*EzL#Nl0)hn0?ip z~%jfO9Nz%-r}v*8stoz9@jJJJwXR)wk+W2I#d!Ja1s>vn z{9}b{NzQ4i3YA-CH7JRp!~UJMbGNQH(SUX!Oi_3QEwJ)M&dE`|T)!p5NA^KH$r%)Z zK3PPEfjBexBDFacg^E(}BEk8f;%t!qgxVejsV^+ejK!l z3>;{q02Fu{Dj^FRfPyeh;FokcbX;^X>JI8@EX6^&8dpLqD}S$7n0&LZM{liod|SS# zI~4I^4C{_c8)IL3Xdevvalu$Nih3O5|78k^oDNo+6`5&_Tp)S$cexGG zmH8A@Bf8y0mYi?IcOoB^iZoE>Uqo%ZfpS;3wsV3&GIV8J@6NZb)E>VXdvqk1y)M3W zJ)=)aL_a!C>5u~&$VRp{ z0ac_B#a_=oaW*z_6pIWx-T1~zPE{mN<@47X9A|gwlWabeNJW+hu`Jj(8LCgS7EA9;HoiM5~eR{dr(FFybMlm7G2ZtgoOn*K7D(V;urI$zR9NnXMM zEp~}TAhkh~D`bo$56dON5DX@~GlN}AQ z{^n;kZcy(dQ*C2p#j}bu9%&K#rFY7+%iHw&JVQvJ2ez)tm1|Z)| zM%+iZ1kvq41Dc00AOuj7R2wQn=2}uU1jRwxOlJKVqSHYUg=jRcKp4ao9Slk0);|Te zyb=-3-uko;&8ZO}rw4a;S$X{@f9z(UjGt%C6<$4y($x)eH=B}7eaIHM z9ISQEdc{sdBJij8*T(=-O&5TY!qE|pVEA?`z>(1wSDe%q&tcK(1FxH^NnAoKsejr) zc(mp`E4lnrA}d_Jfif2wbbr{?bV56Opxn8ks4!IW)ipseEs1-p)nE4WR6ga5DxDbC zRy*TD{HGxZvQB{c6BURMmHUJ0<UXjkg zl0|tJK&=@0QlZ)>b=nPcvs(ke;6CuflI?m=_??T;Sy=c#)81yGbb^J9SlS%MdWhjt$&1U_XN_WLSc@I~O#gmB^!gKhfz*8mFS2!Pc} z00>?PV1sA^@?A%`n{q2SAMF)D8=@$maQ&-GiZ3(MfuS6RJ2tp+mTmWhN8kG;&{S-X zB6juj+VoHF%FsZ)2hl=Pc4r+M7WduxANXpJ%E=GtWItHF=SdJq zXnkOos$FZGFXwiUt?yM$XbZh?gRM8DXkvWiRM1qn;%Kn&&+3vJcxacW(McuGCGS@bc6Ktr zqJtyR1u0x$==akk&2kqPBIl;I?l=*WoQQ0>6_XPF?MP7OQ9VSkkBLMBT8E$rcmKI< z(hMu;WAQA*Xbsg*{6EdQ$DZ4}cKccpK2;jszYTCJI$Y~#pjl9fY3?_V=n7+86 zIkLHQw}sI%v@ip%_d@_M8^aecu7AV60+AVBMB=zJ@Y zexQvGbG~b>Nw(|5$-iyV^o?70KHcuyV0mDZ`n|pD9_9`Iop)z#@0TZe?0w2VDsLJ3 zb;RX0$SFUZ@q!R&Tmti$6S6yo>CW+bk<;99+CDQfIq6Wfo=NCT$Kf*gk~dKPFtdQNRY! zwO%*}S|c)pONdgD4#aR|tfR|dlA{^Ym-sN!Ht@+M_P&5(7vEOzh{z5v$R#Duo$Ybn zN3H*Q8I2$slU+3(7GnVL8U-$|IO&7YxpfPU=Vo$2$w?5a;`fOifINw7;1)-+nhe zv8TpRvpB7J$~M|~jraDasgL1@%BIvqiir%@LmfV-ASdi;Mk>ATOa8Fl#$PJQPgSgd1n{@| z0aBzOK&Lea2#=7$32JmGgr9*cT-A7pI|a&>k9zQL%R&1Rvj1(l`|ZA{PD z4P82)ZU0md^46{NAAOEmgfg!*g;P+hFBvp zQ<*V7ZjTnfWhlT{*O%VD>cfvOYI#Azq^0jyjH|YkE$&i+belEmypH>|TlOySnGh;< zWok4>L>vp02~yR~OsYbQiE=cc9SB1s@9By|es!LwGg`-~V?bSkM600MjN=AQnSe5? zhMqNQ?S`X02P!V-bI1&v)_f!_;eWp?Ewt#|_s09(uNez)TuV zaN2#E;FvRlV%gCAbV}{(38vJJiF78AXfQJ@yP9QQI)3X6hn;?CJ;8kGKyK6QtbSC- zDHgGjk+e_?t?AJ4?VR(2sjM`hMF`U3Kbb-{dKOGgeKN;z)qKy&@rg;x&pzzMjT_8n z43KL&7%`BIlnpT#LYG-wHgjtH_ z3)Gx^1wcr80F3kyIEiNjWXRmW9qn)#trx5^GTJwRyJSJ7^Btq5Up#pezkZQz90FXc zrsKk@)eF;DpP>JitQcI~wK@HQkYyfTN;qd^>R{KX{rw&}uK3yubL0)fOA}{Gm^2B^ z-zzO2MKO@L_Z0I^%l-o*BNYE>2>&B~mZ5*>vXMMo56cz%*sV6VlSi0c(H4H!XiYnE z(JF?`fNf!rIph3-BYQ+z-h#YH`dCkz>D@D8Rvv^*UKdU+E5CH)_E8^)_3gl-s+WgD z&!SY>k!Po*s_&Y{S%D1Y#7{C}U?C2UgPFtygqj;%>6*N@BHA?uv55VN74%@}>hkj< zbcONB(nF|rv+HDJm;AY~v{JOmp^W6Q;MvC~_I$f}GQ^_mXsTp4x#|~XfG7)kN7N+E z5M97V#5(dnph#qt04psyy%5FEKy}+m_PGdt_kmV<_jvko`}+~2zB8LqDR2YgW9A_z zfBOj1Ic2V*@9ZOfpA!UEib}Pa0#?)3e%Bj!8GRZldoUh?$-YAVBd4gG{?I|;Tw4*R zQThVAu4jFMI-8E6^lNinmu%~saF7*|0rZ|I5^veYEoRmA=N-v{Q+G#F27^@S&%y zPm|9Z8qjv+6^sAGA@eAEt^N%i18GZ4ADyvo&V3F#;HieT>`AU|nZNC#Kq{$C9)W^* z;0{FllD_brXloS_mU}=6|5G(WxscybkR9*M_~WQ7UhJ&K$=$QsRpBpBy}I)D>g$Kh z2dsl5{CLKM1a*HI0oU=S-w1?RzFf>R0a+b04MZ6C?|Ah@lN`C{e@B@d&of2Bem&$( zzt+!kfRN6?^l%1I*EZ__E) zMvVrv20=a*K-04A5A1wx$LyoV>P2CJ?qb^e%VD=;(A9PezON6Qy#}loSVlS3f0F4m z6#eDi3Q!aAd(FDH0vePw&W3C3Ncp()=K!4fGCh)uL*m%2iC2-nnz%T)~p1-gV{-l4}nc4NBqM}9pblhz$ zOOR=L%{Zl6{@Y87womQ@7185i&m#tPzG-g9!4ajlk$F|R^R@F9f&-kIR-bDNp8A|` z`?uvFy8Hi0POsC?no>2?WPQ%U?o_}PPPK72+3P)A?am*Q=a(Je(lPU6eO_xo*BMf! zHlj43r(8CZ5W#69!EAWVHM~J1wJhs#woh$l@$~W<#m8$z%T9`L_fohzEKW=NtXyW> zQ}1VPQADxiXDoQ^Q=-_7sE5p4@3GqDCa0QDc(~X&Sf;)~>xTL(Aq+Oq4eU+b5@Z5&{o9u)hB?*$EL zm5lPmBbdXu@F@8dq=_4hrfCD^26g3jI4>Gi>noAf3fgwxRX_j65{OAhjU$$$Hs!Ud~DK@P_>3fdhu{R&cR2av?5yM^WKoe zT8*imRcldV>?_mjiNg22`a&$#^=-tS?Wuvl>3Z6p%8ze*-ngd}eM-PEPl*GUYefUv zfH3_v|Ch`0LccFjZ+-X;qxu*Sr{+aLa6m>`kx7UP3b3_`o)a-T=l<)Fi|(6`y7@~c zAwTq=irgWaYPk$e=pIGtJtnkRu3V$inFgW?6dwp!5fKYHG_`(A-V zXz15xO2db&C)g$1N4Yx+d|3-j&AZ}vB?85@In`Yz3Z+l827|y15ekC$>@hT;EeMmr zMT8_(L*kL&3tL=u)To1lBiICo*rem;7%OTSI-v-)>hrrAZr+kPCP@F zCBq$=Ar2F2a$|}|Wte{6{cD?h0Iz{ImT~E6*PNSjmX3@1EFYYqFE(w!O`&ugyOJ%y z)dI?{&%gha7S^X4e7Ohn(%G2S<+t@twN^zG3T6L}u|R@eVcOo|0!4}{Q<|?%*pp%{ zo(r@eR})FfoRIJ%@J4l;5)fHIp)5QoCE)e7WN4qeOKF9V2DAfZ)TDjEbHLZ$!&2jr zPTAVGCRnhbR-7MRFTU`VRfSn+RHgG9wA+fxiC6fsHPde@KODq-Hq0sKf@{CqQ79E3 zRP_J*V$=WALIYYSG1g&Erp=J}zj6yTsE|Par!)Fr4o(Al6WLWKxD~cZqV6JRLH K`Og2-$NwLnqn17Z literal 0 HcmV?d00001 diff --git a/backend/tests/fixtures/attributed_document.docx b/backend/tests/fixtures/attributed_document.docx new file mode 100644 index 0000000000000000000000000000000000000000..adaccee900b40c66e55393ab77c95825c95314be GIT binary patch literal 36620 zcmagFWmp}_wm*yqcXtU8+}&M*1()FN?(Qyu;O_434hfpz1b25fu+jg{%$aj%&bjw} zzcde3YyD(Zbyv|{D1C&4!UO{Yg9RfOWY?`$F8-1X4hA*`0|tfxYSj_5w{tPIbJ17z zbTD<+W%RJMX-<|`ToFbMyLiD$VG<r0_yfzcSt+m8-CI7NT!}ujD{oa>&r9q&_r-e#KO#Dn&H%p_C9J17>aIJvx+>C2a;tQF~qVx{Y+%QaC zTb2*rG$%OoPTe4{K6>bdXr6XZvL+BLj=DNt9i3z(6joVa3_7B=@F!P(7lnhW^6@w7 z9XgA>F(tQ`34eETpjd!J&dU-bm3p>yCVWRAcUq*urJ7`zK~v(8bG_RnU89q^nq$LC ze=)Mh2!cK8%&u)8qx-l9+c8W-dCaYf_By&%=>*^1#+w~q-pPtOokf?n zR-CG@<#DhyzSq4Na@I*-5Dj%BneJDZ4A!prYM+q%!TWof;`sN1z@-Si`?*aXAFhy}_6?zb;XWV+ z*4#@!cbDwBt^^er%h9LM=)5Eo6`}QhF?{lVL@2Gq6Bq1%UMyYP6B@fGFL=U~*8@}f z$0&MIeH4BGjovy07#ITRr@oV^jWZMD?`w79q%0&0YQTk{#5c-PyB0O^qGdhtL-``X zzO;$+5<5TXl3zXDpTu=FvHOV+E_RH~7BUM7mgq`sf;Eh^=7aJV`!zRcu9`aaS7B4& zx&p~ORJTL4Y%-z&6v)D+RQH@_>ZF2n>lnj2G>zv!MQTLF^=WOTk;_8LlhAGvTs~;H z`ZCPg!`Ms8wQxk0R=4lkxJNw*wYW$d1PC7|IH0MdE10n0a2FxVIAYw2$sA0_RV1`$ zVRt-sDp}}MUNIg9iz%w;dAz@cne0Lz;x=#MFa6oJBel=)o?=qNYs$=ZJ@}Kq!!+?J zbA9TCc60oaFYNc9I-D2(EKnzT{&Xm4ns0&%ga#^*iM_Fslf8p8ld-*%>F-JIo2aO` z!;Ct7;S*Jsnx#A%sTr6IV<|%TgPiOW)VA3v836t5i$KEu%1)c@2koZO^!w+VhuF;M z>hngFOA)Qcv7?3wHs%3Y(+;)Qx*o={eucT39Wp z^u7=s2F{2$H1r`Dc9ii$=1)G&dd@M$7ZE$#vF^eZ4bKwnq^X2F=>egcknlJopdE$~?&aqi|azhtkqiEPxg zG2V$wYIhWNiLoyDT2o)Ix)#S)ZRVz+3c;(RgtGAK&tuPyw-q2$-n!zYztNDIgj#K7 z#!nUX6saLd(n%)ru0}sEB;bKcOYT3%>~>`Z=L9rnHc((-sDB?bLkEZ7Lzb;NZnwgW z+Vw;SrRx%9`A*~$Zeq}a$WO5~{JO-f*10iq*)LG-EZcs*zz8g!FeX4GUnAdB)7aCZ zh}pX)1q(H-abUqWCC4@R=Gv&y{gufBa%ZRt=yELeVKiMw{pqplc^C#bmvjxKnlW$k zMx%)<;&%?tigeR1zmSF3o4iXp6&!qiJevwjoVwLN3|%HGH7~8FePULOs)T&0wW#zg zn-<`Z+5sGEr6eOEOF|gFutAM`R$6{oj*!HWCnaDp=)Rn=fU-jHb^$b)9HHfz+v?!( z&r%4V5Yszc&muky!SCW27i|jr{QyWhRA!%lSOLR{YC35oRcR>Kh`5eDOX$@U8O`J2 zfP4}A5c7S8BH7B&zW)|2S2J+cF`@0Qv79df`Qsfj)FYqh(lfR=?Ju!++C&LSE$JBk z^<(xor(E2KN!8bon&45_yHESsbr@fTEeV%9(BAt_{S=g_AFr#F-Da#J6Q~)XexE}0 zWJE$hXH0Xhad05EsI9A@6N{D0fsf`!xfSZy4!&nk4gJ+x09Q%MO?#W<-U9yxmGR1! z)cY7xjCeR)2Y71Jpe$R~J zuj6sQm@&ideWN1e)f1pBWGM#9jG!`xrOTR73<#;~BYf#s5@t#-92pZT&idxhXG9gGf&}Blmrn}u%~?T9DR?T#5;t^G)-rDbchydQH$ClB zIL0NeJnAo|Qo}ak%yGIMeTlrZAp>M>)aVk%I#Cius~9;6dZEjxm1yH(jR_v2>?S$h zk%x-G`S%}4d7>*6!@S$8f7l(<-BOURBrewM#dZld!b1nMQ);yK_G+g*J9*54Cl_9!UYE#FOEJSa-BS!*svYz@||TF5{@*w zUaT5_+R2{W`C!LhX7ol-=#gJ=4OTj^Duzhjx1Mpx%=>PIg58kY4o=~i0FYsCZb1b?(l=^)Z ze=40kzv;C1d%^YL1MZ)DXrJe#ti0I0oM#Fxj@rKhn6`t8?I6z zlB)i`ZpAR3h?s`ATgygm@2B~daWD5Mp*IIj`$4q(&YhcmfCt`L{o0-IVY(rrLEEM;0JxCl_!uW_YME`GnDbs zv&S`G?jJ8gqLTt=8tad<*S6lglK#Ygg75vQ{D~vdscfoTdT&C zhgtk>J6C;sKI>fcya2==&6HD#2OT%om{2zS(j-WHq;t;5-iB&{y+-}D4aNU>d_?@XjexA{)oj!bL7-Y4P3Ac49(8b~B=)BY6jVR%@H9{|S`o3C1 zzTBZqU7hi!p*a38uLpx>{;)5%YcDPtpP%demb*6i1*+XI&cjo=&U3gFPu_g>F8l;) za_tu1Fot?kV%mQ7o+%ux#M^H_?7r1&_e$mv^mOP*asp`{`#8PsStzMrA6E9=xDF6_ z%CCmsIwJ_F?u=wS^_U8AdkNIxGqzzH(spE#3@cH#%1~I8Ne0wNjH(2O#C;=Zd?IH3 zT-|zZN6##Wsp0ihfssM8s(uBV6v??C+Qj#$;fa+%P5pF2pX%&EzwOh%3|Zm5={`(I zFTE4AQ&@2|g}|pR>3Dy(@zTXlFGP}3JMI2+RXahMbs)TTG-;_CdEO2%cvQFHTZ3t= zWo7i7x_GeX=v;+eRQ4E;cG>>4Xn3PaV>Og+{t*1urCYNw{d6|_=y31&GAeU-BXo8l znDt4rS$BT!a&Z6V=IDITIg6bxw4Jb6Blc}@-=l&!>*j-6Fv9YWy$4GFOOi#!%+}Os zE2tk4)^t&I6%yRSXkWm^i4BWbd-`5F`>rSmqF3KEO??8Na?4`uHR74FExQt0qe2%o zdEIcH&R#E%Z0WLyM`qA`u*mLUXGDC6^gqbN6jN{~30|00s(o7(e^w=nmzDk`U52jw zm8~*_eh-t5>?6aXENr4eM{kw?1|idf`dHT}=0a5$OJ*VXC!4F>)|YbpH0>z**QYhh zZoZm}p`zueIu^DKnI%oz?IhG9PDoi;|Wro+`rdzq)8h5p371vJ0hQl_}`ZgB%iD14y%Wb@}4IpcdtHRfcX-+JkFx zf7cO4Rk+4R4N2;P{>C7Bjhzsb&;?zT_q)>q)QSC9C)Cz^dvK9|#n&|BQ!+v^c_Mrp zkXeWL3;T>c5tK>_ilg}>RU3pB3c~$I>Yr3l<~ zB4@FtDYwoZ480s)aqW519%6M@-wPnPzm~r?$wFsxpDesYU@G<3gtE34kDd+xb_|^* zTW3OX&AENY&gIc@n^*c60Y;sa=ZoRBCxC0auHCS=$n)Uhgj>GFlE;nC!^muI?UqOF z(Y&DI!QTG&f~|+x4I$lDfH+B<-8SR0aPsH43yv@NZ%+=0Vho&!J3cdv`Pip3Q2_meu0NzKR$Y7UHYeD1;ebWZ;Xf$wj zjJ1bJa;IX*g*nF`eD>6E!mHUOZ%dqXr6ib5wh`0)ejG+e!y7s1ivDg^h7g>(9gV@( zcVuDSob%wVWlMyj=3OkafW;gViqq{?UQDWH98p%U2x_0$4p_ha=bDYSG`}6UL3_!^ z2YHzq%k}7y&rJBnUO(Fi^mg&LB&J9oC%u2t%neV53)&ShiaR6XWH{(r1RaSc*|q)9 zB*9VM{0s?mfKU1`^Y!GTikyCi+h8iy;{&=8?|@U2oJ{gfGGZfDhP#NuJ!#w%;_zy? z|3+z*wCM&C$swx`G4o+{^2KzOu8; zQS#H;K^T5iMv8oj&>53%G%lBmXZfl?u~RC{MxJivJ&@YZA6NjkW8L&n8F6hHSWE9) zf!CpWOo?QZjK3txS~t%D$)(Pa%3i8#r9kjK%k5$UE_CSti^tPDtJcWi#G{DSK!6Zz zKW7m*OBRf<-6|QIk+dh_FTwqA&Mzae(J*2qL?T_{f z*qz!L7VkCr!j^wrdjH}wxXRuEZC<_N>SafF{br&?f)OFg)T!;^@pqx!>}PGtoEm4|g^ zW=nkamM^d$002sLi?a}3!R38L83%8O8cU51RnnWt)K$^1m___>pJ4h{Fl8WOhv!m5+hRD?7m6SR2N#P1|L2 zQ55g#;yMh5A6w02eoZknXGfdxwIZ}wdnIKTQH~a;5K?G4tvY_(ws%Ql-zl1l!FbQ6 z&|xs*#+fPe%URoMnkG@8RJX1)jG$);%JNIrr`*=#y;H9bGDh_D_$+6o9nP2P3uvp3 z&aN@^h@n!O3H9$xV3#4CbXwd6_XO>eb{js2uq-7*=M`C+ijbWAl3;%`Sj72CB5#Ex z5G*b__`#B5jFHP^I&Am@U(s-~8p&zL9VT~~H~bQTcQst0AYb!NPsf$u=#Bn-e)YRS z`lYHlc~$D!7%_E^aMeq$im3_F{j-cr;cMS7YB%3d+pW0p&N+WiS2*u%`QgMapIS?o6mvZc1quSM0?8sXTQ zhL#6mssZY`o=9mqjv=SkUDA$GYpNs(v|{O-16n`7dOKRO=4gm?fb)EOw|sVp{=O*Z zj&gXPVwRl|?a<6oMb?FqS1PY8=`i^r0=z_aCYFg?QK;fs54x~8@UwXg@b$7uFsnB@ zj?Y*%Z&C!_cD*H?%9l+z7i+lvR8gPTEQC@ zG=@2)ph!U^C@e`QGS_k7$U$qC@&T~FXJqztJm{L`;xZz)m-?~@bh^JayjfHe8npHq zxquBn=!K*hpe8i0e!$l7+36h0qkqoqw9cV6@#*Y*E)xU*GI9Nt96W46Zly<)A=im+-+ zcTHnScup)K%nNb$^|BSP)#X$_@GNJ%b;8kfapO9mYr8luM!X*4@Au2bxB%xxubOxW zr?Jt*`!hfU!7&ie(fo7+yzo+7;^~6v#|Zl=vy>P zm}5|ec|TOLw#vAPWYO3eKrd=*q9tlKB8)M82eC0NlJuOlY`;F5Ba1XdEn&)!%sA(+0Is~rD~JZz?k}cr-FT1+&!i#Lt;;M( z3Jn@hQR_-rMG@4nVcaMrFTo(r~QVh!S!;M7q{;@V}!~2wZg#U3f2w1Q-0#WT8*H% znLLQ@HR~9Ma|zSLqbH1=&(%Pm&t=>vvk8n#-B3!;WqXPNmFRu!I`=yv$;VCNw+|4r zw2Eyb7nOy46^Xx=Ywg(*W?+Qo42m2#Qz>Xfo|6`DcbouwH->N6X0k90--|wO4J9{P zdR2DWz;i*W3Y={a+t-{cr-==8jc&%6Au@cHfoDpb%xJ-b9Z+R-bG^KV<7D4#0u;5N&4(rAsOPPQOJxRy%9VcTXZt zSX4R5Wh_|FQfsWHdyMg-*$joqQDT}T^J-J=E=#ouH{rvY8$YrAD%pKGA{pYP_LOwE zwhFrom}4!EF5@zvw;7KYEYySd$Pb{N`{*1&%&5k6Ou0g~tN;zT_7W}r#MDKGnXfnB z!|25o8dQCfudYHQ-px~Krc->22T<-qfacHF%MiPDD9=~~B%V3glm>Zub~nwG;jBwX zE7e*F5q4)<26p9WySSnO_Ny{%S!T`WK7uo0UWRyCBL5sLK9=Y<;d{R#Q@HL=#HmQw zM9>Qb-JXgy1?5yO68SByJ;%SS@HLE=7Tl3X*b^4?ii8D!x9yjWpd_uY`&+~F#p|UX zb_O-RDUngkvSXuM?;-VK!D4P97pqV^Z;o>6L1w9ZfwTmswV61IE3%~?!E{TfGKW}Y zOboY9QLt6OY@`BxBp-guqvECO!TGI{Q66mYwrI8~F64)`|@Iq6?K z?)Zuwu&_e0;ZXt5nJQdxE0W z_M8)t5z=ae(S#p=R!p8}r6`QF6F3kU6@b6)FGNK#8Kk&EPP(ix}a(2Z8*9|1!+`BZUg&@ZSKRUGfw>j0rGFw^*x2{7& zI~V8ZfR2z>?Rl{e{N=Dq3jb#{#YLzteMSjBODW)(tWNwdgBSh@>e=z7$pe|Dz>A{T6QH8)BH=c@GagDox;n+C|9%_x&5yIYdg0fG~a z%C@OBwxn;e#gA|xp|2G60P*AVsng*vIM_MrSaYUf5Y7#gH8I|a!SV|We%S#n=D=8?P#3`# zK2RDw+XW^8lIIbT6vi#2vx(G*Mc0r8GyP25%r|smqCIRGgXw9)=-W>u2NxrjN8+?q zC5YM6Z+Z+`a#y>W#LNZ5|9q=j99i@;G2GyphjSR!-V((nI=kVBup=Qh;YQxbh?>BcM&oUwk*kHs(*Yi*Krgw0L6T|2;+4%)Mj1^*>^o}AatV1EF^t); z#hx~?pOq08%o!?N6o8xs^(`<<7!DQEh~+vcV%Q~gSjnUXBi%JHVD&rtGMm)n-!y8? z&qc9zBg@Z#d{3WmfqcUW4jDzASW`y_9@I5%!T#Z!K^Ep)AiXS1MBrKU7i@9*5<2=Y z^2Y3MRkr$|kN9pamR^5T$;OrOTjgWkZ8oj>p|74kxI8N&+o<|p3aoS_f*!9I+7R%pd?3yKa+U= zJIRL8zMLzSdP*iMiQ61RHDqFI5NyAlr8z=9WtsK;$4)Z0IYR{{sK0TAG?=^8uR~)$ zD_II$^H>TT_2B*Qc550LqXy18B0fh2`W=h|_9>IYE7`0@?0Ixh|FUg@NS2Dw8lCgh zSTF%SAebP+`H%MIukRZpzjy5ehj{8^6tiZzlk!LD9H+EZKmJzIDgxuE@ZwTC!9y3n zTsIfLO!{wyh-b_fWsuuh7Jn+N`)`G*2U$WkanEMWKcnH1tfVFrKBkE6@g`6sC)a|AqsB@&Wm#(l%*VDf&c-^Z zb?YPpI7a&KfRnMI0wk0x5OoSJ^2guQntxEIfT;O3|JZ#`0b1wuIciBQQuQ4fVSJu*ZboF$Iu^@?A4G`95U}r4B@(rQTow$n<e@c4YgC}EC?P*VQbWly%P6`K7;UMmJ7&*Vk@gFH4LM85TZ`i7Wx z?|R-(`4>4B%*V0RjscCfj@W_(?2;rw;oPxUpQD%S$8)5Kv;NZTpK6Dv4~r5S>gd&@ zz1X9t9F*LYYv%ntZGfY zj>|d;)_W?>d><0$Lu$&Y=g(w92@LrF^`|Yga#SPMfrxTtI6=kh2!S88-hD{#GcyBf z)m9VDkk-v!=%K-UV$-@IK*>CQZ=UF=cq>QYMP-%Z7a}V%~X~+4nm+78#Q4 z@_lY4Ry~dxR03+Qb9+domS^{tGg&3`wWI}>q#5e#bmPhw$r9|F;R>!Z?BCm{GfL3pG3v6*GwicUWE9niTs>zkwNHFcTuLpQ=^`xf zBTAC*k(=hc-=WK&fXQQ6yiy9?H_v}Ugz|b^)A>o7L1pc?ldqD^F3)-%Sz;+Z3uR+o zu}@HHIisQ`rMvEsvEWCdIBBW<0#BX+Q((E*t;zt^4LC%VP(ZJmcTMGB+}*_l~$)0FcQHX8GS=$%f7@al6ZYEIFMEa0efz%C9owOxp>e^z8{+ z;^i?_6cr0s^v#Gm@l@%t^v=_bD78>$nyWp&!Y7T7-lzn}jiKSfPep~Lts6}(i{c6^ zEe%oyA8Py<9AO~t3ic=z{-88}3q*vQ%M`@m`~darAxJ>q#_+!W2e>#q6p8RI7%GG~ zA}mE8v`hh%r-(p5oK-b7Uks)Jv6yDyk!j-<|K(73m3&ixSEZFl;EFGMG|uLl6dV(# zcRKVAFs)H$-6%|;%otx7v`kWimdSKtVMJIa5ZX5MZ!`oefgijvI3P5Sz(y_$&YboE zy#*>g;chjKu0Gq~)M126PLB{ActfF9oM<@n){;p?Z&A+3s2tqKufC%s{=r&TB zE>fm4CJ&KvJjL)UhzVm@rn=su=%^6G7reNJDU<^u-jgA*T()Un21C@&E{6G`}xA37`v09RR^Vq~Qdo2pRy*^ubo~ zoLMix6LAart{~tG*Y)y-SAAF2JOClBwIU3sjc=~e1H4Bp`%=`gcf~~z*Lh&wG)%xk zgGd0at1}_c+$OH3_=#Gy<<^@<%OA>Aa0(e1+@ECff@^ash=w zM1^*~ts4&Sqz^(`I(3d$)*?4t12%a2O}G!v8=+bG14Ts6JS)OOZ+FUxu@(ZPTs|+y z@u}_Hg`70kx%MEn96+o0=_N?{#6J*U3E0L6`=AUnuCJ0fs$(9FJp8h+%%1|PeuDS- zTo*i7>pGjRKaAsb9Q!c>=o$cIfGG?dj8sAxdxnC6=1?vt#-8CS_EOc%#Qf=K+F76) zyMnR5D2nkFJ58VYN?-%SPu9o}CP2r?PnR8nMToFdbA2^4x)gw&uK$FTIG3JFD+?Zv zz=Skc^P9&Lf%aoJG!{!QGIQ^dk06QTUBimfJ~i`)vSA58T|yl8eBl7TW<(s_N^ODF zY7cYo{)$#}|Fb07y|D(jmZHqWG?c@~iP~UE(*1Cw)Uxn^QASsmS+D4c+81^A;|WEq z;(*amCWso{lM=SxYqOvfFE$Y)KOzIaq`;FRw@fqgxj zV0i)NWt+jfa{WK3TIy1~bTt2j@zAz+&ecCTD@D5l()=d6 z^!01DUUS*kSl?Vxw!f60sD+$h!d27_$H-^=3WP(1x8l|lpfBq60F>3aZ5Ql}xuMkJ zw^C2khMH~J`Wzg*LoS7y^7@aV@_3yyPVGpu?3UPHd=-KCtA2daXWCrJz~qaIVEdCy zPO;mV0(q?+s`Knd#t7a>B*X$mBQdcLUzfhsK~b``*d(IkbyB-jgbt!{@-R2igi94H zD8DDA3hUa#MZe*ni}-+nNbhvXM+APDrLDNA;YyKl{tZxRjaL2WXUnL)EmdJSk9s3H z<^T97tmX@~=*5ezg(|PBCV#|u$m+ZO$(6i1u8C~gD!achrW%W0^wz;ZLh-PPSm&jQ znD**ZP#(Je=trOGtjI!y87`s2@^9hH#Rm#G-$uZafwurZm6EP?ZXv8kIfmQ(17Y9p z09oI%krJca^4a(b0!z0HJfQlCO^^4Upel(5-9s|Tj9Degj3l&>ESl=&iAR5Ja^Vk` z{-^zS0$9(eRDjh$v1Q!6pxmBy`?0T!1P7%-x#iG8w}fT2I7fJS7Y!ql8K1JmP=`SMq*=7ZlBdq@xycy11elV3jJ_3gI z{6B)90g_AlU31ixxR zpQ^urXNoDDyDT_i$~H@D?=xBV03*QiO>?V$6oGfgC(x>wVjP@U&vMy*s%z4X6es@x zwtugF*_joi04+rOV3|fBxNh&ZOSW^i8!6v=+~*I@o{wP_%j#9^Bf z{exR2>R35caxZRiOZUvRQ7e8OZ^bj zq_5w3E`bS_d$P#*yB&J;A0x#kvXJiLV0(7Q`(eLeI9t3^;L z#XL{oAX}&$kt&LL_OJJ)RTNc;%PI_Rf*}o%K5(hV$Jte&3x_^d%ffCe^yrgZf4Sf5 z1bno>fTD(tbChHZbb#2>h1yCne#hwj1-$=~)1d7vE{MiE69@bX8=?r@`}SQFSMh{i zfa7OV+PZVGY~@Ta2k^5|SM$Y-3@UEfPOsep!XRWU9EIZtpDl0l0>?>6xY?Wz z$)<8HA|no)mu))>abH`EGl5*IEdqer@&gjN=QdX6gOmflOW=XfSJV%Y$)i7OH)A1= zWjkF%=-Q?jU^heG`&Nr2LTqijHBDeJ0Y*}W`s54@+T*^ai(ql5_0@r4Q!&$CF=udLt>E3e#f_nu@Tgw_xa*& zo0~K_a?^JcYM-o`&QC|j$+=h0>fx_qWJ*=X3R*9#BgijI9mp7yFo>sL|sdY3)lcomjZH5(TZhIHw6tp6)yyEA@~Y?;b2-_{CS|)V1|c-l z`^tvKT+B=h6H@C>Y?s4=vW-n4>OWRh6;|2o#!GQN`3Rd4$QzlWA90kfu6}iB|I9>Z z-PZ@oX@J3}yZOA46Ya8Oo14tQKb;x5VZ}qIV$M9A(#H=2d4VX02$S9qyXo!;LjhJz ziYSR)pC2oWC}|F3{CgK5Z<@;Q64(EY)ys%^&5(I&Ag8md=7>9?q8E|sFREZm5S77i zs$ea8Y{+=u9=zW+`zL~dKa4TN7OH3NWv{EMmhdu0$2O4)B9qA)8$K_#PC)I7g8Vi* zoBcqe0~8=3EC;IzfGQHJm3{$j*OwF0P|eLZItFEws;MQVmCoU2ZtNzJvK)}2nJ(@% zNyEiO`I&LC>s_TEBnBO5AU(Y6K2P5)XC`W47*%@LNv?l0Ptske1`UY0=`hR|zdFnX zhC&r$dNP~?x~Do77|uUN^dWGkk$U2k%K}u{1~qsi(=|g=XZKy?r61AD;c0(o93R7H zX0B(=t_mC(3u-@@Sk7kA#4yUNqEBNo%W|I4Y6VS^kYovJ)~j;OXWFlRx^8Y5FUPmw z^08bmHLlw<0zKcYpu6RLz0Dd+G4@l|hYQTAvO4n-w`op!d)H&ho%U&ac^5dy-qS}+ zkVgr#gI@RwnXk~p{H!%7X}CgZ`J&Z_`5Liart4OyB&W32$L*U!(dEq&7~&TYvd=ez zE=QUs%p7PC)FdIWg1}7#?g9x9+V(<=UFlR~(0+B9mlK2b_$VI)-yHzY&?NJ|Ojs!} z4?+YQ2&os0N?+HTXs}nonJ;SK>1@yFSLrGmny7AXSq64-BdL^6qLdF@*YypCHynDv zi1f{E77JZPGR+e;v}P574ubCxhF-ky{^C$$g+cq!XmD7MhkK)b5&};tkUXOVzOEFR zCn}UDdL!@j2!l29>ar!x-d;&WyKDCW5)0lJ7mgNH^#>HKMB`#UbsYJ#J+Qo%z@Dgiu(pw1*H_ zRS{JP2Nmjzv=cr;b;I_}3S~+SPy9$&n>y2Qw6_qnO?BCA_DcP3RO>OiT&6AD<$tNr z^??0T1ys(3gLZ6#lUk)U7Qg?P<>pwInoHSb?cjbS3fi_t{I|R+#+XOyYDCpZbCjmA~YYl!ZCccMemvLe6vd4&?!Q}M-=a+v0 zRsR75|8Jm15K#WB@i{;Y9w5{xDQ$SFz@W2h*ctE#!j6^z=-J|Jm?;eoP{;Zr`d#50 zyK=xq$onE;>3sE~ye5q@fg+_h$C*d6(BD3anBlXI(z*#kmv;s%ihlGBZWG~YJ{=u; z5%1PSV9IiemVANmp?FUQEeh{S4&7WWdIrji=blIT(*d+Mw`0j^?QmM=rx;0iP$kxR zLw_lr&fho~N)#zIps>bbZa$I1(V!}qVbGMFVOoLm{7-ZIKn~0oU}+v&Iuni@4B3?} zHVx5Gx~iTqWdu;Vy2wJud*j|7{Qr4A-@r3ymWKlj%()f@4F8|!^EE78EF{d#OpRUs zJgV=jn_4K@aP|bmYCaxT%t-+w`#$bFg$^;IH@j+RhiA`|Z~Iy`=gIvB;v46~zFoWD5`WD??uP%( zh5n8`^~UPiT{K{4+p|%x=WTy_?>zGjc=7i5dg9*sk~=-VrhEMaIPa{jMY*`hy#pQL zU+-QK;Ahas?(FTZB5uiC+n;^4N6Ace&s5A+M;Uk5=UN$cpEBU)e(qg7+c#`qp&Z$d z23$n1y_C17BI#Q*MTg(Kxxxg?T;E&yTJGG%an_EzI$h72JvA&*``Zao_b&N)dcSa5 z-niE5`+gRvz4I@z1*V$$`TM`M==!@`?W~OZ7WD!d+XFBB{5=KryUIK6JPJ$Sc;1Z0 zos&j$Lo&l#Q}?|5+q&IX_MDvtGXdw80IxHy<%Pr~#RExN67HKief{v9UwPpx!;8EA zHAvU1uRB*#Ow*%$wRVo&o48+^BFC9L?8xkIe5kc9eUnczY!3X?PgZ*Ex$J8f7u2n) zU)o-sEa|jh#DDR|e!BbVX2o^0WqFk$mzNrT;rHCNZi7RTo5UyU^~@--`{Mt$xTf2{ zFJMPJuva>~a^sgu%tN^H+8{v0N6bsY|K|G4w_bJM*;%{6;D%B;U;F)qvbwhlT})_V zp>y8X$*g_p%)E1++?$n`=;O3dF%GLgux`a1`CaAc{7YvRU`_wKS}#-1#s=TiWI3?r zcK62U1?c0{{(84T(r};7bIZa4eD$k;xf?wFHE@5yy5{gmvG^%6P(yvA8~f4jzQaS4 z$j;=-vR7`Q#pFSya=H?eeQHVQ?W(rbWdhs&N&b~luhIPrS2{5b;_+7d{de-atu`D= z9C{(pSImb7pXWM7g|1#9O-1C_r&0Zu%1KVG)r}3s($Zj(6sAfN_4xR4z|+$fZ~!oS z>49C_NXB1bAw*d;Is2%<-sj= zXH2~qfqmr(0}%?Q90~I4TjSbRd*PUz00YiZjW1VKGbd%!J~IK1{V1i?D20`|_*|Ug z+*DdwHryvZm0`EL%y0_jM%y-(oZL0xxC?VG*qx_kq`U_|w_uW<+_k=U_8?BY2+TMkUFV!OhOw-Zfd$apM z$M)_gSKz2##g|+zcak{5GQU1Juf5XwgVGn>fp8K}`bSF0e$(C3c{_dk{N4OEdIzK$ zeTX%Dgf~sWnsvP?f#LoyKB)t7D)yr)^-0L^(}OWzV;%S-YmMF$`Q%XJ(GhHUBt|x% zwXR33(qb{T=s)alYudF;b4_QSmA0CntC`Ys9Wc2)5KG$q(vNYdW-xB&$<+zDFhfC! z6dQvyouAk2`+D3!ZxChXi_I+EsDyy~Ub7y$>N~^QoAztZuWY#1hqepWuGTy3wM1kz zBi6U4mgU7HlW@_F2gh#x8uxKN!6Zg#@5!?wh+xIKVt7r8B^~H0G@Ay91x$AxzmMW8 zH31?)FB*OL{gU(su?|xEo4%@itl=~jMCpT=XK@pT32st$BjM>Tfts32zn{Ee+T`Q{X58G(iWVsr0h3J({K1cQ2rk9ca;BK1ujP`v_o3K zF9>eApLXDcFqW-g$iX2XFphs{{gx>5#93!q7oszpWu#J5|a&BH!Iy$32Dak?rYFJ=e-4b%FW$%=t42 zF|>{YAV*IaFx980X$?}gk!}qdflj0MG66b4ge=~b@za%)Tg7EU_J}do8wPI&&*$NL zd)Z?H_ZTq*GD05Xe(P@BZja}@?LKb@)-%V}cO2EDJ6s5*?&BypuMZwT*fS^ZqnFj} z9xVRs&h@7w#+r96ohDH^HN=S9?Lw|QCrHFGa~sQhdI|!}CL5Cgf|kmc_-K==*LKG4 zEx>iW@6q>z`NwhK?(*%kW&Px5C!&a(P?h#`?yM;vwDf0yX}ix0PkXO>sIgG>!np3; zCSdS;{J3JK?TqjGZot&6``1*Ca|gh_ZLXgIoT%OCv#%(25uGorxs9+6j1xmkLF2l*_KC3s(%;D(rz zG6gp2BU|QT&)9Jk>Bc6wJwaKP+9#9*mTDj=DpK$Zka_6ISelcb{6bn*J4a#L4)(G~fLL1uY z30}HDjp!D6tcBh7or%?eee00D5r-av%~7TBC_w8^>Y?zk5e%9S4HMZCm$bnBA?m4( zv{x?qLGcNBFA}J<(1B|$@gfFMGc3l6j1`!!5?4N6dKZqUG67UdjIDt05_7y%7y2nE z->-Vb9*!vws4wXpbe+FZaN9l;5;rs$j+YS3 z-1EE9ewp1FYj|!F(&HqQt6+R?TfX38B4n^bvA>&RPmfc(>QNl^*lhRn2DYu{7=0;r z92B!lNv!9mvdbryOt0QOnB+N7U#`xXzL~;lhJL0(I5uoj&5>z-;42WhGBPc$9y(Q>k(A zVwUix)P3w#w!5fvCncxnt26GDaqrF4qPwRzz27FX2i^c2=E+>Wrw)IvN9>i|XXKGV z#|D9Qm(P#l3Tq11ivr==pixKv}{cJJN1J@I*KU5#v=C4kdU8qVT zf=2Tu>e-DoBj0yYAGGqZ>+};Txk*E}Z&H#m0!(;ZoGc}$M^EZrS?IGFfV*?k2A1w5 z56oUQRXg%5sRpIk6&uS1gZ|&Eamf`1bpD%RKCT8>nj3gltd$$c}8&%Y<(Jn3TFC zL;vw=Jg-h<{+czt2c9$DpBR0Ht&>K;dqQJ+Lu{GH#jaZrci`riX%CZmqaimO(6u3d z&?`RM%y&${+*NUgK`AwV=X~iJSv64MJHhcHHMXZzHXwlTBA*K~H=4lp(XW6kpJM~X zHCrB~4u4BdXh2pEw^y*~8{JM7^{xj)S2c8$2vJSvQQhYPg5#va>uz-}_S^qQ*joVC zv1CocVrFJ$X117_nVDI($YN$@W@ct)Sxgo)%VJr)*YC~D&inq|{o{r@LJ{}o$;|4m zK2_aS$Gs~4^!N3uVQ)%flCyJxsMJY+h8e}x&0$CB@QaJ~*u(WVP+9WJ)YRnekd z1{ZvR9E7$G+l!`UGjf~b{7KQ2>5@)cuC%6Qi~jYyy9Q(Dt!xIS$|rlbn1iv{)a_JQ z{ApK1CkSgT zeI|HSwzE@Evoi=Dbgt$qsD01oMoPQ1V`FsA=RD@`t>4vAz`0#d4^bLBnkL*#=O04c z7uq{3j^3Yf_RV_~OF4>unDaJUpBeBhHH|YamP1Z#3)X&^+|+F*MQ$;ko}xhS-KyXR zD9IUIVG@38cVpvr{BZi|f6Cfp|8=}Y;D@bBvIauvnxo4@@xJX?$F1@xzI#`ye1nL- zNBMcQJK>~^yNZua`(w7GD(5u$V~ZX>c)du|r6*Y}qMk0u_u6L-C}$5<7NPHLL0mvi z*%Aq!8}K`F1aObq&Q*7#7q|P=9}j7-p)4JJzL(Lr zN_m6cPKq`!-MpJwlZpcur+0w2ub;iz^&dbIxhG$chB8)I_>VrANE*Y4c*Gx0u;L3% z77<<9GWG=>H{M^Ip2kNve~q@`CO!!xq3(OiMvXHwLr`SMkP<_MNqr%rIPgom2So}o zGeZ&`X(GT@aW<8G^@sCH#H+)e4|B)0c5WXV1=tn6vT;FtH|$ zr&~EK+yOh8Loek@|L@hAD!~`$YOdd}<{zFJ)@*Z8&kVNm@A@rwl3gu763AEIo~Hvec7GEh$A5yPWW1%cQ8U*cJy7vVM7U z2@N?|poYl2+;`S-$2^OZuOFGlHr>eb-peEv;-}D2wFz1|tgt%DDp9IVUgE+2d59=_ z`BpOjgZjIv(M8WMGoPvTa^JF6FPc%U#~h22#kv=e9oXM)85n+VBO~ttlg~SHAURVp z>MebEcsIUwJrIDAX6TYZ>JNO3)U?6i@U@??Zgg6S3DCngKM_YsiWWV4&7Bb7dus1A3Y*+8ZN=C zWLAIc_Uz%Vg`WlQO4<@MqNMLVXibno=%e|1d9%alagjZ%4&HYRehvB79NIewlEydt z1(C&Chj?|}K96lk! z(VCX$1r>3plhJ5$-?|Z!&#LX`0MQ9xPHFpaeNC#r=E>+PCRIIo|0;x)G`+U_?h4K# zPs{TY9m;gD*k48jM$4{9`2vBiFxk<@ufK&EYNceQipO=3D*~pfV{B<>Lz52<>`Knc z2Q|b?CH|-g_iB34<@*Y)*Lq>^75uRd(|jk>FC7j?hn+1&5p~>Vv`)~;GV%>v!#X>- zf$mfG+8DsbK?Yh{DH_MI1KstmSWp|8^Rj_(PbDO~hMXIj^#X1i++LiHUKd-kfkaqx zlP_YM-MhAgSMVFS&STbvDaPLnv73XuK;c@Vh1xjZkCuGUf=MZAzq&~5ZfWuWxJ6U$ zY?-pgex2{e9^P{0Gx-Ivk|+PUkvZP?WG=7OTh_^$bvfAW?Y-C8q#L%WJD9u(Z!iUQlnwB=Zk@8G<0+>--+w0Cqr4(Wd~9ZG?8@m?Wi`3yiqIx){&F*teF(9=xC!-$|981n%CGJ~ zP&z=wm!3DKRr0lkvKsIPbY5Ef<3+=iImG1^W|RahMuU_#^OUqZ^ts)r`dX;^>E0=p z_}qE8KUn0tPX>4!=XnS5m|PK>E{lP&=s`u6bSh>0vLBYI8XQu+Z0-yJ<6Si5>*W&{ zlSl;t5f+C@`7dOTA+5_rt?GR`{Mgi8%@kei&qXUd%G#V+Y?+>O!L8=Sts&foCwP|Y zI-pG25Fr(<>gfu6=haGNsG+SX-BmZL+j|Q$aiV(-Du^BvyVdNc)#%qLGu1E-)zA)d z{GqKeaGUVQK&ASwyVY8!)kX+dJdi&hmw}>E{PWMM)k-90JZ;e0=XR^j7;X4m+JE1( zk}-B%EQED@y*?4Z4yUeh~b8$0o!WJtdyW(Y*5GMzO$Kc>o(m1KykNv+(L6D zUZ8sy?s74bp^Kfin|%scTmL$zP4H;Dm&qPxv4^uAkBpaa(~-$Mbc!?d&myi4CK_iRfmgkcjL9JRyc zuk|I9j=_$O|2GW_33r0mI&n-1J~gL;r3B%4-_YR=bXu(x+N`T6J1&`invjVRPfe! zIp{%vr6MerPSAVDJ=;A!Tv9VB(>!FV7>QQeuF3=BL4DB68=)x4zi~ge0MvJ)d^KK`{_eMP%swMk1-xL#->V(6b8b%WC+l4LP2nvY=2*1vW{!}##sBabyj`@KtLWI*4({-HiQa)O#hTg%9c!KXAsR_2QYDR1)NlrE1kThM- zhBOdiwKK()Mij`<5+Wqe^GrgC3!t{5)c(ryn=qJQ9!ACjRcJlPfn5j`kn^ih4X^0V zO$b!O%NJp=kszqwk=>J1j`7Q8N3D{tD2+Nr9_2{>ZNUA*feS#wU{J`b`OuCtSevx! zmlA=kPn6ef}8F`Qc)P> z$`VbCtgiArJB;~CDGCH;ozETZj72NBWgZ4Z)Jk~JAp%Cah13s441y{Grq7f)BV(-| z3*!?YO2h2GT#OP9-NcwEE}Q4kt_?N-A=;R9gDn^aL=3{q27HBG=rvrS*EnH=43$3H zZfk{(^}NCn1jWM=%x#0L$r{WYsFT(Yg{=8DmIwM+L!J$T41|^8Ph6J6Ge`G>(zG{Y z0&q6`chf9sg8{Wa5Ut>oTziOs@e6}dgh1`xEIvuk9%838GI!>-g{As}; z8#YIP9QcD_*x_FCPq#kP}317fXR;6!GZGR?uAJsi-7>z>jmtCdNJ{1v; zF3ujWJ{ro^f82LHnyG#3ujHJsB!q12?<#UGhUnEF&ze%NHB~gO)M%hQ+?>=7@z&0C+3Ips=|!y-I#O zAoUAO?RJ-b&`R!O9qrMB#h0({?<4N-{=B&g(q}Sl$+$_2CAFsV=93yMRQ0>GYl*in zX7mH+p)1>AD~tlmQnc*3q(-j8pl*J+q!Ng5Vg63=YI=Z+3s2S1-OD~ zw5}T@b{{fV08AGM9*4Rvqx3Ow{aTLsj!s#y6>=l(WdlIg8ZtOm~=2L2PJV})>Jmn8|oMOaw>v-k|D%O|nxdPYpd&Ndx5yy9O7iGeSa;&81B8B0cjc@M>7`qPu zyrJ{dT$E@=7J53?X-2k@*;e8B0dF^1_jiD6X_?*Wnb$AtPcPNA9(DsQM!rgBk=-8& z^WA3p0%E`h?(b`pmN38N1N+-d18-?LsmkRH#- z>i7RskAp9^3~3Ae`8k>P$-|x5O`+sg>cigd0WA0PU;{k0f`;`sa(oN}xbH+Yk88ds=-5tgQ7A`dfv00K=9v4lXwBF}K@(H7St!YI&sWC=4znkLG4a0$jV@%vIS1U)yU zpC-%}XlZ_Yy8=Zdic^5CXqaRUsVe?N9Tnt%K8W6 zrF|fhxsL$7A=uKKQIi;bwdPuZUrd$7S0nB;Ob8xg>1m^37EN4j57s1qj zXnkey#rIRCB*W__g`bj zNjX@hGkvbvH;?5+$&b$8DVzpR8Sq~`rVsX?IW0xxmcg43;nIs3DLdDP)N|s8R#`aW zYz3lwtyL(gq8Z$-i-=rt3kt%4ND*oiReHSC?zG|)QhQyjEO~x@Y(-Ch6K(zy2a4plA5 z!tRyDq&P|vrl4iet+>+w(%Rb4t;A)17Y?CR_N{uV(WHug1deT9S1_zxVOKB?R=e3S zd0to6W#DbVEe8mjqP|$k;O0z9G#M46-z?nWLm?F&^K-rp?cvgsh=&V|=55h2;N1!mU6n%Xj$UMKMKL|NCE7pr-w?;!*kE5RYX5D^UL*R*V=O z1P>oT9EUa}-dcN;c}z^+tlB1F*HVywSVI4t9-npskW=-w=|C9nR@jjU!hj|JO^WQzU${L2cl$3lFU%WYp2v|2$!fHluoaEKnco|Xs%cv zQ|Sn1N3u|?Nam6aUn3gvM4SUA60D2xf&~Y|XO;cM>Du$rlN!!ZqF3XDiu@%D!8PBp z+#1brk{6rO&WZyO@JK4Ai~)3 zkgTPucFWXv6xC_{G(qb_ij)JmR+x(r0_obt(;>syv!LkFZH7=LbcAR-8J5G~v4?QN z*tPwZVwE^`s+6c91IqPzPLOP{X#CmHRXgaH*#!Q>RmoyBzp4?m9>N%2Dw@!##zYdF zSj#yYL<}GXS(|1#ELpfdK>e#;6*4Z%A4N?-m4t=%Et-YOV=yY+uCv6;Og+8zC65}V z!x9LE0!bTr@-6N$7h-%Ys$Je8hWO2}JK6>=NsB8F3*Ah{Gu-WTno@N#7!q5BKsvUb@bA1cs zf@CT;jOk)7PgOaX9v{LutVUPA4Y_z<4(jHw_=5P{Ea9RzJW#<<&GDV?e#3iIJvqr0 zTC1=_G@AJNB4N#4%=^teq}r)4fgj=NAOmjRH!IO*Bd2xt1C+W>uqGo9GPtja3l<(vE3W$V~nxP`fdUH@EjUfa~$&`#@{3&px zP%ZX}5S}U1;6;`8Ny}A+&mv4cDkPQS;Z&Kuu;uV;-SMJioC$fBBJ))B;Lu%^lY3O* zCX%d`>M6z0q3K|h#k3;xB3>$`{AkRtMa0q3kFd%L;#HZWEcVl9zVzGXMf~Z%Xw+9k z##O{3JgM>`)+Qy(jCXzjVp?g2coDdjpv}Lk=0R2qMOkN(45Vf z!B?d9BVwzmrkUhLh$|0!V1+btKY=y+(*Sw`w+hYd=r0QB)D884H0F$j)LfWD#Z7D4|Ll?Wk(oN%#@A`M>kUl!=r0W64z#rR{vbZ123 zGtvEO^qPL5JAfH+1s9`0j)MLEh|)xOIV3os+EK z!VNAAdk=xM30MlUCLs2X`mP`Sio8Kb;cZAtI536X5u?aVW1YuSN24&Nm z-DI7{$1-^tjqkp4mui!T5X!_7AC?8A^Q9!cwR4ox0r=bjp@jo{>Dhr$^cT+6;Ju=C zB^^O~40|#q^}h6iNF$6{9LbFhgQDxUU`!-5&JmlMk3DL?((l((h@Uskz#8~6V{Yp3 zlhPnps291(Xu5Miz(lx&0y4@pv?nZY>8E%TQ?5wXM2kt3i_w&Wd2C)u6{@839jeY| z{*Pu2)Ua%fsr*{2ewIC^{Gdp7Jn`>8B-bdSveVw{A%bwBI9~vJeRB#6p4b0^--XzT zuSLTbEll8?vP-tV7#EPH*>X&D@^}u({Z$Ft7_4*=9W;0s0dF>8LV&XH(W zH)P9k6b}U<4g|XN*N)Qw5!~-*KX8Yx9L89%<6KXrfr z4T&P}F+TesH*8iMO!5};2)%NwyyxulPU$CtfSfmVuF!bG9D!cb(onKYv_#ek5ddH4s9j?`jY-Z3oh59g-kK4IV_2Ynnb&+DiV>Z z1G^OawgBISu#l( z-BPc@xkB)jC0H3eBegP;##XQijwTmak}JE+)4CuCRia)4rP%cU>oc$WfF(|Y2%0Rv zp?{Ex!lsonbZF+k;sT{j#~L4#4%F$UhM`rBDr_XiQNO@xN zsB5~BXgb0*vMz&_SG%yLARXH&Xz|dd)rU-J!pzMtQ?C!XRsiwr4j2a97JC_-QcHrH zs@Ywig>h7(<|mNY#yjqYS(!EN!&QLqJr;jj`@FhbvwCX`U+Vw|gIR5=T1g z=p%9e(m+Cj5XMEc=NBf@N`jD&J3?N*itGZ!D>BtDXsuSZo}Jqm#zS;AtCtGng5ea3 zL80oqhw$>0k~4r74=KYGu?k8R{1s|uLE#P+`@J%Qc!e-n%F@9wk-yY(QMVLU7Y|Rm1J1vvfZ3KbXiuo`m3uNEEtY(WJ zBp-gi1L3*E-hgi)7nmb{1EPFvtlj`#dQbgj+P)bk00<=TcObs-lpi;crnLMX*8> zEeCYVzKjS*{|5byy@yDjSP6IxA|2uD@y1sH*bhAmK;LCKhx@O196M-#ypSEG+C$7k z`$>#9)X#+H2=MH`q>%dm9)$YaF~PU|${R?^jN#`b_v2umPc^W=9m5i;=`!TVZ4_X@ zzqu=;iH35)xPhF~@;ajXLu}Pg<&PvL{6c_i<&)Y^CKw51-k_0xYVr2}9T@u`C3C+; z0Fz!)d0aEs0KozxZ~ait=&^r>aJ}Ds2-J#^jsCO-x6RBSIX95Fj)2zKLo6`?$WerO zM?yGiOi*ywwCCJI>|@_UG@oJw0Jg3AJX*T-ayC9c)``(AmgZbX2;H}0Do{SMWjQp=Fi&OZTtdQDet9@-Z2mp;O??U+1nDI6gp>v>q@@M!i+qYFKf>)K% z?7ROn*cbUy?(?|;tB?iT;Dsp_Ma+NESqtA>#gAG2h_J#UQrj}&;{g()R)g#Q3#_gn78%BX{JO}{&5 zZ0-(aLcAFQMCz5c!_xuK<^>0=S}cgrvyT$avN+{gh0Nr+rX4df-7-9+Q~LnK;7Z#W zd2d_+>&MJ!L9&i~xToXKXCN;$AyDBQGdj=NTK&PfruA?nxuzfCipmzeH6aDM@ny+j zsn;#|4P3wVNN<|_DB0%tVZv~IyrA!-5;vx5kPa??ldu@vq*W?Ubueu#yWxFjR_)@- z>@u&?s^1U2gj-uxB%vJh@U@+U(^=aDCNF?yPT_sPOxN-d%02|G9r&g_a~;%?94P~)bCvoh-DCbU=L zwgPogls_O^nRuLs5Ok%boTEwuy8siV3o#3HMj2_miF)^-I*-rGO3TkmTEwDLk*AV| z7v`qcwaju>recjFh1a+DNL7-DN&a-M?hIZwOG1r*BnKRh^0V$`xv_^6!|?gXR)+nZ zNvg6O8KoZ;b*L!yl;p+d+z&9eTWhGozz+|cf2tl9F%<=W|L|@TJ1l~q65@^z)a3<- z{A0x-**z)ePC1`ttTPKDppXBQB$>>qe5cA z@drGF0UYjTp*%~Fxs>o4gKytDlV+iU*g>wu#tBd$qw{ES{5ijY_Ex#qQt%)|@4U@w zvjq&qVLU$H?P9P8a4kyqLhTu1H5B75lkQ%^ zux`~*l~-z&WGN=C1UW0W&2k+UDQc5S;Jj_zG9=HD8eFohyp#dbMR7zxy2#=Q#fErT zWH2FxHsUk2wQ|>-mzM=aOSuu5@MRHWLP{eE)#74is=qk1Ia&=~Ix2b?9E+`IZo7x^uU@*M?PU>{?I zbR!lBUlp8>#~;lh4uT2w|&kxE%y+-!yc&u z4R4^!9`jShyXE*6A*LNC8V@lF<<@|$GBUN`8Kr0W-+)IX?b&Ur^{F^pZs+{ zAeD{svx4CkWT-2#0#Yy-!00bc)VPsU{{HBR3(2x0{dt6$kQCpdyucJDeiQc@pFv*g z)?CnVkW`?EHfFUe9Jj!aaRtl`{?}*84DCP_lskGaD5t>y*wBuOBW__8I*@6{x5#|l zN{>5+j=Q*Q7}#064e6jMwi#JaUQC=t3$q|x@COy@xYaJU^T9oZ3+@>8Sb2g{oC^NI zVr+rqPv^if>NM6E7H$l|zS#mmIp+mJ(%UL?!0`s8Bk$B9C z@Gd7HE216S6uC5hW{rdw(0?0OC~ z+%c-@8($=H`G{ zQ}TJ=2pRDKeXVVTY}g{5GGBV`gPx`Ky#7@VO_Z9JLN=AJU>^qDB)g5H!l`@*4|dTf zmj$$#kMgFd(mH7Qtmuy-`P1Sgsq8*nSj+&<$6!ur%GtVXZivKZK4XhHh;M9dsMTh% zf8G@4V}*aKPwttYCi-iJq>*v0kC>ew1vs*5vM0$b9&IhPj|6}r_82TCZP8JeEhb)Q zh!WLKh(|XcC*PDJCjVvS+gem+dUfRtT|+bghh)&5Bd5NXz-$^FSF>nDg2XW1%_^aR zE`dZ2MbmUwxNfYcPVzJft)+2nHg?1B-I(^gzn=siu=}{y!D=oKgMxv-Sj)Udht>ff z+#vR+ZoR)4U>XDjLo0QLYLbNd7TaG8a5hI?z(*!c;~VitX3ZQ2CTR3(AtGQ{Nm6Q0 zl1Koi!5dZJm=YTw)5Tim07#}tkb?;`7Md=@5E$h*@lcwkdVBeFkwubxU>I6$NOTR1 zbjm#AftPeF*MoIrnRx#pRK%O*uJYJPI;9Kr-8|mzKKb#9L}Hi9IMv#Um9H=TcSq4jK+8n*44L$^4|?o*$V)osa9*a9{m z0BZOSWOvV*0GF?@O}ZX2o?@PMd4YcW{G+$WWoT4^F$?}{9(=MfKMZ5Q9FQnAEe*%r z+)AnKn-LpRN>~IFwIM%`ANKc-kIQ|2c0+pWbYM>;io-+iwr2I>#Bl1?zVP-q7_b1F za_;)y<(#PU_b$~V#Rn0Z+KS?rMkwGyEEgMN&vnZrm%35h^v!I}cG*%iD>mA~Py6)u z+)g@)#dJ0Z102YdI9NwS!E4H9U7Q~V-NbDV@?U?U*Y8iAUk>g>IX@|D)M6YL9m6H1 zkYfd64y7x^MesYm1s1NXVq$gD);k3s8O>(BOu_R~#$$Nq1yj5@W`Q#JV`%S)RgBUCVk%$H()*9rBD8u^3|ViX?19MAP%3;dR1vvV19lX zM_U3ud9$1&?)^p(p`k@Waw`U!2P4R&-l42F0lf{xT5hH{HeUf{{ZXDMT&3CyT%l&X z{R`7r-AI;CAg9sxxk?z5aWMpYz09@F4`g3joJ7x-Vg}SOu`7hDn z#OKO>iqVxw+6I=7Ycc6_wZI*Yoaqw$dc2>c{cQ3B2V@Cgl0^Aw!?l$#AU9(AW z>h`?yS>pM*4&~SGz-^2>SmWmdf%{IdSn^ntj=EovnrkyN?+RGz1bf7mJB^2o^W_UO6%N=I;n5aYS!Xit= z*Y&p9RQ4+WrE=64$!&5&?n%zB*kF7Ug5%PB3k7`g>ozH&8SM!H5bJ93wQ>D4IzJ_7 zq2=&6tiCnUbk?1GCcD>bbj@H^gXeiZNy_i{>yftfaNondExG#rWxv~xX&*haT z+J`DYfKpQ}fWJR=H~*?u_RY*#<-acY=-GbjK_)n$S+Aafg9Md*(qYt%;4cWmVbMb*#^o#xQpZTT!`$Q8QF(aytD1OoxX8w5l=eXa*_-kO zP9$aeR_HZvHB&Cy>WVAWB`MCk``E zX1?jSAB3h!_mW&8d<2blP0CJPbjxEWN!VK5@)+GG>}*y+^BhKWK8lvnlj}FZ4$poP zaM^8rsBTX1@Q}sC7&afr>vp%`|8(q?JI}=gP~Ytp22eEW-;S9B>XND&8QcC!dpG?czKy zX{X451E8>7MI=Fqxj$!~jX&?-@wHpO(KS$r($KX1>cerQ8vtc4@;=}teCeJHF`Q%`BnuB3qV_NsE*aMeSPf*!1(Fdm4;f*G&lNOeA7p%b9E zF(cB9V#l8iJ4=rQYHM)8^$3fE62XbnAd$q{<>}r|CS!>SsAP^_ny~W>TUQ-3q7xHL zV$m^IIvQmuyxIyZ5fRXMKMjcVNCd5%UAJYmm~Nho2WvZl2xnwKe6soP2GA=Oqyup3tnhCLV6g zxM^bQPTIHQT=X9Kca9)sW8*sNU2cbcx@KASIm4%?yXy;si03zVrK5}npGueh zEVJ#UW)UpMdSQMVf4DGzHn;T}O!emS+snd*zs=lIph$&RP2e(l@WyRg-KGe%=#OPS^iB{2u2#fCRQ(##T4r15B?bOo}r4}${P3_?iKnoFmv?pSwfFAoosnO0U6 z8EoL!`n4;UYV)0xyR$9DVzkd_fGx1+rKI9OhS|UqTqmQQHzgC3OHvdXw@zJg&M7jq zLrQWK7HBxnan(@ezKASP)yY{DuG?i;;$UZ(^UHnSdN$e@Sy5Q(2E2EBqo6mR3N8?# zFcln_Gt0;{G%3&37A{I=eRU#BpE=MMazccGMGZSWAG zQX4>CVS&6SKo^P%GzLxNI1S|apqP7Pu7iXwCnDes0S`fy{a3QY{$hRz^Z1Ur!t~By!1iB!w_B@v&fvF0P3AByXTm?4PzqnM(2!Abfp!picF~knU7dCz-2>F*z6$ z3?kBvC(GYw^NSn@<(g_mOgn^?+n`A@Fzx%o{q*bAXGx{T?5H+`#lDtKE#r*_!<8_T zpxxsQc@Y0*$h4-}OlWozGPR2@FBM7a^9qgwl-VRRO}ui`I{_c|G}@L6;NSCNQ6X9_ zDbVDvIIiJE_|hbkEQT0v~?nR)hYRY1M> z&a_iz-8T7Mz9oKToBzF!XEUV9$f%+8bzNAnqtcDqKo!OG!b%WWRrs z;2}iQSqxuZnm7v1FSz@zvX_$7)lRdxxJ+3retBzEt`v@?AQ>yYpXPb+`_kYPIl5>- zHv8ouT_Hu1TCNetin0*}<)Dm74*Z!fYsNWB!#RrpxnovnR4cX4r}%42VHs0->g-cf zL3J0^&w&7kvdxHPP7K$*z`&N~tuhWe!@)ee=mUHSp9he3LXB$Vg!=6k+y41E>kIJF zr+E8UHu~avkCIbW;LrN?M3b6C$-3v zaim>KHsKZ~aSvmfW;Zqynod^|q;6}FCR2fmLud(N6@)n03PH+CHUsEu^U?sRK2nQG)^rzh4W8yq1q}r&8|C*vcHJ z$zJR*8)|YG?Xm>53{P!++OJ42x#vN_uG%Kb22cGI)}mrcvj}@c#x677QnhdJ<8_c< z>!{py0pVb;T3Oz0`|RyTS>8S@S5(RONeJEXoKvH5LedeSw^P`g%2h1#lxB%&?s;i*MDiJ9U>GEnoYqQKN?oF_m;D2~?Fs$Sf(j(qw>Vy!rG zQqcIELV|3={j_acl^sMv8`c~`ya@OFy-QNXq{IhVU%q>2Qj$IH*({9*7@klXnsW|H zUUKi)m@iHwJZC7-PFksn~c$r5+a>6#*~T-jv!mim@%Zqpj{!gX$~Ye zMdN6J+~hjQn+75E+UjEX`Ym@{zj#02u#We0H=PdJIhO0aR_{0DKW}_FempCsfEykJ zAbb5&AKt{x#nr)1)6SOB%GJ#7uiIXAR6Wc96B^KykJxy^cnMD!+JZ6?C2~GYjnvCe zh=i_$BID1ku-+Z_Z;qo!c5)vegls(WtzuFxjxb)A;_`$~cSnMENWtg9LXcTCNR}!V z5?XF;?%>1OmkDj^*$t)nU-@V~$u&0jL$j0g<*EDP9!KY&a5( zM^@y$Yrg9ESc=Fm3^fp3Mwiv}Ih*9cwQVq&rp`f2YyU$R|HR=^H$V?1Ko`M3bp2fg z_CM$$u_>D2rGUyuflO#(-+Bga_r{s?z%0?iB!Q@HmZbIY-TlaIN!LZAD?YsB zc-T!wzty;2E%D+B90u7NR?X5-&QXv-saEplhh%m2va^wn@zn6}TY+=82a=DHaJhcn zb{ef-fM7r%yCpbGpT0TS=(3nB*xpwvv?>x4|yB>!^h9lTX@ z5MI4C9QKLO6>&~>Z}^*Ui`k2

zWB_gxCf@BcAMi<=D%w*kz_gafj>|BN~8-Rz9b zoB^{kf3mx2%}x6~F0AhDQo(p}-RAH_HZZoP0zq}XXmQf!J0J?Iy0hFdW~PlRnM;#t z+Nx}UvItgLl&_P~sJ`3WAKvnB;J@Fl0~59o!qHqft^##Rw?*iVWTuw)Kb}%gC@!9q zq1RAQe$2M{3R->ce#FkaBCpSbqtMirQfc!1q`X=EDVZK+%NxxZHgKP^r+J-Jdtf(J zKCMBEN9r{*JpUDc$5bPHhgn)>m^IiIQl&Dl@QsB7X8(5HElY9JMBS;YHjFZ?d?x}H zrnm#$CK#!9{XpD_LU9kALX&_^Lvik6WLdpVsozOuDW92CtsFw?A!zK*oi5LR z9p-Z|EI>etmQb)60oO%Ri&yR{&6PaYZ>DUMEcUB^>nQH#{H8Eje`rcgH!Op`F2vGO zaKW0U-4ZiPAO9QnC-JSbq+)s@CIwaIj|DP0N$f1ozI27}oUw7)2!`xQk>wOLrs;6E zbW6gd+Hwp^QPgDU{je`e!~Fj1)*rQi6}XnO_BW0$r!o}^*1Ug4JN;6aHNp6@ zJj*;tQR#9k5{9Q$VV%ES5tLD@N`UrT{x?Ka4t56`PdNFh9ay}+`(&b6_3dWW#y1{{ff{1{CFabRlpg{Lu-4tNt1-L%#jjU`LtR2lPF4b)uQPt4+JH0rm z?WU2a=+PhtI$8RP=!I!PIIMn_Bykkc*E)qrCCZ^@P#Aq%NRiKue*t<3<@5fmM^kmp za#EZN;87M6lz>EnMZ+2)vA$vFLo)!q+1Md(@J#-INZ%^@}p7@JO7sq9L-lx- zc8eMns3e}H-y2(X7h#8e50IeA=+?5vNMyz0!wj$16N~MpDN&3{*)l0!@8pkZJ@j~r zk0eLpEfL5TcCs$Bw88FSEN@u934;EFSm8#Co`W`QNn%tof}! z);3JGY`3gGZZa4Uf0Emywg$fXt&d;s@k64vW}YAMA3QocIQ}B;J8G5w|wQ z6XLuX2sm(<=_gaaqAiw9nrWHj5UVM+GrU;{^$GTu>#c+pnVMn5$!qLiXFE3K9oqLV z`C8?I$~<^MODh+jtF|m52Pm)Xoh^P>hi$?Z3+U$&eMOlSw2Oyl6V$QCrI`Ue!rnh? ze!5-ZsuGcvC!A;1T<~;goL*22Q~Cqy5wm=;tj@o-o-_EBDCc2*XRCtjQvsIvr%M!Ca)f{wXJj?iNz?$0l~P{#L+^0>}k| z=&N8$^Y3mTSj>49LS;XuwZ4Tl_Zbs$#A_4^NJPg8Ex%?TB4#f$+slSk2lO=rjvB+9 zlbyXRwW2vx!{Sc@dz)FJR@}hwD+m*qNNVROWKGgR?DP`?tjPA;3Zf4qGtHQ5p6?TJ zi)?5YSTJ;hPISBOIfhnKpeK;V2cglE^z-lr2lKH8$?>h`gGWiwDVLk1S)o4w`+K9V zhZdQeN8=@bvARvl>MZ%u<+XefFyK`++>$WQ;^pBccezLY+jEnkU+33dZPJ0Hip%8L zawvh%!Cdn>ftUAwNokJH(b{p2Z@$aqa;mzk3Zp?k~N&J?2tN&{(ZK^HG zRZ-%H&4L>zet29|dbmsMW$fy<{`1k3Cnet8P+9$b0_V?j#mcXaIIo;3oc_(%%HKO! zYxmYexBdsGGPkO}vf_IwlV_ov-tw~mc;Ir$2L5GDY*HBu8nF`pjXS;_vJdcPWD;Qp z4Ucm$DA}3Bt_@l`aVoHddI~t^2I_XeKm$;Wfg#!%I2fFoR}vpmS&&*B3u>LA8&Ij{ z?eGJr^d4{=nH#1XNH;JZ0~!HojRs@p{xx2C?+KX2j#&IDK4qZ zNd-?=qK5(cfs!!&4UEiID4NlZmqa%QeV!CyPDwt@9HdDUbnWO9F9_{P#Zc{N(=g~J zpie{~OqfyvH34N>0$o4)s4+sndO1`-`WQ005$OGIgb@mLSbO8>2B3Fz5eED~F#ub) z7u_uMz9Yh{kM-z|LG&ciO+oK!AxzA;wBkyZ^FRjxu0x5=*k3i-^gIHC~36;4kxb_*ry`QL;s zFaOdkB5q2h--?RgxCH7XYUFLA+5{Hr)+fK@MPtYQ88=4wg=v4$yW;3KVJ-z}rBg2% z?S@dPqxk42wl0ZU#Bhi1%hvPW{VTgwy2q%{>S79PpR#c+ssuVjy%bxJIM)mGKk)p~ z<19u_xU^1M5(cOra8axBE_v4)tz_@9?PXhMe!aO%ZEF=IJiWro|6oJv@KUhNnnOE+*UfLn jfjqbOIz9Tb`6svZ(5|#&7?TaDy+>urtk=8vHqHJ3!F#

a`TFi4E`Tr z5aeL^!05otsKme|$jB_n`2PrlJkTGkj9>ur0thfN0o};P&cVsW4OFmIfPsmTnVE@& z8RTl9Y%Nfpfklv2NYT)dO*k--U8zvSsBz*#4rQl}2StM}eo!$^Dr(~75)+q@lu}hw z*U;25F*P%{u(Wb^admU|@bn4}2@MO6h>S{3Nli=7$jmA(DJ?6nsH|#kX>Duo=uqec!9r-=(U9^_Ou4*DRPRCJL`OvU7(>PL{*z&<0+V@+iF4DK<6ziu(` oFf#%pk6Dnxp5d{^qOQMd{&WB-hEY5k2BT?UGz|>)H1Pi>04m~TqyPW_ literal 0 HcmV?d00001 diff --git a/backend/tests/fixtures/attributed_video.mp4 b/backend/tests/fixtures/attributed_video.mp4 new file mode 100644 index 0000000000000000000000000000000000000000..04b8b731d8c8a81a31fe6fdf0796b73e40b067c8 GIT binary patch literal 5646 zcmZuV2{=^!*H=Q8w7f{xv1BRB3{uM2BSlE|^_scNWX4Q0V^`UVD1>AS$r2$&k+r0R zv`F^sk$91%&G!99Z{NTF`+etm&OPUz?RU=aoM)ap2ZA7!JHww!q*KTcv(bFfh8Wn0fjz=>Q6%487LJrT2)a6 zt)dEq1O|h8NLks}*H_7hh({;~ho7!(b0 z$Eu?6h#QH5^Tc8fp%0-^uoq17rz2RjpXwpB9|og}A|qG=;)kMpy8-S1ic0qfuHflH z!(%Z@Xy61MC^FFx!MiXcVt|JW4fgUtuo!g|jzFW3VHecM)Ib?{w_EQnF=$& zI}ThCg9a15KnP$+gPAt&G?S;JEj^>UcyMcRUmIYq0Dym9o)EQRR>q=ww>ucwX!uIPVZ;2KYUz!VDI_(o#r>EPTvA30De}UY#mZK?m(iq{5 zSM6l)X-0Ri5^|?MIWJqb^Ixy!+VxkYmg~eyK)`~){s(_qIkwD|^$+e!ORFD>-?c!K z@MYJRXp3obI@}=>6yI?yXxHVs^LF^MnFPc7M|s(?{?FL&x(Li@b4yanHImCJzV0l0 z6cR+fCRi#R-B)q7C)l$=iS-ia4teyeRm=;q4u9u}_Idpe_IA#cuHYEGmNE_p!6toU zPJ?&(xSMBNyk!cJxvg8aEZHk^9Xe0fXSL1DP5N3T@ppAmj%obL3*(5Wb1CJMgSG0- z2{e5pr8@1QINbFIG3DvilDYTvyJ?+yEJchfVM3$J`}5De$$LRs4h>BVJgea_^fs<` zX{-17$-x6#q@!<4Ep}tF+ce<(%im#RzV0sD_Mqa!pQ|KWGx^0vI`2O+lE5BA=4mZV9{?#KT<4&Uko=AM~+F&k}q5DwB@|y!CPZU+&AFL zoA7CBjf?1q)3|sxNGZeDxYXm6!ZoXK$S=+`t4=6+{vhf-%RqZdk;ZFW&X#L`5A%Pi zZfcmlpwEswaArVgK%q)-yT#9I<)W#fY({*EjhOxiM&l=~(i#*c16*&ftHhQPM*ZHZpWl62 zDsShN3?2)9SEW5W>ygc7^eB;1tDkJ5mq@B2$1ceKaCpq|A*0MD>Wm%Ft4ZUDe20)k zWcfiPZ|HI)RB++wF~-iu$~;FTJeTX8@Zom}6|#G1DYDC-Vh{V4jt6CqG>-*TB$~ZG z@3Z?{TaTCDl3l%d-!}=C@%C0qNNydlYXILbSue3{Hs| z?xj=N?hf5os?yrl6OropzWXh^dS+qAvl?8JRxFU%&82V|vnU&+6;3W;@zN% z#Z7lvA!+q+Ugg;Mi8mDv(;qV~p4t2*rAL_{{P;#?|G~%SR4QW$?QXm&s!8T=yNG=| z6LAfR-dw^7ZFN<=uTCG7@-8p8&czs334e1a4{8-&8CL6IS*?Mea((TN*d<&0QS7e; z1Ao=pUX7%*JEPZ+?AZCPlxxDG_>Kv8HlZ|EHM!U17QuYKt^ zp||~6r>!1N7A4cFrPW)o&0SgfG%^zXFhqw}yg;e8iu9sa(X@$(R2*UrIuw86w7g?> zw9aS3bb8sy;^7GwS?YM`)WVY3wl1fku3V>}fvqb8J^hmx1PXQ5rnE_^u}PQgi1X_6 zNrjQHO<2Z@euWIjEzwWkiSqUzQ*&iWkE80@+MvF{Dzh(TEzgS@3|N}rvaQc&=txVz zb0vwL^E=D6&n`N8?|PeHD|qA19h3Q4M%P8R?NK7W(UH#=Ws~gd&uK8$7Yj)ZlA~eP zS@}yu0h_%s4cZqCEce|b7@NlN(Uc(o632a5qcWE7wX)1gq?e(|pR?hbR?|+%e&F|l-_fguy zO7?NkjF9D%kGS#CBV9(tAG@qwSw&`?;R|?5s1b!mzMF}Ty6du%1@+@P4q)Y_4i4LM zPs;pYG@Eu#_Xn)7e%>~kr`U0VZm7LbvF1A*AQWYDX{ur0NV>ItXX^0RwIo^YHfaB9 zL*?AVc-A1oxe>hvzQI&wdx0?T%-op!fwpgoB5#F5E0{g|kBZzaIv48f%D1~UI8c|x z-;F%sJyg7GX@B1~KTzq?ZM%U}S}VtorQfWXKFPJ6&vIxk-dObWXVDp|NAaO1-emD9 z%_ul_v&vbAeY{yb#~$6p-0XMqVdPjO)5cUszO4nbPkYafGv2MlW7mbl_hn$a&R%ml z>>+!tKgkNc&Pi5B| zp8nAb*`u*4FO9kC9m>6n*M8c}$yG*PC^1{Hynj0@B1)^Zvi+;n@%V0H#l&nqaZz|C z2cOj1LceiH##YGqwQ#$%Y5JTvEZ@= z$67{}#sM;&>L4u;_l2;ZljZ9 z^3l7m-SO94O9T-Vs14@vwGn!W8g7Sy+H$vHCQ*JAn5>tUL|Pgs?0&bCs% zsXIE!`*97+&IC z$<#MnT*plcEkb*x*)9~@X$rhXnVXt$H;l$Rex+UzoRx`D-h)RhPhOsGtcxzg1y@QhgYRsC zAH4F?wAdXwUx2InYxT8^57#WmeV^w!n8R^=UXlFe)r>Mvbmsj2=FFwM#!3Dv3N-dM zR300J*?u3_^nQ%SEg0Y_c=l$W*4if?h6%fquPdAQ4g-U)ld5;~p0W+ zzwQqeG3wVV2#YyxO#L)|r7K3*-_G{IxiblOskxz7AB!iEtqL~r-NV9YpL}f$j2W;o?WaHtHKv2stN^MUnr~v za9?BNo#E@7W*NK1VZ%{f|K-`6JC1u?p;7_LF^jZh4sZHy9(rzLOt@qLDl@mL@ z_co4*V*)Izcw+rjOM0JcUR`&8)cixu?2LcYV5515uKeV)w^>o4n$P!`J&6iI3k~F^ zFxc*FvhwDCgpJkXDJ-=dKrus{oW?Fbd+t6EZs1FV6Qp<7$fzTt)MyP zfXVPS|ApJet}G8qi~G*bu%vzo!7Ulxv{ZOFC9F+Zy**#ivDM~4)!s|Tb!yA+~z329XEySkK zE*2&~A~`VBR9@-!rPMKK>xMF~}%6~ocSyuI+PEcJzmSDZouA+IL+!ho2dCt`z zPBx8XWwsypHDfrO$28mNPnrc|lDE#@)_QtUr7gT-R_E%6A+yyZ)i%jap(8@E0gYXT z)xH)12KOrRFgDM%VB;f2-aKhmlI3@$OH`?ET`xy`IW{I3C>mDqxbqRHu2?s<>Cfuo z$${6Y5QlSV`|FaUf%iE=uU`L|df-&)Z6t9`s_mI`jjB1e<}qLTS4F$9<%q@zTGk z?Idqv4bl0=c*H5A>d}3|a>nzhXE~V_rZe}`i~gD`ieO0s~1D>vtE9Mjt+y zvNH}Z_)U&JS5R#k6rsgqGkoHGr1UD&1a(~U!j%pmVzr~(+|s2_j@V6iR907a zIjStae2H6Z&;-89Z$*$c0)<%hEh6Wcef zKpaY{p5`k8XW=Jz6Q6H&FFUVTe)_PnBTuab%aSu-We_{~C}=m=E>j(58U2>J&_Olu@1W`-7a zZkpd$hieaeDa<&FXu*a4Ap@&^H_g{p>sE|jEDU;vKTK;Ff}qV|WD3Oxq)HA<@AKX#+q6X520RMsH~bU{ zcoJ=c0}tkM=r_-#*34zmQP>MlLYOuzZ@{|19WZ=84IOow3mXz;4)Zfnj6ve4s97W0>&U zAC3EuEz<}L1ZdKRp64IB;mh`qTm;0xF%D1b7785a$Rn>U^T18Pr3m~;aq{u~G}L^PO6W%lehuqYCL zO`TzZ7Yvr1kQUfuGM$-0p(9?*MI;U7Mj?Wkp_0H#4eaA!28!%WXP~?&43ry!BEU2s z1T31EBY-Nr~;Ve_HVs8>_q{4Hs-041$fvWP@zBlRnQm> zMX>(F><5}X1q=$aaM3?4zy!I23Bss@2?93(sTJVJ%q}bZX~bNc(!t&o@q@`!60)%} drT*t=X&iN9GDQCLk<*t4t5=K$N>5M!e*l)m0+9d! literal 0 HcmV?d00001 diff --git a/backend/tests/fixtures/sample_document.docx b/backend/tests/fixtures/sample_document.docx index 101ac248bf32b8e2d4eff9c6b26802e56a43659e..1625c8682fb0958488800ad721d984b5fc7fcce1 100644 GIT binary patch delta 302 zcmZ28pJ~l}Cf)#VW)=|!1_llWB|DRiyoVW?f%N9fjB*gh7pAEYMh}Y^6If~~+aCyH z31=omN`QN=K3HmlRh}uBp;+*m8O-1>(TAvUES&(+b*20p8%S#M*}AJ>dTISPu+n)A zA`qop8)rZ?2ef`<1uJpswgc0x-DY5VU$+gI{@ZO0rmcG*{Hh*zF#kx86_{r1bp_L& qy%711UVAYARIe?V=IV0*)4qKW^ZWWd!Tei&&R|-hAEGXy-vTagAEc@Hx(1L@6|8Ra01FHBP*j2;#-Ca}~}wm%TY z63$GBlmPc!eX!I9t2|RML$TmBGnm0&q7PBySULfs>q_}IHjvchvvpU&^wRooV5RdK zL?BAHHqL-(4ru+z3RdFMZ3m`XyUoD#zHS>Z{kPj1Ok4Lr_*FgbVE&OFD=^L2>k6hl qdm-{2z4l=Ksa{(!&DG}srhWS$=J)k^g88@loWZn0KSW(ZzYhTPUv;wp diff --git a/backend/tests/fixtures/sample_document.pptx b/backend/tests/fixtures/sample_document.pptx index c9228b7583b815d5d8497456a7d957668c34d33f..676badf102fc7e7cef76701f19440dd598bfdc8a 100644 GIT binary patch delta 894 zcmbQ$!8EIbi8sKTnMH(wfq{cT$bUbg4|?&%=AYuGXuj}ZUzRW$${!3lLP9wH~R<8 zg@wX@utz_J)qn*y7evWGg1JA27ZPIgW91M{?KfEWqI7?7xNrWF z+X+@Oxgq}(Sf;e#Iz*{|nH|{A6UxlM^vNH11vFnzu<0!(XFLF5nCOHZCs1u^3?l9+8ZM69M7q3#(} zOtl6f)?WircMeI+q81`nUW-uo5GtlnhY;(kgQz=+BxYO>5i6>Pse`y*x&h{X>B(&k zwqQRTYX}C@GK~&kI=;~xOs{Q(gwe0YP%uBJ31Zg5CWu+Tn;>Cj+YHgy&{UBJvMPnsIHAq5P2cr6;&U}Pwo*J&n7Yq;b0&R`;Kh-3| z!NS0hF}cxAd2&Jx>*ixdLI|aN<`DgN%;j0(w#C(QO}4g$*!JJn4`ExgBSdwelM+I3 zLmk&-33G_*a<_K~vuArlRA>5Zg$e!>1i8ulnCXvRW(J0{+zbp#lLOU7CI{4UZ}tzG z3k!w+V2^$bs{spaE{KwW1ap53FC@g~$I3%Q4#o>YM6M>Bfp}m^+HbJ#Md|+FaNqnT zw-c;nazp+nuuN&eb%;{`GCQ!JCzP3i>62wPV49=c8ce&D+k@$jaxXA_wHzWZQURfZ zD_dK)Ag0UVETMz1en&Ug2*4Nm!3SO3S!1(Br)4+h*(WELftc{ zm}(6|tiJ}L?i`YsMJ+_EycVJEAyiDE4k6Z62T^wtNzAw&B34unQwMRsbOX%&(v#a7 zY{7mw)({M)Wf~p8bbO;Xm|oim38P<)pJMfeHsEQCPyI3Pgr~Q= z@2k_t{B(K3ZrnI~BP(0X`q(v(U*Ej&ch<`t~< zcWURg7G!0d@P%`(*V7pvl0RQ?U7k^MOqpe={T0E5&uUotJSAPZ)Di^KlqUCHbaD^5 za+Let3rTsqP;Go^5?V@1){Q z%d6jOZ|WE2-C5YQ{q8r__0bXYJt})2uTNo}%ba(k^waA{4-5YvNjdvE^x$>Zv)}*3 z>0EEpzps8>zRo=Hx@7f+|2KYUX-NMl75~JI5*D#DW9DrB$Y{mJ3=EOYdK^<2!HlDv zO%R3`cO?@@Wb!fIXJDE2e1%|{$;tvR!2b5e`-DspoIycwB9nBlP+JH2bS%mJWoVW2@kwQ$hDxN9<_h!n(r5oyN7 wldVMLz`ST6PkeH@h$h$}dCOJy$N?P^&H^+8uA70OfsuLgei1dcAR&+-0D7CpZ~y=R delta 620 zcmaE(`C5}Vz?+#xgn@y9gW*=B!9?BzoIome>!FP1cN4G4)t~n|Vj$2S?^=|7?vjSNlY))1 zvt*7JTJM)WN>IekjG3dn`7@&>8#6FOHtTatVFWXd zaW+91-rSW;Ad$((d7pu0Ht-dKWhSc#yaWpf2>L+;5{0Z0!8&=VkQrFtJt1kv2a|sZ zDS>(N!qQBN9Fy%hq_hLP8JR?w;gJ%%>BO0@{)`L^7EBBb62Pc}friE%lN*KA!P-{~ zOGnyrGcZ)-=ogn%=A;(uRpjQN8#KLZx6A>cZegH7K(%ntz_^o_fg!#qHK$l#4@9DA zoGdRQ&A4c?wTK+pt{5OsY;uK&raUOE None: _write_minimal_pdf(FIXTURES / "sample_document.pdf") _write_office_documents() + _write_attributed_fixtures() print("wrote:", ", ".join(sorted(p.name for p in FIXTURES.iterdir()))) +def _write_attributed_fixtures() -> None: + """Files that carry real embedded attribution, for M10's harvester. + + Separate from the `sample_*` set on purpose: those deliberately carry *no* + attribution, which is what proves the harvester writes nothing when there is nothing + to read. A single set carrying metadata could not test both halves. + + Every one of these uses the same cast — Jane Doe at the BBC, Panorama, 2019-03-15 — + so one assertion shape covers every format and a mismatch is obvious on sight. + """ + from PIL import Image + + # Video: MP4 tag block. `title` is set deliberately and must NOT be harvested — in a + # media container it names this file rather than a containing work, and mapping it + # would rename half the library on upload. `album` is what stands in for the work. + run([ + "ffmpeg", "-y", "-nostdin", + "-f", "lavfi", "-i", "testsrc=size=160x120:rate=10:duration=1", + "-c:v", "libx264", "-pix_fmt", "yuv420p", + "-metadata", "artist=Jane Doe", + "-metadata", "album=Panorama", + # MP4 has no publisher atom, so ffmpeg drops this one — deliberately left in + # to document that, and why the video fixture has no publisher while the MP3 + # below (ID3 has TPUB) does. + "-metadata", "publisher=BBC", + "-metadata", "copyright=(C) 2019 BBC", + "-metadata", "date=2019-03-15", + "-metadata", "title=Encoder boilerplate that must not be harvested", + "-metadata", "comment=https://example.org/panorama", + str(FIXTURES / "attributed_video.mp4"), + ]) + + # Audio: ID3. A bare year, which is the common case and the reason + # `published_date` is a partial-date string rather than a DateTime. + run([ + "ffmpeg", "-y", "-nostdin", + "-f", "lavfi", "-i", "sine=frequency=440:duration=1", + "-c:a", "libmp3lame", + "-metadata", "artist=Jane Doe", + "-metadata", "album=Panorama", + "-metadata", "publisher=BBC", + "-metadata", "date=2019", + str(FIXTURES / "attributed_audio.mp3"), + ]) + + # Image: EXIF. DateTimeOriginal goes in the Exif sub-IFD (0x8769) where a real + # camera puts it, not in IFD0 — a fixture that put it at the top level would let a + # harvester that only looks there pass while failing on every actual photograph. + image = Image.new("RGB", (320, 240), (60, 120, 180)) + exif = image.getexif() + exif[0x013B] = "Jane Doe" + exif[0x8298] = "(C) 2019 BBC" + # Assigned back, not just mutated: Pillow serialises the sub-IFD from the value + # stored under 0x8769, so mutating the dict get_ifd() returns is silently dropped on + # save. That mistake produces a fixture with no DateTimeOriginal at all, which a + # harvester bug would then "pass" against. + sub_ifd = exif.get_ifd(0x8769) + sub_ifd[0x9003] = "2019:03:15 10:11:12" + exif[0x8769] = sub_ifd + image.save(FIXTURES / "attributed_image.jpg", exif=exif, quality=90) + + _write_attributed_pdf(FIXTURES / "attributed_document.pdf") + + import docx + + document = docx.Document() + document.add_paragraph("Gecko Asset Manager") + properties = document.core_properties + properties.author = "Jane Doe" + properties.title = "Panorama" + properties.created = __import__("datetime").datetime(2019, 3, 15, 10, 11, 12) + document.save(FIXTURES / "attributed_document.docx") + + +def _write_attributed_pdf(target: Path) -> None: + """A one-page PDF carrying an Info dictionary. + + Same hand-built approach as `_write_minimal_pdf`, plus the /Info trailer entry that + holds Author, Title and CreationDate. PDF dates use the D:YYYYMMDDHHmmSS form, which + the harvester's date normaliser has to reduce like every other format's. + """ + objects = [ + b"<< /Type /Catalog /Pages 2 0 R >>", + b"<< /Type /Pages /Kids [3 0 R] /Count 1 >>", + b"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 200 200] " + b"/Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>", + b"<< /Length 62 >>\nstream\nBT /F1 18 Tf 20 100 Td (Gecko Asset Manager) Tj ET\nendstream", + b"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>", + b"<< /Author (Jane Doe) /Title (Panorama) /CreationDate (D:20190315101112Z) >>", + ] + + out = bytearray(b"%PDF-1.4\n") + offsets = [] + for index, body in enumerate(objects, start=1): + offsets.append(len(out)) + out += f"{index} 0 obj\n".encode() + body + b"\nendobj\n" + + xref_at = len(out) + out += f"xref\n0 {len(objects) + 1}\n".encode() + out += b"0000000000 65535 f \n" + for offset in offsets: + out += f"{offset:010d} 00000 n \n".encode() + out += ( + f"trailer\n<< /Size {len(objects) + 1} /Root 1 0 R /Info {len(objects)} 0 R >>\n" + f"startxref\n{xref_at}\n".encode() + + b"%%EOF\n" + ) + target.write_bytes(bytes(out)) + + def _write_office_documents() -> None: """The Office and plain-text fixtures `extract_text` reads. diff --git a/backend/tests/test_attribution_api.py b/backend/tests/test_attribution_api.py new file mode 100644 index 0000000..e805b72 --- /dev/null +++ b/backend/tests/test_attribution_api.py @@ -0,0 +1,388 @@ +"""Attribution over HTTP: editing, inheritance, filtering and search (M10).""" + +from pathlib import Path + +from sqlmodel import select + +from app.models.asset import Asset + +FIXTURES = Path(__file__).parent / "fixtures" + + +def _upload(client, *names: str): + files = [ + ("files", (name, (FIXTURES / name).read_bytes(), "application/octet-stream")) + for name in names + ] + return client.post("/api/assets", files=files) + + +def _upload_one(client, name: str = "sample_video.mp4") -> dict: + return _upload(client, name).json()["created"][0] + + +# ─── editing ───────────────────────────────────────────────────────────────── + + +def test_attribution_is_editable_by_hand(library): + asset = _upload_one(library) + + response = library.patch( + f"/api/assets/{asset['id']}", + json={ + "creator": "Jane Doe", + "publisher": "BBC", + "source_title": "Panorama", + "published_date": "2019-03", + "license": "CC BY 4.0", + "source_url": "https://example.org/panorama", + }, + ) + assert response.status_code == 200 + + updated = response.json()["data"] + assert updated["publisher"] == "BBC" + assert updated["credit"] == "Jane Doe — Panorama — BBC — 2019-03 — CC BY 4.0" + + +def test_a_typed_credit_line_overrides_the_composition(library): + asset = _upload_one(library) + response = library.patch( + f"/api/assets/{asset['id']}", + json={"publisher": "BBC", "credit_line": "Courtesy of the BBC"}, + ) + assert response.json()["data"]["credit"] == "Courtesy of the BBC" + + +def test_correcting_a_field_recomposes_the_credit(library): + """The reason the composition is not stored.""" + asset = _upload_one(library) + library.patch(f"/api/assets/{asset['id']}", json={"publisher": "BBC Two"}) + assert library.get(f"/api/assets/{asset['id']}").json()["data"]["credit"] == "BBC Two" + + library.patch(f"/api/assets/{asset['id']}", json={"publisher": "BBC Four"}) + assert library.get(f"/api/assets/{asset['id']}").json()["data"]["credit"] == "BBC Four" + + +def test_a_partial_published_date_is_accepted(library): + asset = _upload_one(library) + for value in ("1994", "2019-03", "2019-03-15"): + response = library.patch(f"/api/assets/{asset['id']}", json={"published_date": value}) + assert response.status_code == 200, value + assert response.json()["data"]["published_date"] == value + + +def test_a_malformed_published_date_is_refused(library): + """The string column is what allows a partial date; the validator is what stops it + becoming free text, which would break the lexicographic range filters.""" + asset = _upload_one(library) + for value in ("summer 1994", "15/03/2019", "2019-13", "2019-3"): + response = library.patch(f"/api/assets/{asset['id']}", json={"published_date": value}) + assert response.status_code == 422, value + + +# ─── inheritance ───────────────────────────────────────────────────────────── + + +def _clip_of(library, parent_id: str) -> dict: + response = library.post( + f"/api/assets/{parent_id}/clips", json={"in_point": 0.2, "out_point": 0.8} + ) + assert response.status_code == 201, response.text + return response.json()["data"] + + +def test_a_clip_inherits_its_parents_attribution(library): + parent = _upload_one(library) + library.patch( + f"/api/assets/{parent['id']}", + json={"creator": "Jane Doe", "publisher": "BBC", "source_title": "Panorama"}, + ) + + clip = _clip_of(library, parent["id"]) + assert clip["publisher"] == "BBC" + assert clip["credit"] == "Jane Doe — Panorama — BBC" + assert set(clip["attribution_inherited"]) == {"creator", "publisher", "source_title"} + + +def test_a_clip_can_override_one_field_and_keep_the_rest(library): + parent = _upload_one(library) + library.patch( + f"/api/assets/{parent['id']}", + json={"creator": "Jane Doe", "publisher": "BBC"}, + ) + clip = _clip_of(library, parent["id"]) + + updated = library.patch( + f"/api/assets/{clip['id']}", json={"creator": "A different interviewee"} + ).json()["data"] + + assert updated["creator"] == "A different interviewee" + assert updated["publisher"] == "BBC" + assert updated["attribution_inherited"] == ["publisher"] + + +def test_correcting_the_parent_reaches_every_clip_of_it(library): + """The property that made copy-on-create the wrong design.""" + parent = _upload_one(library) + library.patch(f"/api/assets/{parent['id']}", json={"publisher": "BBC Two"}) + clip = _clip_of(library, parent["id"]) + assert clip["publisher"] == "BBC Two" + + library.patch(f"/api/assets/{parent['id']}", json={"publisher": "BBC Four"}) + + refreshed = library.get(f"/api/assets/{clip['id']}").json()["data"] + assert refreshed["publisher"] == "BBC Four" + + +def test_an_unattributed_parent_leaves_the_clip_blank(library): + parent = _upload_one(library) + clip = _clip_of(library, parent["id"]) + assert clip["credit"] == "" + assert clip["attribution_inherited"] == [] + + +# ─── filters ───────────────────────────────────────────────────────────────── + + +def test_filtering_by_publisher(library): + first = _upload_one(library, "sample_video.mp4") + second = _upload_one(library, "sample_image.jpg") + library.patch(f"/api/assets/{first['id']}", json={"publisher": "BBC"}) + library.patch(f"/api/assets/{second['id']}", json={"publisher": "Channel 4"}) + + body = library.get("/api/assets", params={"publisher": "BBC"}).json() + assert [row["id"] for row in body["data"]] == [first["id"]] + assert body["total"] == 1 + + +def test_filtering_by_publisher_also_returns_the_clips_that_inherit_it(library): + """A filter that only looked at the row would answer "everything from the BBC" with + the documentary and none of the clips cut from it.""" + parent = _upload_one(library) + library.patch(f"/api/assets/{parent['id']}", json={"publisher": "BBC"}) + clip = _clip_of(library, parent["id"]) + + body = library.get("/api/assets", params={"publisher": "BBC"}).json() + assert {row["id"] for row in body["data"]} == {parent["id"], clip["id"]} + + +def test_a_clip_that_overrides_the_publisher_drops_out_of_the_parents_filter(library): + parent = _upload_one(library) + library.patch(f"/api/assets/{parent['id']}", json={"publisher": "BBC"}) + clip = _clip_of(library, parent["id"]) + library.patch(f"/api/assets/{clip['id']}", json={"publisher": "Channel 4"}) + + body = library.get("/api/assets", params={"publisher": "BBC"}).json() + assert [row["id"] for row in body["data"]] == [parent["id"]] + + +def test_filtering_by_creator_and_source_title(library): + asset = _upload_one(library) + library.patch( + f"/api/assets/{asset['id']}", json={"creator": "Jane Doe", "source_title": "Panorama"} + ) + + assert library.get("/api/assets", params={"creator": "jane"}).json()["total"] == 1 + assert library.get("/api/assets", params={"source_title": "panorama"}).json()["total"] == 1 + assert library.get("/api/assets", params={"creator": "nobody"}).json()["total"] == 0 + + +def test_published_date_ranges_compare_chronologically(library): + """Lexicographic comparison is chronological because the format is guaranteed.""" + older = _upload_one(library, "sample_video.mp4") + newer = _upload_one(library, "sample_image.jpg") + library.patch(f"/api/assets/{older['id']}", json={"published_date": "2018-12-31"}) + library.patch(f"/api/assets/{newer['id']}", json={"published_date": "2019-03-01"}) + + after = library.get("/api/assets", params={"published_after": "2019"}).json() + assert [row["id"] for row in after["data"]] == [newer["id"]] + + before = library.get("/api/assets", params={"published_before": "2019"}).json() + assert [row["id"] for row in before["data"]] == [older["id"]] + + +def test_unattributed_returns_exactly_what_is_still_missing_a_source(library): + attributed = _upload_one(library, "sample_video.mp4") + blank = _upload_one(library, "sample_image.jpg") + library.patch(f"/api/assets/{attributed['id']}", json={"publisher": "BBC"}) + + body = library.get("/api/assets", params={"unattributed": "true"}).json() + assert [row["id"] for row in body["data"]] == [blank["id"]] + + inverse = library.get("/api/assets", params={"unattributed": "false"}).json() + assert [row["id"] for row in inverse["data"]] == [attributed["id"]] + + +def test_a_clip_that_inherits_is_not_unattributed(library): + """It is not missing a source — it shows its parent's. Counting it would put rows in + the backlog that there is nothing to do about.""" + parent = _upload_one(library) + library.patch(f"/api/assets/{parent['id']}", json={"publisher": "BBC"}) + clip = _clip_of(library, parent["id"]) + + ids = {row["id"] for row in library.get("/api/assets", params={"unattributed": "true"}).json()["data"]} + assert clip["id"] not in ids + assert parent["id"] not in ids + + +def test_a_whitespace_only_value_still_counts_as_unattributed(library): + asset = _upload_one(library) + library.patch(f"/api/assets/{asset['id']}", json={"publisher": " "}) + + body = library.get("/api/assets", params={"unattributed": "true"}).json() + assert [row["id"] for row in body["data"]] == [asset["id"]] + + +# ─── search ────────────────────────────────────────────────────────────────── + + +def test_an_asset_is_findable_by_its_publisher(library): + asset = _upload_one(library) + library.patch( + f"/api/assets/{asset['id']}", json={"publisher": "BBC", "source_title": "Panorama"} + ) + + for term in ("BBC", "Panorama"): + body = library.get("/api/search", params={"q": term}).json() + assert [hit["asset"]["id"] for hit in body["data"]] == [asset["id"]], term + + +def test_a_clip_is_findable_by_the_publisher_it_inherited(library): + parent = _upload_one(library) + library.patch(f"/api/assets/{parent['id']}", json={"publisher": "BBC"}) + clip = _clip_of(library, parent["id"]) + + found = {hit["asset"]["id"] for hit in library.get("/api/search", params={"q": "BBC"}).json()["data"]} + assert clip["id"] in found + + +def test_correcting_a_parent_reindexes_its_clips(library, session): + """The keyword index stores a snapshot, so it is the one place inheritance cannot be + resolved on read — without the child re-index, a clip stays searchable under the old + publisher and invisible under the new one.""" + parent = _upload_one(library) + library.patch(f"/api/assets/{parent['id']}", json={"publisher": "BBC"}) + clip = _clip_of(library, parent["id"]) + + library.patch(f"/api/assets/{parent['id']}", json={"publisher": "Channel 4"}) + + found = {hit["asset"]["id"] for hit in library.get("/api/search", params={"q": "Channel"}).json()["data"]} + assert clip["id"] in found + + stale = {hit["asset"]["id"] for hit in library.get("/api/search", params={"q": "BBC"}).json()["data"]} + assert clip["id"] not in stale + + +def test_attribution_does_not_leak_between_users(library, session): + """Every filter above narrows within one user's library; this is the one that would + be a disclosure rather than a wrong answer.""" + asset = _upload_one(library) + library.patch(f"/api/assets/{asset['id']}", json={"publisher": "BBC"}) + + stored = session.exec(select(Asset)).first() + stored.user_id = "somebody-else" + session.add(stored) + session.commit() + + assert library.get("/api/assets", params={"publisher": "BBC"}).json()["total"] == 0 + assert library.get("/api/search", params={"q": "BBC"}).json()["data"] == [] + + +# ─── the library-wide re-harvest ───────────────────────────────────────────── + + +def _run_job(job_id, session, monkeypatch): + """Run a queued job against the test database. + + The queue builds its own session from its own engine, so it has to be pointed at + this test's — the same helper `test_embedding_backfill.py` uses for library jobs. + """ + from app.jobs import enrichment as enrichment_jobs + + monkeypatch.setattr(enrichment_jobs.queue(), "engine", session.get_bind()) + enrichment_jobs._run_job(job_id) + session.expire_all() + + +def test_harvest_queues_one_library_job(library, session): + from app.models.job import KIND_HARVEST_ATTRIBUTION, EnrichmentJob + + _upload(library, "sample_image.jpg", "sample_video.mp4") + + response = library.post("/api/assets/harvest-attribution") + + assert response.status_code == 202 + body = response.json()["data"] + assert body["action"] == KIND_HARVEST_ATTRIBUTION + assert body["asset_id"] is None + assert len(session.exec(select(EnrichmentJob)).all()) == 1 + + +def test_a_second_harvest_while_one_runs_is_a_409(library, session): + library.post("/api/assets/harvest-attribution") + assert library.post("/api/assets/harvest-attribution").status_code == 409 + + +def test_the_harvest_attributes_files_uploaded_before_it_existed(library, session, monkeypatch): + """The case this endpoint exists for: the metadata was on disk the whole time and + nothing had ever looked.""" + asset = _upload_one(library, "attributed_image.jpg") + + # Wind the row back to how it would look had it been ingested before M10. + stored = session.get(Asset, asset["id"]) + stored.creator = None + stored.license = None + stored.published_date = None + stored.field_provenance = "{}" + session.add(stored) + session.commit() + + job = library.post("/api/assets/harvest-attribution").json()["data"] + _run_job(job["id"], session, monkeypatch) + + refreshed = library.get(f"/api/assets/{asset['id']}").json()["data"] + assert refreshed["creator"] == "Jane Doe" + assert refreshed["license"] == "(C) 2019 BBC" + + +def test_the_harvest_never_overwrites_a_hand_typed_value(library, session, monkeypatch): + """Which is what makes it safe to run twice, and safe to run unasked.""" + asset = _upload_one(library, "attributed_image.jpg") + library.patch(f"/api/assets/{asset['id']}", json={"creator": "The actual photographer"}) + + job = library.post("/api/assets/harvest-attribution").json()["data"] + _run_job(job["id"], session, monkeypatch) + + refreshed = library.get(f"/api/assets/{asset['id']}").json()["data"] + assert refreshed["creator"] == "The actual photographer" + + +def test_the_harvest_reports_what_it_did(library, session, monkeypatch): + from app.models.job import EnrichmentJob + + _upload(library, "attributed_image.jpg", "sample_image.jpg") + stored = session.exec(select(Asset).where(Asset.creator.is_not(None))).first() + stored.creator = None + stored.license = None + stored.published_date = None + session.add(stored) + session.commit() + + job = library.post("/api/assets/harvest-attribution").json()["data"] + _run_job(job["id"], session, monkeypatch) + + finished = session.get(EnrichmentJob, job["id"]) + assert finished.status == "done" + assert "1 attributed of 2 scanned" in finished.detail + + +def test_the_harvest_skips_clips_which_own_no_bytes(library, session): + """A clip inherits on read; there is no file under it to read metadata from.""" + from app.enrichment.harvest_attribution import run + + parent = _upload_one(library, "attributed_video.mp4") + _clip_of(library, parent["id"]) + + result = run(session, session.get(Asset, parent["id"]).user_id, lambda *a, **k: None) + assert result.scanned == 1 diff --git a/backend/tests/test_embedded_metadata.py b/backend/tests/test_embedded_metadata.py new file mode 100644 index 0000000..70f3d83 --- /dev/null +++ b/backend/tests/test_embedded_metadata.py @@ -0,0 +1,219 @@ +"""Attribution read out of a file's own container metadata (M10). + +Run against real files rather than mocked tag dictionaries: the thing most likely to be +wrong here is what a format actually stores and where, which a stub cannot be wrong +about. `tests/make_fixtures.py` builds the `attributed_*` set; the `sample_*` set +deliberately carries no attribution, which is what proves the harvester stays silent +when there is nothing to read. +""" + +from datetime import datetime +from pathlib import Path + +import pytest +from sqlmodel import select + +from app.ingest.embedded_metadata import harvest, normalise_partial_date +from app.ingest.probe import probe +from app.models.asset import PROVENANCE_EMBEDDED, Asset + +FIXTURES = Path(__file__).parent / "fixtures" + + +def _harvest(name: str, asset_type: str) -> dict: + path = FIXTURES / name + tags = probe(path).tags if asset_type in ("video", "audio") else {} + return harvest(path, asset_type, name, probe_tags=tags) + + +def _upload(client, *names: str): + files = [ + ("files", (name, (FIXTURES / name).read_bytes(), "application/octet-stream")) + for name in names + ] + return client.post("/api/assets", files=files) + + +# ─── per format ────────────────────────────────────────────────────────────── + + +def test_mp4_tag_block_is_read(): + found = _harvest("attributed_video.mp4", "video") + assert found["creator"] == "Jane Doe" + assert found["source_title"] == "Panorama" + assert found["license"] == "(C) 2019 BBC" + assert found["published_date"] == "2019-03-15" + + +def test_id3_tags_are_read(): + found = _harvest("attributed_audio.mp3", "audio") + assert found["creator"] == "Jane Doe" + assert found["publisher"] == "BBC" + assert found["source_title"] == "Panorama" + + +def test_a_bare_year_stays_a_bare_year(): + """The case the string column exists for: ID3 routinely carries only a year, and + turning it into 2019-01-01 would invent a precision the file never claimed.""" + assert _harvest("attributed_audio.mp3", "audio")["published_date"] == "2019" + + +def test_exif_is_read_including_the_sub_ifd(): + """DateTimeOriginal lives in the Exif sub-IFD (0x8769), where every real camera puts + it — a harvester reading only IFD0 passes on hand-built files and fails on photos.""" + found = _harvest("attributed_image.jpg", "image") + assert found["creator"] == "Jane Doe" + assert found["license"] == "(C) 2019 BBC" + assert found["published_date"] == "2019-03-15" + + +def test_pdf_info_dictionary_is_read(): + found = _harvest("attributed_document.pdf", "document") + assert found["creator"] == "Jane Doe" + assert found["source_title"] == "Panorama" + # D:20190315101112Z — prefixed and separator-less, unlike every other format. + assert found["published_date"] == "2019-03-15" + + +def test_docx_core_properties_are_read(): + found = _harvest("attributed_document.docx", "document") + assert found["creator"] == "Jane Doe" + assert found["source_title"] == "Panorama" + assert found["published_date"] == "2019-03-15" + + +# ─── what must NOT be harvested ────────────────────────────────────────────── + + +def test_a_media_containers_title_is_not_harvested(): + """In a media container `title` names this file, not a containing work, and is very + often an encoder's boilerplate. The fixture sets one precisely so this can assert it + goes nowhere — mapping it would rename or mis-source half a library on upload.""" + found = _harvest("attributed_video.mp4", "video") + assert "Encoder boilerplate" not in str(found) + assert found.get("source_title") == "Panorama" + assert "name" not in found + + +def test_technical_tags_are_not_mistaken_for_attribution(): + """Every MP4 carries encoder, handler_name, major_brand and compatible_brands. A + mapping that took whatever it recognised would file "Lavf60.16.100" as a creator.""" + found = _harvest("sample_video.mp4", "video") + assert found == {} + + +def test_files_with_no_attribution_yield_nothing(): + assert _harvest("sample_image.jpg", "image") == {} + assert _harvest("sample_document.pdf", "document") == {} + assert _harvest("sample_audio.mp3", "audio") == {} + + +def test_retrieved_at_is_never_harvested(): + """When *you* fetched something is not a fact the file can know.""" + for name, kind in [ + ("attributed_video.mp4", "video"), + ("attributed_image.jpg", "image"), + ("attributed_document.pdf", "document"), + ]: + assert "retrieved_at" not in _harvest(name, kind) + + +def test_a_comment_is_only_taken_as_a_url_when_it_is_one(): + """Downloaders put the source URL in `comment`. They also put everything else there.""" + from app.ingest.embedded_metadata import _from_container_tags + + assert _from_container_tags({"comment": "https://example.org/x"})["source_url"] == ( + "https://example.org/x" + ) + assert "source_url" not in _from_container_tags({"comment": "Recorded off-air, poor audio"}) + + +def test_harvest_never_raises_on_an_unreadable_file(tmp_path): + """This runs after the row is committed; an exception here must not reach the upload.""" + broken = tmp_path / "broken.jpg" + broken.write_bytes(b"not an image") + assert harvest(broken, "image", "broken.jpg") == {} + + +def test_an_unknown_document_format_is_simply_empty(): + assert _harvest("sample_document.txt", "document") == {} + + +# ─── the date normaliser ───────────────────────────────────────────────────── + + +@pytest.mark.parametrize( + "raw,expected", + [ + ("2019", "2019"), + ("2019-03", "2019-03"), + ("2019-03-15", "2019-03-15"), + ("2019-03-15T10:00:00.000000Z", "2019-03-15"), # MP4 + ("2019:03:15 10:11:12", "2019-03-15"), # EXIF, colon separated + ("D:20190315101112Z", "2019-03-15"), # PDF + ("20190315", "2019-03-15"), # compact, no prefix + ("", None), + ("not a date", None), + (None, None), + ("2019-19-40", "2019"), # degrades to what is still trustworthy + ], +) +def test_partial_dates_normalise_without_inventing_precision(raw, expected): + assert normalise_partial_date(raw) == expected + + +def test_a_real_datetime_is_accepted(): + """Office hands back a datetime object rather than a string.""" + assert normalise_partial_date(datetime(2020, 5, 6, 12, 0)) == "2020-05-06" + + +# ─── through the upload pipeline ───────────────────────────────────────────── + + +def test_an_upload_lands_attributed(library, session): + response = _upload(library, "attributed_image.jpg") + assert response.status_code == 201 + + asset = response.json()["created"][0] + assert asset["creator"] == "Jane Doe" + assert asset["license"] == "(C) 2019 BBC" + assert asset["published_date"] == "2019-03-15" + + +def test_embedded_values_are_stamped_as_embedded_not_human(library, session): + """The distinction that lets a later pass propose over a camera-supplied name + without ever proposing over something the user typed.""" + _upload(library, "attributed_image.jpg") + + stored = session.exec(select(Asset)).first() + import json + + provenance = json.loads(stored.field_provenance) + assert provenance["creator"] == PROVENANCE_EMBEDDED + + +def test_an_upload_with_no_embedded_metadata_stays_unattributed(library, session): + response = _upload(library, "sample_image.jpg") + asset = response.json()["created"][0] + + assert asset["creator"] is None + assert asset["credit"] == "" + + +def test_a_harvest_never_overwrites_a_value_already_there(library, session): + """Filtering to empty fields is what makes the harvest safe to run unasked, and safe + to re-run over a library that has been hand-corrected.""" + from app.services import assets as asset_service + + _upload(library, "attributed_image.jpg") + stored = session.exec(select(Asset)).first() + + asset_service.apply_metadata(session, stored, {"creator": "The actual photographer"}) + asset_service.apply_embedded_attribution( + session, stored, {"creator": "Jane Doe", "publisher": "BBC"} + ) + session.refresh(stored) + + assert stored.creator == "The actual photographer" + # The empty one is still filled, so a partial correction does not block the rest. + assert stored.publisher == "BBC" From 35e8eb49db1274641db260d3084b76dcf5eb9fd5 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 20 Sep 2026 08:06:01 +0000 Subject: [PATCH 3/4] feat(attribution): a Source tab, and filters that follow inheritance MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit M10, step 5 of six. Attribution is now reachable: a Source tab on the asset panel, creator/publisher filters and a "missing a source" chip in the library, and the resolved credit in the Info facts. The panel renders resolved values, and marks the inherited ones. That marking is load-bearing rather than decorative: an inherited value that looked typed is one a user would "correct" in place, which silently detaches that field from its source so that fixing the original later no longer reaches it. The clip is told where the value came from instead. Saving sends only the fields that changed, for the reason the description editor already does: the API stamps every key it receives as human-written, so resending the seven you did not touch would mark them hand-verified — including any a harvest had filled, which is exactly the distinction the "embedded" provenance value exists to keep. The credit preview duplicates the server's composition so it updates as you type. Deliberate: the alternative is a round trip per keystroke to render pure formatting, and the server's `credit` stays authoritative for anything saved. `unattributed` is only sent when the chip is on. `false` is a real filter server-side — "everything that *has* attribution" — so sending it whenever the chip was off would hide every unattributed asset from the default library view, which is the opposite of the point. The filter inputs commit on blur or Enter rather than per keystroke: every filter change reloads the library, so a controlled input bound straight to the store would be one request per character. Escape is stopped at the panel. `TagInput` and the transcript editor let it through to `DetailDock`'s window listener, so abandoning a half-typed value there closes the whole asset; that bug is not in this milestone's scope, but this is not a ninth instance of it. Six existing test files build their own asset literals, so the ten new fields land via one shared `noAttribution` spread rather than being typed out six times. 280 frontend tests, 916 backend. Lint and build clean. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_019RWM7S6z1UPsZqso4p2HAj --- frontend/src/api/assets.ts | 73 +++++ frontend/src/components/AssetDetail.tsx | 11 + frontend/src/components/AssetThumb.test.tsx | 2 + .../src/components/AttributionPanel.test.tsx | 186 +++++++++++ frontend/src/components/AttributionPanel.tsx | 303 ++++++++++++++++++ frontend/src/components/FilterBar.tsx | 120 ++++++- frontend/src/components/SelectionBar.test.tsx | 2 + frontend/src/stores/library.test.ts | 41 +++ frontend/src/stores/library.ts | 33 ++ frontend/src/test-fixtures.ts | 37 +++ frontend/src/views/AssetView.test.tsx | 2 + frontend/src/views/LibraryView.test.tsx | 2 + frontend/src/views/SearchView.test.tsx | 2 + 13 files changed, 813 insertions(+), 1 deletion(-) create mode 100644 frontend/src/components/AttributionPanel.test.tsx create mode 100644 frontend/src/components/AttributionPanel.tsx create mode 100644 frontend/src/test-fixtures.ts diff --git a/frontend/src/api/assets.ts b/frontend/src/api/assets.ts index b1d4088..3a840b7 100644 --- a/frontend/src/api/assets.ts +++ b/frontend/src/api/assets.ts @@ -1,5 +1,6 @@ import client from '@/api/client' import type { Tag } from '@/api/tags' +import type { ActivityJob } from '@/api/transcripts' /** * The asset library API. @@ -49,6 +50,31 @@ export interface Asset { /** Batch-loaded for a page by the server, so reading this is free. */ tags: Tag[] + /** + * M10. Whose work this is — as opposed to `source`, which is how the file got here. + * + * These are the *resolved* values: a clip with nothing of its own carries what it + * inherited from its parent, so a clip of an attributed video shows a real credit + * rather than eight blanks. `attribution_inherited` names which of them came that + * way, so the panel can mark them instead of letting an inherited value look like + * something typed on the clip. + */ + source_url: string | null + creator: string | null + publisher: string | null + source_title: string | null + /** ISO 8601 partial: `YYYY`, `YYYY-MM` or `YYYY-MM-DD`. Partial on purpose — a book + * is from 1994 and nothing should invent a January 1st for it. */ + published_date: string | null + retrieved_at: string | null + license: string | null + /** An override. Null means the displayed `credit` is composed from the fields above. */ + credit_line: string | null + + /** Read-only: the line to display, composed unless `credit_line` overrides it. */ + credit: string + attribution_inherited: string[] + upload_date: string modified_date: string metadata_modified_date: string @@ -82,6 +108,15 @@ export interface ListAssetsParams { max_duration?: number uploaded_after?: string uploaded_before?: string + /** M10. Each of these resolves through a clip's parent server-side, so filtering by + * publisher returns the clips that inherit it as well as the asset itself. */ + creator?: string + publisher?: string + source_title?: string + published_after?: string + published_before?: string + /** True for "still missing a source" — how a backlog gets worked through. */ + unattributed?: boolean limit?: number offset?: number } @@ -97,8 +132,33 @@ export interface AssetUpdate { name?: string description?: string | null summary?: string | null + + /** M10. The only path that can *correct* attribution: the harvester fills blanks + * only, and an AI may propose but never write. */ + source_url?: string | null + creator?: string | null + publisher?: string | null + source_title?: string | null + published_date?: string | null + retrieved_at?: string | null + license?: string | null + credit_line?: string | null } +/** The attribution fields, in the order the panel shows them. */ +export const ATTRIBUTION_FIELDS = [ + 'creator', + 'source_title', + 'publisher', + 'published_date', + 'source_url', + 'license', + 'retrieved_at', + 'credit_line', +] as const + +export type AttributionField = (typeof ATTRIBUTION_FIELDS)[number] + interface DataResponse { data: T } @@ -144,4 +204,17 @@ export const assetsApi = { remove(id: string): Promise { return client.delete(`/assets/${id}`).then(() => undefined) }, + + /** + * Re-read embedded metadata for every file already in the library. + * + * For everything uploaded before M10 existed: the EXIF and ID3 have been sitting on + * disk the whole time and nothing ever looked. Returns the queued job, which shows up + * in the activity feed like any other. + */ + harvestAttribution(): Promise { + return client + .post>('/assets/harvest-attribution') + .then((r) => r.data.data) + }, } diff --git a/frontend/src/components/AssetDetail.tsx b/frontend/src/components/AssetDetail.tsx index cd054da..2c6e50e 100644 --- a/frontend/src/components/AssetDetail.tsx +++ b/frontend/src/components/AssetDetail.tsx @@ -6,6 +6,7 @@ import { Link2, Mic, Pencil, + Quote, ScanText, Scissors, Tags as TagsIcon, @@ -25,6 +26,7 @@ import { useTagStore } from '@/stores/tags' import { formatBytes, formatDate, formatDimensions, formatDuration } from '@/utils/format' import { useAutoGrow } from '@/utils/useAutoGrow' import AssetThumb from '@/components/AssetThumb' +import AttributionPanel from '@/components/AttributionPanel' import ClipEditor from '@/components/ClipEditor' import DocumentTextPanel from '@/components/DocumentTextPanel' import EmbedButton from '@/components/EmbedButton' @@ -495,6 +497,9 @@ export default function AssetDetail({ + {/* The resolved credit, so it is readable without opening the Source tab — and + visible on a clip, which shows what it inherited. */} + {usage && usage.total_events > 0 && ( , + }, { id: 'info', label: 'Info', icon: Info, content: info }, ] diff --git a/frontend/src/components/AssetThumb.test.tsx b/frontend/src/components/AssetThumb.test.tsx index f2dbb2b..890a774 100644 --- a/frontend/src/components/AssetThumb.test.tsx +++ b/frontend/src/components/AssetThumb.test.tsx @@ -2,6 +2,7 @@ import { render, screen } from '@testing-library/react' import { describe, expect, it } from 'vitest' import AssetThumb from '@/components/AssetThumb' import type { Asset } from '@/api/assets' +import { noAttribution } from '@/test-fixtures' function makeAsset(overrides: Partial = {}): Asset { return { @@ -9,6 +10,7 @@ function makeAsset(overrides: Partial = {}): Asset { name: 'Test', description: null, summary: null, + ...noAttribution, asset_type: 'image', source: 'local_upload', parent_asset_id: null, diff --git a/frontend/src/components/AttributionPanel.test.tsx b/frontend/src/components/AttributionPanel.test.tsx new file mode 100644 index 0000000..8fae602 --- /dev/null +++ b/frontend/src/components/AttributionPanel.test.tsx @@ -0,0 +1,186 @@ +import { render, screen, waitFor } from '@testing-library/react' +import userEvent from '@testing-library/user-event' +import { beforeEach, describe, expect, it, vi } from 'vitest' +import AttributionPanel from '@/components/AttributionPanel' +import { useLibraryStore } from '@/stores/library' +import type { Asset } from '@/api/assets' +import { noAttribution } from '@/test-fixtures' + +function asset(overrides: Partial = {}): Asset { + return { + id: 'a1', + name: 'Interview', + description: null, + summary: null, + ...noAttribution, + asset_type: 'video', + source: 'local_upload', + parent_asset_id: null, + in_point: null, + out_point: null, + original_name: 'interview.mp4', + mime_type: 'video/mp4', + file_format: 'mp4', + size_bytes: 100, + duration_seconds: 60, + width: 1920, + height: 1080, + codec: 'h264', + file_url: null, + thumb_url: null, + missing: false, + tags: [], + upload_date: '2026-01-01T00:00:00', + modified_date: '2026-01-01T00:00:00', + metadata_modified_date: '2026-01-01T00:00:00', + ...overrides, + } +} + +const update = vi.fn().mockResolvedValue(undefined) + +beforeEach(() => { + update.mockClear() + useLibraryStore.setState({ update }) +}) + +describe('AttributionPanel', () => { + it('shows the fields that are filled in', () => { + render() + + expect(screen.getByLabelText('Creator')).toHaveValue('Jane Doe') + expect(screen.getByLabelText('Publisher')).toHaveValue('BBC') + }) + + it('previews the composed credit as you type', async () => { + const user = userEvent.setup() + render() + + await user.type(screen.getByLabelText('Creator'), 'Jane Doe') + await user.type(screen.getByLabelText('Publisher'), 'BBC') + + expect(screen.getByText('Jane Doe — BBC')).toBeInTheDocument() + }) + + it('prefers a typed credit line over the composition', async () => { + const user = userEvent.setup() + render() + + await user.type(screen.getByLabelText('Credit line'), 'Courtesy of the BBC') + + expect(screen.getByText('Courtesy of the BBC')).toBeInTheDocument() + expect(screen.queryByText('BBC', { selector: 'p' })).not.toBeInTheDocument() + }) + + it('sends only the fields that changed', async () => { + const user = userEvent.setup() + render() + + await user.type(screen.getByLabelText('Publisher'), 'BBC') + await user.click(screen.getByRole('button', { name: 'Save' })) + + // Not `creator` as well: the API stamps every key it receives as human-written, so + // resending an untouched field would mark it hand-verified when it was not. + await waitFor(() => expect(update).toHaveBeenCalledWith('a1', { publisher: 'BBC' })) + }) + + it('clears a field by sending null rather than an empty string', async () => { + const user = userEvent.setup() + render() + + await user.clear(screen.getByLabelText('Publisher')) + await user.click(screen.getByRole('button', { name: 'Save' })) + + await waitFor(() => expect(update).toHaveBeenCalledWith('a1', { publisher: null })) + }) + + it('cannot be saved until something changes', () => { + render() + expect(screen.getByRole('button', { name: 'Save' })).toBeDisabled() + }) + + it('refuses a malformed published date before it reaches the server', async () => { + const user = userEvent.setup() + render() + + await user.type(screen.getByLabelText('Published'), 'summer 1994') + + expect(screen.getByText('Use YYYY, YYYY-MM or YYYY-MM-DD.')).toBeInTheDocument() + expect(screen.getByRole('button', { name: 'Save' })).toBeDisabled() + expect(update).not.toHaveBeenCalled() + }) + + it('accepts a bare year', async () => { + const user = userEvent.setup() + render() + + await user.type(screen.getByLabelText('Published'), '1994') + + expect(screen.queryByText('Use YYYY, YYYY-MM or YYYY-MM-DD.')).not.toBeInTheDocument() + expect(screen.getByRole('button', { name: 'Save' })).toBeEnabled() + }) + + it('marks an inherited value as inherited', () => { + render( + + ) + + expect(screen.getByText('inherited from the original')).toBeInTheDocument() + }) + + it('does not mark a value the asset carries itself', () => { + render() + expect(screen.queryByText(/inherited from/)).not.toBeInTheDocument() + }) + + it('copies the credit', async () => { + // Spied rather than replaced: userEvent.setup() installs its own clipboard stub and + // defines it non-configurably, so assigning over it silently does nothing and the + // spy never sees the call. + const user = userEvent.setup() + const writeText = vi.spyOn(navigator.clipboard, 'writeText').mockResolvedValue() + + render( + + ) + + await user.click(screen.getByRole('button', { name: 'Copy this credit' })) + + await waitFor(() => expect(writeText).toHaveBeenCalledWith('Jane Doe — BBC')) + }) + + it('keeps the draft when a save is refused', async () => { + update.mockRejectedValueOnce(new Error('nope')) + const user = userEvent.setup() + render() + + await user.type(screen.getByLabelText('Publisher'), 'BBC') + await user.click(screen.getByRole('button', { name: 'Save' })) + + expect(await screen.findByText(/Could not save/)).toBeInTheDocument() + expect(screen.getByLabelText('Publisher')).toHaveValue('BBC') + }) + + it('does not let Escape reach the panel behind it', async () => { + const onKeyDown = vi.fn() + const user = userEvent.setup() + render( +

+ +
+ ) + + await user.click(screen.getByLabelText('Creator')) + await user.keyboard('{Escape}') + + // DetailDock listens for Escape to close the whole asset panel; abandoning a + // half-typed credit must not also shut the asset. + expect(onKeyDown).not.toHaveBeenCalled() + }) +}) diff --git a/frontend/src/components/AttributionPanel.tsx b/frontend/src/components/AttributionPanel.tsx new file mode 100644 index 0000000..1e3a183 --- /dev/null +++ b/frontend/src/components/AttributionPanel.tsx @@ -0,0 +1,303 @@ +import { useEffect, useMemo, useState } from 'react' +import type { KeyboardEvent } from 'react' +import { Check, Copy, Link2 } from 'lucide-react' +import type { Asset, AssetUpdate, AttributionField } from '@/api/assets' +import { useLibraryStore } from '@/stores/library' +import { useSavedFlash } from '@/utils/useSavedFlash' + +/** + * Where an asset's content came from, so it can be credited when it is used. + * + * `Asset.source` says how the file arrived — uploaded, generated, cut from something + * else. This is the other question: whose work it is. See docs/m10-attribution.md. + * + * The values arriving on `asset` are already *resolved*: a clip with nothing of its own + * carries what it inherited from its parent, and `attribution_inherited` names which. + * Those render as inherited rather than as ordinary values, because an inherited value + * that looked typed is one a user would "correct" here — silently detaching that field + * from the source, so that fixing the parent later no longer reaches it. + */ + +interface FieldSpec { + name: AttributionField + label: string + placeholder: string + hint?: string + type?: 'text' | 'url' | 'date' +} + +const FIELDS: FieldSpec[] = [ + { + name: 'creator', + label: 'Creator', + placeholder: 'Author, photographer, speaker, director', + }, + { + name: 'source_title', + label: 'Source title', + placeholder: 'The programme, film, article or book this is part of', + }, + { + name: 'publisher', + label: 'Publisher', + placeholder: 'Outlet, channel, studio, imprint', + }, + { + name: 'published_date', + label: 'Published', + placeholder: 'YYYY, YYYY-MM or YYYY-MM-DD', + hint: 'A year alone is fine — better than inventing a day nobody knows.', + }, + { + name: 'source_url', + label: 'Source URL', + placeholder: 'https://…', + type: 'url', + }, + { + name: 'license', + label: 'Licence / rights', + placeholder: 'CC BY 4.0, © the BBC, unknown', + }, +] + +/** Accepted by the server: a whole year, a year and month, or a full date. */ +const ISO_PARTIAL_DATE = /^\d{4}(?:-(?:0[1-9]|1[0-2])(?:-(?:0[1-9]|[12]\d|3[01]))?)?$/ + +type Draft = Record + +function draftFrom(asset: Asset): Draft { + return { + creator: asset.creator ?? '', + source_title: asset.source_title ?? '', + publisher: asset.publisher ?? '', + published_date: asset.published_date ?? '', + source_url: asset.source_url ?? '', + license: asset.license ?? '', + retrieved_at: asset.retrieved_at ?? '', + credit_line: asset.credit_line ?? '', + } +} + +/** + * The same composition the server performs, so the preview updates as you type. + * + * Duplicated from `app/attribution.py::compose_credit` on purpose: the alternative is a + * round trip per keystroke to show a line that is pure formatting. `credit` from the + * server remains authoritative — this only ever renders an unsaved draft. + */ +function composeCredit(draft: Draft): string { + return [ + draft.creator, + draft.source_title, + draft.publisher, + draft.published_date, + draft.license, + ] + .map((value) => value.trim()) + .filter(Boolean) + .join(' — ') +} + +interface Props { + asset: Asset +} + +export default function AttributionPanel({ asset }: Props) { + const update = useLibraryStore((s) => s.update) + const [draft, setDraft] = useState(() => draftFrom(asset)) + const [saving, setSaving] = useState(false) + const [error, setError] = useState(null) + const [saved, flashSaved] = useSavedFlash() + const [copied, setCopied] = useState(false) + + // Re-seed when the panel is pointed at a different asset, or when the server's copy + // changes underneath it — a harvest job filling blanks is exactly that case. + useEffect(() => { + setDraft(draftFrom(asset)) + setError(null) + // Deliberately not `[asset]`: the store hands back a new object on every update, so + // depending on its identity would re-seed the draft mid-edit and discard whatever + // was half-typed. Re-seeding on a different asset, or on a server-side change to + // this one (a harvest filling blanks), is the whole intent. + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [asset.id, asset.metadata_modified_date]) + + const inherited = useMemo( + () => new Set(asset.attribution_inherited), + [asset.attribution_inherited] + ) + + const dirty = useMemo(() => { + const current = draftFrom(asset) + return (Object.keys(current) as AttributionField[]).some( + (key) => draft[key] !== current[key] + ) + }, [asset, draft]) + + const dateValid = + !draft.published_date.trim() || ISO_PARTIAL_DATE.test(draft.published_date.trim()) + + // The override when there is one, else the live composition — the same precedence the + // server applies, so what is previewed is what will be stored. + const preview = draft.credit_line.trim() || composeCredit(draft) + + const set = (name: AttributionField, value: string) => + setDraft((previous) => ({ ...previous, [name]: value })) + + const save = async () => { + if (!dirty || !dateValid) return + setSaving(true) + setError(null) + try { + // Only what changed. The API stamps every key it receives as human-written, so + // sending all eight would mark the seven you did not touch as hand-verified — + // including the ones a harvest filled, which is the distinction `provenance` + // exists to keep. + const current = draftFrom(asset) + const changes: AssetUpdate = {} + for (const key of Object.keys(current) as AttributionField[]) { + if (draft[key] !== current[key]) changes[key] = draft[key].trim() || null + } + + await update(asset.id, changes) + flashSaved() + } catch { + // The store restores the server's version and surfaces the message; leaving the + // draft alone means a rejected edit is still on screen to fix. + setError('Could not save. Your changes are still here.') + } finally { + setSaving(false) + } + } + + const copyCredit = async () => { + const line = asset.credit || preview + if (!line) return + try { + await navigator.clipboard.writeText(line) + setCopied(true) + setTimeout(() => setCopied(false), 1500) + } catch { + // Clipboard access needs a secure context and a user gesture. Prompting with the + // text beats a button that silently does nothing. + window.prompt('Copy this credit', line) + } + } + + // Escape must not reach DetailDock's window listener, which closes the whole panel — + // abandoning a half-typed credit should not also shut the asset. + const swallowEscape = (event: KeyboardEvent) => { + if (event.key === 'Escape') event.stopPropagation() + } + + return ( +
+

+ Where this came from, so it can be credited when you use it. +

+ + {FIELDS.map((field) => ( +
+
+ + {inherited.has(field.name) && ( + + inherited from {asset.parent_asset_id ? 'the original' : 'its source'} + + )} +
+ set(field.name, e.target.value)} + /> + {field.name === 'published_date' && !dateValid && ( +

+ Use YYYY, YYYY-MM or YYYY-MM-DD. +

+ )} + {field.hint && dateValid && field.name === 'published_date' && ( +

{field.hint}

+ )} +
+ ))} + +
+ + set('credit_line', e.target.value)} + /> +

+ Leave this empty and it is built from the fields above, so correcting one keeps + the credit right. +

+
+ + {preview && ( +
+
+
+

Credit

+

+ {preview} +

+
+ +
+
+ )} + + {asset.source_url && ( + + + Open the source + + )} + + {error &&

{error}

} + +
+ + {saved && ( + + + Saved + + )} +
+
+ ) +} diff --git a/frontend/src/components/FilterBar.tsx b/frontend/src/components/FilterBar.tsx index a2fda48..8630375 100644 --- a/frontend/src/components/FilterBar.tsx +++ b/frontend/src/components/FilterBar.tsx @@ -63,6 +63,12 @@ export default function FilterBar() { const toggleTag = useLibraryStore((s) => s.toggleTag) const setCategoryFilter = useLibraryStore((s) => s.setCategoryFilter) const setSourceFilter = useLibraryStore((s) => s.setSourceFilter) + const creatorFilter = useLibraryStore((s) => s.creatorFilter) + const publisherFilter = useLibraryStore((s) => s.publisherFilter) + const unattributedOnly = useLibraryStore((s) => s.unattributedOnly) + const setCreatorFilter = useLibraryStore((s) => s.setCreatorFilter) + const setPublisherFilter = useLibraryStore((s) => s.setPublisherFilter) + const setUnattributedOnly = useLibraryStore((s) => s.setUnattributedOnly) const setDurationRange = useLibraryStore((s) => s.setDurationRange) const setUploadedRange = useLibraryStore((s) => s.setUploadedRange) const clearFilters = useLibraryStore((s) => s.clearFilters) @@ -128,7 +134,10 @@ export default function FilterBar() { (categoryFilter ? 1 : 0) + (sourceFilter ? 1 : 0) + (minDuration !== null || maxDuration !== null ? 1 : 0) + - (uploadedAfter || uploadedBefore ? 1 : 0) + (uploadedAfter || uploadedBefore ? 1 : 0) + + (creatorFilter ? 1 : 0) + + (publisherFilter ? 1 : 0) + + (unattributedOnly ? 1 : 0) const anyActive = activeCount > 0 || Boolean(query.trim()) || Boolean(typeFilter) const durationLabel = () => { @@ -236,6 +245,50 @@ export default function FilterBar() { + {/* M10. Both resolve through a clip's parent server-side, so filtering by + publisher returns the clips cut from an attributed video as well as the + video itself. Committed on blur or Enter rather than per keystroke: each + change reloads the library, and a request per character would be one + request per character. */} +
+ + setCreatorFilter(value || null)} + /> +
+ +
+ + setPublisherFilter(value || null)} + /> +
+ +
+ Attribution + +

+ A clip showing its original's credit is not missing one. +

+
+
Duration (seconds)
@@ -320,6 +373,24 @@ export default function FilterBar() { onClear={() => setUploadedRange(null, null)} /> )} + {creatorFilter && ( + setCreatorFilter(null)} + /> + )} + {publisherFilter && ( + setPublisherFilter(null)} + /> + )} + {unattributedOnly && ( + setUnattributedOnly(false)} + /> + )}