Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 4 additions & 3 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -41,9 +41,10 @@ user row. The changes that makes possible are specified in
in audio and video come back with a timestamp you can jump straight to.
- **Clips and sub-videos** — mark a range non-destructively (no new file, always in step
with its parent), or physically extract it as a standalone asset.
- **Opt-in AI enrichment** — transcription, image description, summaries and tag
suggestions. Suggestions are reviewed, never applied silently, and a field you edited
by hand is not overwritten by a later AI run.
- **Opt-in AI enrichment** — transcription, text extraction from PDFs and Office
documents, image description, summaries and tag suggestions. Suggestions are reviewed,
never applied silently, and a field you edited by hand is not overwritten by a later AI
run.
- **AI asset creation** — generate images from images, and video from one or more
images, with the prompt and base assets recorded so a result stays reproducible.
- **Cost visibility** — an estimate before anything paid runs, and the real cost tracked
Expand Down
55 changes: 55 additions & 0 deletions backend/alembic/versions/20260916_1111_add_documentpage_table.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,55 @@
"""add documentpage table

Readable text pulled out of a document, so `summarize` and `autotag` have something to
read for a PDF or a Word file. Until now they refused: `enrichment/source.py` raised
`NoSourceMaterial` saying so out loud, because the `extract_text` job the plan lists had
never been built.

One row per page, slide, sheet or chunk rather than a column on `asset`, because
SQLModel selects every column of every row and the library listing selects assets by the
page — a 200-page PDF's text in an asset column would be read from disk to render a
thumbnail grid that never looks at it. No foreign key to `asset`, matching
`transcriptsegment`; the delete cascade is explicit in `services/assets.py`.

Revision ID: 56ac14e89a0c
Revises: c4954d1715e7
Create Date: 2026-09-16 11:11:08.578568+00:00
"""
from typing import Sequence, Union

from alembic import op
import sqlalchemy as sa
import sqlmodel


revision: str = '56ac14e89a0c'
down_revision: Union[str, None] = 'c4954d1715e7'
branch_labels: Union[str, Sequence[str], None] = None
depends_on: Union[str, Sequence[str], None] = None


def upgrade() -> None:
op.create_table('documentpage',
sa.Column('id', sqlmodel.sql.sqltypes.AutoString(), nullable=False),
sa.Column('asset_id', sqlmodel.sql.sqltypes.AutoString(), nullable=False),
sa.Column('user_id', sqlmodel.sql.sqltypes.AutoString(), nullable=False),
sa.Column('idx', sa.Integer(), nullable=False),
sa.Column('page_number', sa.Integer(), nullable=True),
sa.Column('label', sqlmodel.sql.sqltypes.AutoString(), nullable=True),
sa.Column('text', sqlmodel.sql.sqltypes.AutoString(), nullable=False),
sa.PrimaryKeyConstraint('id')
)
with op.batch_alter_table('documentpage', schema=None) as batch_op:
batch_op.create_index(batch_op.f('ix_documentpage_asset_id'), ['asset_id'], unique=False)
batch_op.create_index('ix_documentpage_asset_idx', ['asset_id', 'idx'], unique=False)
batch_op.create_index(batch_op.f('ix_documentpage_user_id'), ['user_id'], unique=False)



def downgrade() -> None:
with op.batch_alter_table('documentpage', schema=None) as batch_op:
batch_op.drop_index(batch_op.f('ix_documentpage_user_id'))
batch_op.drop_index('ix_documentpage_asset_idx')
batch_op.drop_index(batch_op.f('ix_documentpage_asset_id'))

op.drop_table('documentpage')
5 changes: 5 additions & 0 deletions backend/app/enrichment/autotag.py
Original file line number Diff line number Diff line change
Expand Up @@ -78,6 +78,11 @@ def _prompt(session: Session, asset: Asset, material: source.SourceMaterial) ->
if material.truncated:
header += " (the opening portion only)"
parts.append(f"{header}:\n\n{material.text}")
elif material.kind == source.FROM_DOCUMENT:
header = "Document text"
if material.truncated:
header += " (the opening portion only)"
parts.append(f"{header}:\n\n{material.text}")
elif material.kind == source.FROM_POSTER:
parts.append("No transcript is available; a single frame from the video is attached.")
else:
Expand Down
15 changes: 13 additions & 2 deletions backend/app/enrichment/bulk.py
Original file line number Diff line number Diff line change
Expand Up @@ -19,9 +19,15 @@

from sqlmodel import Session

from app.enrichment import autotag, describe, embed, summarize
from app.enrichment import autotag, describe, embed, extract_text, summarize
from app.models.asset import Asset
from app.models.job import KIND_AUTOTAG, KIND_DESCRIBE, KIND_EMBED, KIND_SUMMARIZE
from app.models.job import (
KIND_AUTOTAG,
KIND_DESCRIBE,
KIND_EMBED,
KIND_EXTRACT_TEXT,
KIND_SUMMARIZE,
)
from app.providers.base import ProviderUnavailable

logger = logging.getLogger(__name__)
Expand All @@ -40,6 +46,10 @@
KIND_SUMMARIZE: summarize.run,
KIND_AUTOTAG: autotag.run,
KIND_EMBED: embed.run,
# Safe to include where transcription is not: it calls nothing and bills nothing, so
# the mis-click that makes transcription too expensive to offer here costs only time.
# It is also the action most worth having in bulk — documents arrive by the folder.
KIND_EXTRACT_TEXT: extract_text.run,
}

# A selection has to be reviewable before it is run, and it is the thing standing between
Expand Down Expand Up @@ -140,4 +150,5 @@ def inner(_stage, _pct, _detail="", *, stage=stage, pct=pct, detail=detail):
KIND_SUMMARIZE: "Summarising",
KIND_AUTOTAG: "Suggesting tags",
KIND_EMBED: "Embedding",
KIND_EXTRACT_TEXT: "Reading text",
}
5 changes: 5 additions & 0 deletions backend/app/enrichment/describe.py
Original file line number Diff line number Diff line change
Expand Up @@ -71,6 +71,11 @@ def _prompt(asset: Asset, material: source.SourceMaterial) -> str:
"visual — who is on screen, the setting — which the transcript cannot "
"tell you."
)
elif material.kind == source.FROM_DOCUMENT:
header = "Text of the document"
if material.truncated:
header += " (the opening portion only — it continues beyond this)"
parts.append(f"{header}:\n\n{material.text}")
elif material.kind == source.FROM_POSTER:
parts.append(
"There is no transcript, so a single still frame from the video is "
Expand Down
Loading
Loading