Repeated knowledge-ingest of identical content (same sha256, new asset_id -> new canonical object key) reuses the Document row but appends a second ORIGINAL_PDF artifact, because ensure_artifact identity is (document_id, storage_key). process_document_id then crashed with MultipleResultsFound on scalar_one_or_none and Celery retried forever. New latest_artifact helper picks the newest row deterministically (created_at DESC, id DESC) — every duplicate references byte-identical objects (sha256 verified at ingest), so the latest reference is always safe. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
83 lines
2.5 KiB
Python
83 lines
2.5 KiB
Python
"""Shared ``document_artifacts`` upsert helper.
|
|
|
|
Single source of truth used by the scanner, the per-document pipeline, and the
|
|
table / figure processors. Each caller previously carried its own copy with
|
|
slightly different signatures; this module replaces them so that the artifact
|
|
row schema is enforced in one place.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import uuid
|
|
|
|
from sqlalchemy import select
|
|
from sqlalchemy.orm import Session
|
|
|
|
from app.db.models import DocumentArtifact
|
|
|
|
|
|
def ensure_artifact(
|
|
db: Session,
|
|
*,
|
|
document_id: uuid.UUID,
|
|
artifact_type: str,
|
|
bucket: str,
|
|
key: str,
|
|
page_number: int | None = None,
|
|
checksum: str | None = None,
|
|
) -> DocumentArtifact:
|
|
"""Insert a ``DocumentArtifact`` row if none exists with the same key.
|
|
|
|
Identity is ``(document_id, storage_key)``. Re-running the pipeline never
|
|
duplicates artifact rows; metadata fields are not updated in place because
|
|
derived bytes are versioned by their storage key.
|
|
"""
|
|
existing = db.execute(
|
|
select(DocumentArtifact).where(
|
|
DocumentArtifact.document_id == document_id,
|
|
DocumentArtifact.storage_key == key,
|
|
)
|
|
).scalar_one_or_none()
|
|
if existing is not None:
|
|
return existing
|
|
|
|
artifact = DocumentArtifact(
|
|
document_id=document_id,
|
|
artifact_type=artifact_type,
|
|
storage_bucket=bucket,
|
|
storage_key=key,
|
|
page_number=page_number,
|
|
checksum=checksum,
|
|
)
|
|
db.add(artifact)
|
|
return artifact
|
|
|
|
|
|
def latest_artifact(
|
|
db: Session,
|
|
*,
|
|
document_id: uuid.UUID,
|
|
artifact_type: str,
|
|
) -> DocumentArtifact | None:
|
|
"""Newest artifact of the given type, or ``None``.
|
|
|
|
Duplicates of one type are legal: repeated knowledge-ingest of identical
|
|
content (same sha256, new asset_id → new canonical object key) appends a
|
|
second ORIGINAL_PDF row because ensure_artifact identity is
|
|
(document_id, storage_key). All such rows reference byte-identical objects
|
|
(sha256 verified at ingest), so the newest reference is always a safe,
|
|
deterministic pick — never scalar_one_or_none here (G6 MultipleResultsFound).
|
|
"""
|
|
return (
|
|
db.execute(
|
|
select(DocumentArtifact)
|
|
.where(
|
|
DocumentArtifact.document_id == document_id,
|
|
DocumentArtifact.artifact_type == artifact_type,
|
|
)
|
|
.order_by(DocumentArtifact.created_at.desc(), DocumentArtifact.id.desc())
|
|
)
|
|
.scalars()
|
|
.first()
|
|
)
|