Files
LegacyHUB/app/storage/artifacts.py
Vadim Malanov fe346d983b
Some checks failed
CI / Backend (lint + tests + compose) (push) Has been cancelled
CI / Frontend (lint + type-check + build) (push) Has been cancelled
fix(pipeline): tolerate duplicate ORIGINAL_PDF artifacts in document lookup
Repeated knowledge-ingest of identical content (same sha256, new
asset_id -> new canonical object key) reuses the Document row but
appends a second ORIGINAL_PDF artifact, because ensure_artifact identity
is (document_id, storage_key). process_document_id then crashed with
MultipleResultsFound on scalar_one_or_none and Celery retried forever.
New latest_artifact helper picks the newest row deterministically
(created_at DESC, id DESC) — every duplicate references byte-identical
objects (sha256 verified at ingest), so the latest reference is always
safe.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-11 20:14:52 +03:00

83 lines
2.5 KiB
Python

"""Shared ``document_artifacts`` upsert helper.
Single source of truth used by the scanner, the per-document pipeline, and the
table / figure processors. Each caller previously carried its own copy with
slightly different signatures; this module replaces them so that the artifact
row schema is enforced in one place.
"""
from __future__ import annotations
import uuid
from sqlalchemy import select
from sqlalchemy.orm import Session
from app.db.models import DocumentArtifact
def ensure_artifact(
db: Session,
*,
document_id: uuid.UUID,
artifact_type: str,
bucket: str,
key: str,
page_number: int | None = None,
checksum: str | None = None,
) -> DocumentArtifact:
"""Insert a ``DocumentArtifact`` row if none exists with the same key.
Identity is ``(document_id, storage_key)``. Re-running the pipeline never
duplicates artifact rows; metadata fields are not updated in place because
derived bytes are versioned by their storage key.
"""
existing = db.execute(
select(DocumentArtifact).where(
DocumentArtifact.document_id == document_id,
DocumentArtifact.storage_key == key,
)
).scalar_one_or_none()
if existing is not None:
return existing
artifact = DocumentArtifact(
document_id=document_id,
artifact_type=artifact_type,
storage_bucket=bucket,
storage_key=key,
page_number=page_number,
checksum=checksum,
)
db.add(artifact)
return artifact
def latest_artifact(
db: Session,
*,
document_id: uuid.UUID,
artifact_type: str,
) -> DocumentArtifact | None:
"""Newest artifact of the given type, or ``None``.
Duplicates of one type are legal: repeated knowledge-ingest of identical
content (same sha256, new asset_id → new canonical object key) appends a
second ORIGINAL_PDF row because ensure_artifact identity is
(document_id, storage_key). All such rows reference byte-identical objects
(sha256 verified at ingest), so the newest reference is always a safe,
deterministic pick — never scalar_one_or_none here (G6 MultipleResultsFound).
"""
return (
db.execute(
select(DocumentArtifact)
.where(
DocumentArtifact.document_id == document_id,
DocumentArtifact.artifact_type == artifact_type,
)
.order_by(DocumentArtifact.created_at.desc(), DocumentArtifact.id.desc())
)
.scalars()
.first()
)