Files
wyndham-ARR/arr_storage/store.py
2026-07-29 16:38:05 +08:00

498 lines
19 KiB
Python

"""Immutable staged-to-committed object workflow shared by local and cloud backends."""
from __future__ import annotations
import hashlib
import os
import stat
import tempfile
from pathlib import Path
from typing import Dict, Mapping, Optional, Tuple
from arr_ingestion.contracts import ROLE_CONTRACTS, SHA256_RE, IngestionError
from arr_storage.contracts import (
OBJECT_METADATA_SCHEMA,
BackendError,
BackendObject,
ObjectAddress,
ObjectBackend,
ObjectKeyPolicy,
StoredObject,
role_limit,
valid_delivery_filename,
)
_REQUIRED_METADATA = frozenset(
{
"arr-object-schema",
"arr-state",
"arr-role",
"arr-job-id",
"arr-attempt-no",
"arr-sha256",
"arr-byte-size",
"arr-mime-type",
}
)
class ManagedObjectStore:
"""ARR-owned immutable object store and ingestion ArtifactStore implementation."""
def __init__(self, backend: ObjectBackend, policy: Optional[ObjectKeyPolicy] = None) -> None:
self._backend = backend
self._policy = policy or ObjectKeyPolicy()
@property
def key_policy(self) -> ObjectKeyPolicy:
return self._policy
def stage_file(
self,
*,
job_id: str,
attempt_no: int,
role: str,
source: Path,
original_filename: str,
expected_sha256: Optional[str] = None,
expected_byte_size: Optional[int] = None,
) -> StoredObject:
address = ObjectAddress(job_id, attempt_no, "staged", role)
object_key = self._policy.build(address)
if not valid_delivery_filename(role, original_filename):
raise IngestionError(
"ARTIFACT_REFERENCE_INVALID", "artifact filename is invalid"
)
if expected_sha256 is not None and not SHA256_RE.fullmatch(expected_sha256):
raise IngestionError(
"ARTIFACT_REFERENCE_INVALID", "artifact hash is invalid"
)
if (
expected_byte_size is not None
and (
not isinstance(expected_byte_size, int)
or isinstance(expected_byte_size, bool)
or expected_byte_size < 0
)
):
raise IngestionError(
"ARTIFACT_REFERENCE_INVALID", "artifact size is invalid"
)
_file_kind, _extension, mime_type = ROLE_CONTRACTS[role]
with tempfile.TemporaryDirectory(prefix="arr-object-stage-") as temporary:
snapshot = Path(temporary) / "upload.bin"
sha256, byte_size = self._snapshot(source, snapshot, role_limit(role))
if expected_sha256 is not None and sha256 != expected_sha256:
raise IngestionError(
"ARTIFACT_HASH_MISMATCH", "artifact identity does not match"
)
if expected_byte_size is not None and byte_size != expected_byte_size:
raise IngestionError(
"ARTIFACT_HASH_MISMATCH", "artifact identity does not match"
)
metadata = self._metadata(address, sha256, byte_size, mime_type)
try:
self._backend.put_file(
object_key,
str(snapshot),
mime_type,
metadata,
if_absent=True,
)
except BackendError as error:
if error.kind != "conflict":
self._raise_backend(error)
existing = self.inspect(object_key, original_filename)
if (existing.sha256, existing.byte_size, existing.mime_type) != (
sha256,
byte_size,
mime_type,
):
raise IngestionError(
"ARTIFACT_CONFLICT", "object key already contains different bytes"
) from None
return existing
return self.inspect(object_key, original_filename)
def commit(self, staged: StoredObject) -> StoredObject:
self._validate_descriptor(staged, required_state="staged")
committed_address = ObjectAddress(
staged.job_id,
staged.attempt_no,
"committed",
staged.role,
)
committed_key = self._policy.build(committed_address)
try:
verified_staged = self.inspect(staged.object_key, staged.original_filename)
except IngestionError as error:
if error.code != "ARTIFACT_NOT_FOUND":
raise
committed = self.inspect(committed_key, staged.original_filename)
self._require_same_identity(staged, committed)
return committed
self._require_same_identity(staged, verified_staged)
metadata = self._metadata(
committed_address,
staged.sha256,
staged.byte_size,
staged.mime_type,
)
try:
self._backend.copy_object(
staged.object_key,
committed_key,
metadata,
if_absent=True,
)
except BackendError as error:
if error.kind != "conflict":
self._raise_backend(error)
committed = self.inspect(committed_key, staged.original_filename)
self._require_same_identity(staged, committed)
else:
committed = self.inspect(committed_key, staged.original_filename)
self._require_same_identity(staged, committed)
try:
self._backend.delete_object(staged.object_key)
except BackendError as error:
self._raise_backend(error)
return committed
def upload_committed(
self,
*,
job_id: str,
attempt_no: int,
role: str,
source: Path,
original_filename: str,
expected_sha256: Optional[str] = None,
expected_byte_size: Optional[int] = None,
) -> StoredObject:
staged = self.stage_file(
job_id=job_id,
attempt_no=attempt_no,
role=role,
source=source,
original_filename=original_filename,
expected_sha256=expected_sha256,
expected_byte_size=expected_byte_size,
)
return self.commit(staged)
def inspect(self, object_key: str, original_filename: str) -> StoredObject:
address = self._policy.parse(object_key)
if not valid_delivery_filename(address.role, original_filename):
raise IngestionError(
"ARTIFACT_REFERENCE_INVALID", "artifact filename is invalid"
)
try:
head = self._backend.head(object_key)
except BackendError as error:
self._raise_backend(error)
descriptor = self._descriptor_from_head(address, original_filename, head)
actual_sha, actual_size = self._hash_object(object_key, role_limit(address.role))
if (actual_sha, actual_size) != (descriptor.sha256, descriptor.byte_size):
raise IngestionError(
"ARTIFACT_HASH_MISMATCH", "stored object identity does not match metadata"
)
return descriptor
def inspect_committed(self, object_key: str, original_filename: str) -> StoredObject:
descriptor = self.inspect(object_key, original_filename)
if descriptor.state != "committed":
raise IngestionError(
"ARTIFACT_NOT_COMMITTED", "only committed objects may be read"
)
return descriptor
def materialize(self, object_key: str, destination: Path, max_bytes: int) -> None:
address = self._policy.parse(object_key)
if address.state != "committed":
raise IngestionError(
"ARTIFACT_NOT_COMMITTED", "only committed objects may be read"
)
if not isinstance(max_bytes, int) or isinstance(max_bytes, bool) or max_bytes < 0:
raise IngestionError("ARTIFACT_REFERENCE_INVALID", "artifact size limit is invalid")
try:
head = self._backend.head(object_key)
except BackendError as error:
self._raise_backend(error)
descriptor = self._descriptor_from_head(address, "placeholder" + ROLE_CONTRACTS[address.role][1], head)
if descriptor.byte_size > max_bytes:
raise IngestionError(
"ARTIFACT_TOO_LARGE", "delivery artifact exceeds its size limit"
)
destination.parent.mkdir(parents=True, exist_ok=True, mode=0o700)
output_descriptor = -1
destination_created = False
copied = 0
digest = hashlib.sha256()
try:
output_descriptor = os.open(
destination,
os.O_WRONLY | os.O_CREAT | os.O_EXCL,
0o600,
)
destination_created = True
with self._open_reader(object_key) as reader, os.fdopen(output_descriptor, "wb") as target:
output_descriptor = -1
while True:
chunk = self._read_chunk(reader)
if not chunk:
break
if not isinstance(chunk, bytes):
raise IngestionError(
"OBJECT_STORE_UNAVAILABLE", "object storage read did not complete"
)
copied += len(chunk)
if copied > max_bytes:
raise IngestionError(
"ARTIFACT_TOO_LARGE", "delivery artifact exceeds its size limit"
)
digest.update(chunk)
target.write(chunk)
target.flush()
os.fsync(target.fileno())
if (copied, digest.hexdigest()) != (descriptor.byte_size, descriptor.sha256):
raise IngestionError(
"ARTIFACT_HASH_MISMATCH", "stored object identity does not match metadata"
)
except FileExistsError:
raise IngestionError(
"ARTIFACT_CONFLICT", "materialization destination already exists"
) from None
except Exception:
if destination_created:
destination.unlink(missing_ok=True)
raise
finally:
if output_descriptor >= 0:
os.close(output_descriptor)
def _descriptor_from_head(
self,
address: ObjectAddress,
original_filename: str,
head: BackendObject,
) -> StoredObject:
if head.object_key != self._policy.build(address):
raise IngestionError("ARTIFACT_REFERENCE_INVALID", "stored object key is invalid")
metadata = dict(head.metadata)
if not _REQUIRED_METADATA.issubset(metadata):
raise IngestionError(
"ARTIFACT_METADATA_INVALID", "stored object metadata is incomplete"
)
_file_kind, _extension, expected_mime = ROLE_CONTRACTS[address.role]
expected_values: Mapping[str, str] = {
"arr-object-schema": OBJECT_METADATA_SCHEMA,
"arr-state": address.state,
"arr-role": address.role,
"arr-job-id": address.job_id,
"arr-attempt-no": str(address.attempt_no),
"arr-mime-type": expected_mime,
}
if any(metadata.get(key) != value for key, value in expected_values.items()):
raise IngestionError(
"ARTIFACT_METADATA_INVALID", "stored object metadata conflicts with its key"
)
sha256 = metadata.get("arr-sha256", "")
size_text = metadata.get("arr-byte-size", "")
if not SHA256_RE.fullmatch(sha256) or not size_text.isdigit():
raise IngestionError(
"ARTIFACT_METADATA_INVALID", "stored object identity metadata is invalid"
)
byte_size = int(size_text)
if byte_size != head.byte_size or byte_size > role_limit(address.role):
raise IngestionError(
"ARTIFACT_METADATA_INVALID", "stored object size metadata is invalid"
)
backend_content_type = metadata.get("arr-backend-content-type")
if backend_content_type is not None and backend_content_type != expected_mime:
raise IngestionError(
"ARTIFACT_METADATA_INVALID", "stored object content type is invalid"
)
return StoredObject(
object_key=head.object_key,
job_id=address.job_id,
attempt_no=address.attempt_no,
state=address.state,
role=address.role,
original_filename=original_filename,
sha256=sha256,
byte_size=byte_size,
mime_type=expected_mime,
etag=head.etag,
version_id=head.version_id,
)
@staticmethod
def _metadata(
address: ObjectAddress,
sha256: str,
byte_size: int,
mime_type: str,
) -> Dict[str, str]:
return {
"arr-object-schema": OBJECT_METADATA_SCHEMA,
"arr-state": address.state,
"arr-role": address.role,
"arr-job-id": address.job_id,
"arr-attempt-no": str(address.attempt_no),
"arr-sha256": sha256,
"arr-byte-size": str(byte_size),
"arr-mime-type": mime_type,
}
@staticmethod
def _snapshot(source: Path, destination: Path, max_bytes: int) -> Tuple[str, int]:
source_descriptor = -1
destination_descriptor = -1
copied = 0
digest = hashlib.sha256()
try:
if source.is_symlink():
raise IngestionError(
"ARTIFACT_REFERENCE_INVALID", "upload source must be a regular file"
)
source_descriptor = os.open(
source,
os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0),
)
if not stat.S_ISREG(os.fstat(source_descriptor).st_mode):
raise IngestionError(
"ARTIFACT_REFERENCE_INVALID", "upload source must be a regular file"
)
destination_descriptor = os.open(
destination,
os.O_WRONLY | os.O_CREAT | os.O_EXCL,
0o600,
)
with os.fdopen(source_descriptor, "rb") as reader, os.fdopen(
destination_descriptor, "wb"
) as target:
source_descriptor = -1
destination_descriptor = -1
while True:
chunk = reader.read(1024 * 1024)
if not chunk:
break
copied += len(chunk)
if copied > max_bytes:
raise IngestionError(
"ARTIFACT_TOO_LARGE", "delivery artifact exceeds its size limit"
)
digest.update(chunk)
target.write(chunk)
target.flush()
os.fsync(target.fileno())
except FileNotFoundError:
destination.unlink(missing_ok=True)
raise IngestionError(
"ARTIFACT_NOT_FOUND", "upload source is unavailable"
) from None
except OSError:
destination.unlink(missing_ok=True)
raise IngestionError(
"OBJECT_STORE_UNAVAILABLE", "object storage upload did not complete"
) from None
except Exception:
destination.unlink(missing_ok=True)
raise
finally:
if source_descriptor >= 0:
os.close(source_descriptor)
if destination_descriptor >= 0:
os.close(destination_descriptor)
return digest.hexdigest(), copied
def _hash_object(self, object_key: str, max_bytes: int) -> Tuple[str, int]:
copied = 0
digest = hashlib.sha256()
with self._open_reader(object_key) as reader:
while True:
chunk = self._read_chunk(reader)
if not chunk:
break
if not isinstance(chunk, bytes):
raise IngestionError(
"OBJECT_STORE_UNAVAILABLE", "object storage read did not complete"
)
copied += len(chunk)
if copied > max_bytes:
raise IngestionError(
"ARTIFACT_TOO_LARGE", "delivery artifact exceeds its size limit"
)
digest.update(chunk)
return digest.hexdigest(), copied
def _open_reader(self, object_key: str):
try:
return self._backend.open_reader(object_key)
except BackendError as error:
self._raise_backend(error)
@staticmethod
def _read_chunk(reader) -> bytes:
try:
return reader.read(1024 * 1024)
except BackendError as error:
ManagedObjectStore._raise_backend(error)
except OSError:
raise IngestionError(
"OBJECT_STORE_UNAVAILABLE", "object storage read did not complete"
) from None
def _validate_descriptor(self, descriptor: StoredObject, required_state: str) -> None:
address = self._policy.parse(descriptor.object_key)
if (
address.job_id != descriptor.job_id
or address.attempt_no != descriptor.attempt_no
or address.state != descriptor.state
or address.role != descriptor.role
or descriptor.state != required_state
or not valid_delivery_filename(descriptor.role, descriptor.original_filename)
or not SHA256_RE.fullmatch(descriptor.sha256)
or descriptor.byte_size < 0
or descriptor.mime_type != ROLE_CONTRACTS[descriptor.role][2]
):
raise IngestionError(
"ARTIFACT_REFERENCE_INVALID", "stored object descriptor is invalid"
)
@staticmethod
def _require_same_identity(left: StoredObject, right: StoredObject) -> None:
if (
left.job_id,
left.attempt_no,
left.role,
left.sha256,
left.byte_size,
left.mime_type,
) != (
right.job_id,
right.attempt_no,
right.role,
right.sha256,
right.byte_size,
right.mime_type,
):
raise IngestionError(
"ARTIFACT_CONFLICT", "stored object identity conflicts"
)
@staticmethod
def _raise_backend(error: BackendError) -> None:
if error.kind == "not_found":
raise IngestionError("ARTIFACT_NOT_FOUND", "stored object is unavailable") from None
if error.kind == "conflict":
raise IngestionError("ARTIFACT_CONFLICT", "object key already exists") from None
if error.kind == "invalid_request":
raise IngestionError("ARTIFACT_REFERENCE_INVALID", "object request is invalid") from None
raise IngestionError(
"OBJECT_STORE_UNAVAILABLE", "object storage operation did not complete"
) from None