Files
archive-clients/src/archive_clients/resources.py
T

258 lines
9.7 KiB
Python

"""Normalize qBittorrent observations into the stable protocol model."""
from __future__ import annotations
import hashlib
import json
from dataclasses import dataclass
from datetime import datetime, timezone
from pathlib import PurePosixPath
from typing import Any
from archive_clients.bencode import Metainfo, decode_metainfo
from archive_control.v1 import resource_pb2
class ResourceError(ValueError):
pass
@dataclass(frozen=True)
class NormalizedResource:
"""A normalized qB observation for protocol data and local file work.
``save_path`` is qBittorrent's API-visible per-torrent content root. It
is deliberately local-only: clients resolve it through their qB path
mapping immediately before filesystem work, and it is never serialized in
inventory, placements, or control protocol messages.
"""
summary: resource_pb2.ResourceSummary
files: tuple[resource_pb2.TorrentFile, ...]
metainfo: Metainfo
metainfo_bytes: bytes = b""
# qBittorrent's API-visible save path is intentionally local-only. It
# must never become part of inventory or placement protocol messages.
save_path: PurePosixPath | None = None
def build_content_tree(
files: tuple[resource_pb2.TorrentFile, ...],
available_file_indices: set[int],
) -> list[resource_pb2.ContentTreeEntry]:
directories: dict[str, set[int]] = {}
entries: list[resource_pb2.ContentTreeEntry] = []
by_index = {file.file_index: file for file in files}
for file in files:
path = PurePosixPath(file.canonical_path)
for parent in path.parents:
if parent == PurePosixPath("."):
continue
directories.setdefault(parent.as_posix(), set()).add(file.file_index)
entry = resource_pb2.ContentTreeEntry(
canonical_path=file.canonical_path,
parent_path=(
"" if path.parent == PurePosixPath(".")
else path.parent.as_posix()
),
display_name=path.name,
type=(
resource_pb2.CONTENT_ENTRY_TYPE_PADDING_FILE
if file.padding else resource_pb2.CONTENT_ENTRY_TYPE_FILE
),
file_index=file.file_index,
)
entry.file.CopyFrom(file)
if file.file_index in available_file_indices:
_set_selection(entry.available_file_indices, [file.file_index])
entry.available_logical_bytes = file.logical_bytes
entry.available_file_count = 1
entries.append(entry)
for path_string, descendant_indices in directories.items():
path = PurePosixPath(path_string)
available = sorted(descendant_indices & available_file_indices)
entry = resource_pb2.ContentTreeEntry(
canonical_path=path_string,
parent_path=(
"" if path.parent == PurePosixPath(".")
else path.parent.as_posix()
),
display_name=path.name,
type=resource_pb2.CONTENT_ENTRY_TYPE_DIRECTORY,
available_logical_bytes=sum(
by_index[index].logical_bytes for index in available
),
available_file_count=len(available),
)
_set_selection(entry.available_file_indices, available)
entries.append(entry)
return sorted(
entries,
key=lambda entry: (
tuple(PurePosixPath(entry.canonical_path).parts),
entry.type != resource_pb2.CONTENT_ENTRY_TYPE_DIRECTORY,
),
)
def normalize_resource(
torrent: dict[str, Any],
raw_files: list[dict[str, Any]],
metainfo_bytes: bytes,
observed_at: datetime | None = None,
) -> NormalizedResource:
metainfo = decode_metainfo(metainfo_bytes)
# qBittorrent/libtorrent may omit torrent padding files from
# /torrents/files while keeping them in the exported metainfo.
visible_metainfo_files = tuple(
item for item in metainfo.files if not item.padding
)
if len(raw_files) == len(metainfo.files):
matched_metainfo_files = metainfo.files
elif len(raw_files) == len(visible_metainfo_files):
matched_metainfo_files = visible_metainfo_files
else:
raise ResourceError("qBittorrent and metainfo file counts differ")
files = []
canonical = True
for expected_index, (raw, meta_file) in enumerate(
zip(raw_files, matched_metainfo_files, strict=True)
):
index = _integer(raw.get("index"), "file index")
if index != expected_index:
raise ResourceError("qBittorrent file indices are not contiguous")
path = _path(raw.get("name"))
size = _integer(raw.get("size"), "file size")
if size != meta_file.logical_bytes or path != meta_file.path:
canonical = False
completed = raw.get("completed")
if completed is None:
progress = raw.get("progress", 0)
if isinstance(progress, bool) or not isinstance(progress, (int, float)):
raise ResourceError("file progress is invalid")
completed = round(size * max(0.0, min(float(progress), 1.0)))
completed = min(_integer(completed, "completed bytes"), size)
priority = _integer(raw.get("priority", 0), "file priority")
file = resource_pb2.TorrentFile(
file_index=index,
canonical_path=path,
logical_bytes=size,
completed_bytes=completed,
selected=priority > 0,
padding=meta_file.padding,
)
files.append(file)
observed_at = observed_at or datetime.now(timezone.utc)
qb_torrent_id = _string(torrent.get("hash"), "torrent hash").lower()
if len(qb_torrent_id) not in {40, 64} or any(
character not in "0123456789abcdef" for character in qb_torrent_id
):
raise ResourceError("torrent hash is invalid")
save_path = _save_path(torrent.get("save_path"))
summary = resource_pb2.ResourceSummary(
qb_torrent_id=qb_torrent_id,
display_name=_string(torrent.get("name"), "torrent name"),
runtime_state=_runtime_state(_string(torrent.get("state"), "state")),
total_logical_bytes=sum(file.logical_bytes for file in files),
total_file_count=len(files),
canonical_paths=canonical,
)
summary.resource_id.info_hash_v1_hex = metainfo.info_hash_v1_hex
summary.resource_id.info_hash_v2_hex = metainfo.info_hash_v2_hex
selected = [file.file_index for file in files if file.selected]
complete = [
file.file_index for file in files
if file.selected and file.completed_bytes == file.logical_bytes
]
_set_selection(summary.selected_files, selected)
_set_selection(summary.selected_complete_files, complete)
summary.selected_logical_bytes = sum(
file.logical_bytes for file in files if file.selected
)
summary.selected_complete_bytes = sum(
file.logical_bytes for file in files
if file.selected and file.completed_bytes == file.logical_bytes
)
revision_data = [{
"index": file.file_index,
"path": file.canonical_path,
"size": file.logical_bytes,
"completed": file.completed_bytes,
"selected": file.selected,
} for file in files]
summary.content_revision = hashlib.sha256(json.dumps(
revision_data, sort_keys=True, separators=(",", ":"),
).encode("utf-8")).hexdigest()
summary.observed_at.FromDatetime(observed_at)
return NormalizedResource(
summary, tuple(files), metainfo, metainfo_bytes, save_path
)
def _set_selection(target: Any, indices: list[int]) -> None:
if not indices:
return
first = previous = indices[0]
for index in indices[1:]:
if index == previous + 1:
previous = index
continue
target.ranges.add(first=first, last=previous)
first = previous = index
target.ranges.add(first=first, last=previous)
def _runtime_state(state: str) -> int:
lowered = state.lower()
if "missing" in lowered:
return resource_pb2.TORRENT_RUNTIME_STATE_MISSING
if "error" in lowered:
return resource_pb2.TORRENT_RUNTIME_STATE_ERROR
if "check" in lowered:
return resource_pb2.TORRENT_RUNTIME_STATE_CHECKING
if "stalled" in lowered:
return resource_pb2.TORRENT_RUNTIME_STATE_STALLED
if "pause" in lowered or "stop" in lowered:
return resource_pb2.TORRENT_RUNTIME_STATE_STOPPED
if "queue" in lowered:
return resource_pb2.TORRENT_RUNTIME_STATE_QUEUED
if "download" in lowered or lowered in {"metadl", "forceddl", "allocating"}:
return resource_pb2.TORRENT_RUNTIME_STATE_DOWNLOADING
if lowered.endswith("up") or "upload" in lowered:
return resource_pb2.TORRENT_RUNTIME_STATE_SEEDING
return resource_pb2.TORRENT_RUNTIME_STATE_UNSPECIFIED
def _path(value: Any) -> str:
path = _string(value, "file path")
candidate = PurePosixPath(path)
if candidate.is_absolute() or ".." in candidate.parts or "\\" in path:
raise ResourceError("qBittorrent file path is unsafe")
return candidate.as_posix()
def _save_path(value: Any) -> PurePosixPath:
"""Validate qBittorrent's API-visible per-torrent content root."""
path = _string(value, "torrent save path")
candidate = PurePosixPath(path)
if (
not candidate.is_absolute()
or ".." in candidate.parts
or "." in candidate.parts
):
raise ResourceError("qBittorrent torrent save path is unsafe")
return candidate
def _integer(value: Any, name: str) -> int:
if isinstance(value, bool) or not isinstance(value, int) or value < 0:
raise ResourceError(f"{name} is invalid")
return value
def _string(value: Any, name: str) -> str:
if not isinstance(value, str) or not value:
raise ResourceError(f"{name} is invalid")
return value