Files
2026-08-16 03:57:24 +02:00

491 lines
17 KiB
Python

#!/usr/bin/env python3
"""
attach_images_to_partdb.py
==========================
Pushes the images downloaded by aliexpress_to_partdb.py up to Part-DB
and attaches them to the matching parts via the Part-DB REST API
(API Platform / JSON-LD).
How it works
------------
1. Reads ``partdb_images/attachments.json`` (the file the first script
wrote).
2. For each entry, queries Part-DB for a part whose ``name`` matches
the ``part_name`` recorded in the JSON.
3. Uploads the local image as a PartAttachment to the matched part.
The script is idempotent: if the part already has an attachment
with the same filename, the upload is skipped. Safe to re-run.
Usage
-----
# Just run it - URL and token are baked in as defaults:
python attach_images_to_partdb.py
# Test with the first 3 entries first (no changes are made):
python attach_images_to_partdb.py --limit 3 --dry-run
# Or override the defaults via env vars or CLI flags:
set PARTDB_URL=https://partdb.example.com
set PARTDB_TOKEN=tcp_xxx
python attach_images_to_partdb.py
# If your Part-DB uses a self-signed HTTPS cert
python attach_images_to_partdb.py --insecure
Dependencies
------------
pip install requests
"""
from __future__ import annotations
import argparse
import base64
import json
import mimetypes
import os
import sys
import time
from pathlib import Path
try:
import requests
except ImportError:
sys.stderr.write(
"ERROR: the 'requests' package is required.\n"
" install it with: pip install requests\n"
)
sys.exit(1)
# ---------------------------------------------------------------------------
# SECURITY WARNING
# ---------------------------------------------------------------------------
# This file contains an API token. Treat it like a password:
# - Do not commit it to git, share it, or upload it to cloud backups.
# - When you're done with the import, delete this file OR rotate the
# token in Part-DB (User menu -> Settings -> API tokens -> delete).
# - For better security, set PARTDB_URL and PARTDB_TOKEN as environment
# variables and remove the values from this file.
# ---------------------------------------------------------------------------
DEFAULT_PARTDB_URL = "http://192.168.64.80:8085"
DEFAULT_PARTDB_TOKEN = "tcp_047767e7659f85ce382971d2d7f439d3cfdd011407935edb8c60c746e55ad20b"
DEFAULT_ATTACHMENTS = "partdb_images/attachments.json"
DEFAULT_DELAY = 0.15 # seconds between API calls; be polite to the server
# ---------------------------------------------------------------------------
# API helpers
# ---------------------------------------------------------------------------
def api_get(session: requests.Session, base: str, endpoint: str, **params):
url = f"{base.rstrip('/')}{endpoint}"
r = session.get(url, params=params, timeout=30)
r.raise_for_status()
return r.json()
def api_post(session: requests.Session, base: str, endpoint: str, payload: dict):
url = f"{base.rstrip('/')}{endpoint}"
r = session.post(url, json=payload, timeout=60)
return r
def extract_member(data) -> list:
"""API Platform returns either a plain list or a Hydra/JSON-LD
collection ({'member': [...]}). Return just the items."""
if isinstance(data, list):
return data
if isinstance(data, dict):
return data.get("member") or data.get("hydra:member") or []
return []
def find_part_by_name(
session: requests.Session,
base: str,
name: str,
) -> list[dict]:
"""Return the list of parts whose name exactly matches ``name``."""
# 1) exact filter
try:
data = api_get(session, base, "/api/parts", **{"name": name, "itemsPerPage": 50})
members = extract_member(data)
if members:
return members
except requests.HTTPError:
pass
# 2) fallback: full-text search
try:
data = api_get(session, base, "/api/parts", **{"search": name, "itemsPerPage": 50})
members = extract_member(data)
return [p for p in members if (p.get("name") or "").strip() == name.strip()]
except requests.HTTPError:
return []
def part_id_from_iri(iri: str):
if not iri:
return None
try:
return int(iri.rstrip("/").rsplit("/", 1)[-1])
except (ValueError, IndexError):
return None
# ---------------------------------------------------------------------------
# Attachment type discovery
# ---------------------------------------------------------------------------
# Priority list of names to look for when auto-picking an attachment type
# for product images. Match is case-insensitive substring.
_ATTACHMENT_TYPE_PRIORITY = [
"image", "photo", "picture", "product image", "product photo",
"preview", "thumbnail", "datasheet image", "foto", "slika",
]
def fetch_attachment_types(session, base):
"""Return the list of attachment types as dicts from Part-DB."""
try:
data = api_get(session, base, "/api/attachment_types", itemsPerPage=200)
except requests.HTTPError as exc:
sys.stderr.write(f" ! could not list attachment types: {exc}\n")
return []
return extract_member(data)
def pick_default_attachment_type(types):
"""Pick a sensible attachment type for product images.
Prefers a type whose name contains any of the priority keywords;
otherwise falls back to the first type returned by Part-DB."""
if not types:
return None
by_name = {(t.get("name") or "").strip().lower(): t for t in types}
for kw in _ATTACHMENT_TYPE_PRIORITY:
for name, t in by_name.items():
if kw in name:
return t
return types[0]
def type_iri(t):
if t is None:
return None
iri = t.get("@id")
if iri:
return iri
tid = t.get("id")
if tid is not None:
return f"/api/attachment_types/{tid}"
return None
# ---------------------------------------------------------------------------
# Attachment helpers
# ---------------------------------------------------------------------------
def already_attached(part: dict, filename: str) -> bool:
"""Best-effort check: if the part has any embedded attachment with
the same filename, treat as already done."""
atts = part.get("attachments")
if isinstance(atts, list):
for a in atts:
if not isinstance(a, dict):
continue
if a.get("filename") == filename or a.get("name") == filename:
return True
return False
def upload_attachment(
session: requests.Session,
base: str,
part_id: int,
image_path: Path,
attachment_type_iri: str,
dry_run: bool = False,
) -> tuple[bool, str]:
"""Upload image_path as a PartAttachment for the given part ID."""
if not image_path.is_file():
return False, f"file missing on disk: {image_path}"
file_bytes = image_path.read_bytes()
b64 = base64.b64encode(file_bytes).decode("ascii")
mime, _ = mimetypes.guess_type(str(image_path))
if not mime:
mime = "image/jpeg"
payload_b64 = f"data:@{mime};base64,{b64}"
payload = {
"name": image_path.name,
"attachment_type": attachment_type_iri,
"element": f"/api/parts/{part_id}",
"upload": {
"data": payload_b64,
"filename": image_path.name,
"private": False,
},
}
if dry_run:
return True, (
f"[dry-run] would POST {len(file_bytes):>6} bytes ({mime}, "
f"type={attachment_type_iri}) to /api/attachments for part {part_id}"
)
r = api_post(session, base, "/api/attachments", payload)
if r.status_code in (200, 201):
return True, f"attached to part {part_id}"
try:
err = r.json()
# 422 errors come back as a dict of {field: [messages]}
if isinstance(err, dict):
for k, v in err.items():
if isinstance(v, list) and v:
return False, f"HTTP {r.status_code} ({k}): {v[0]}"
msg = (
err.get("detail")
or err.get("hydra:description")
or err.get("message")
or str(err)[:200]
)
else:
msg = str(err)[:200]
except Exception:
msg = (r.text or "")[:200] or f"HTTP {r.status_code}"
return False, f"HTTP {r.status_code}: {msg}"
# ---------------------------------------------------------------------------
# Main
# ---------------------------------------------------------------------------
def main(argv=None) -> int:
p = argparse.ArgumentParser(
description="Attach images to Part-DB parts via the REST API.",
formatter_class=argparse.RawDescriptionHelpFormatter,
)
p.add_argument(
"--url",
default=os.environ.get("PARTDB_URL", DEFAULT_PARTDB_URL),
help="Base URL of your Part-DB instance (no trailing slash). "
"Defaults to the value baked into this file; "
"can also be set via the PARTDB_URL env var.",
)
p.add_argument(
"--token",
default=os.environ.get("PARTDB_TOKEN", DEFAULT_PARTDB_TOKEN),
help="Part-DB API token (Bearer). Defaults to the value baked into "
"this file; can also be set via the PARTDB_TOKEN env var.",
)
p.add_argument(
"--attachments",
default=DEFAULT_ATTACHMENTS,
help=f"Path to attachments.json (default: {DEFAULT_ATTACHMENTS})",
)
p.add_argument(
"--delay", type=float, default=DEFAULT_DELAY,
help=f"Seconds to wait between API calls (default: {DEFAULT_DELAY})",
)
p.add_argument(
"--limit", type=int, default=0,
help="Only process the first N entries (0 = all).",
)
p.add_argument(
"--dry-run", action="store_true",
help="Report what would happen, but make no changes.",
)
p.add_argument(
"--insecure", action="store_true",
help="Skip TLS certificate verification (self-signed certs).",
)
p.add_argument(
"--attachment-type",
default="",
help="Attachment type IRI to use, e.g. /api/attachment_types/1. "
"If omitted, the script auto-picks one whose name looks like "
"'Image' / 'Photo' / etc., falling back to the first available.",
)
p.add_argument(
"--list-attachment-types", action="store_true",
help="List all attachment types defined in Part-DB and exit.",
)
args = p.parse_args(argv)
if not args.url:
sys.stderr.write(
"ERROR: --url or PARTDB_URL env var is required.\n"
" Example: --url https://partdb.example.com\n"
)
return 1
if not args.token:
sys.stderr.write(
"ERROR: --token or PARTDB_TOKEN env var is required.\n"
" Get one in Part-DB: User menu -> Settings -> API tokens.\n"
)
return 1
session = requests.Session()
session.headers.update({
"Authorization": f"Bearer {args.token}",
"Accept": "application/json",
"Content-Type": "application/json",
})
if args.insecure:
session.verify = False
import urllib3
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
# Sanity check: can we reach the API?
try:
api_get(session, args.url, "/api/parts", itemsPerPage=1)
sys.stderr.write(f"Connected to {args.url}\n")
except requests.HTTPError as exc:
if exc.response is not None and exc.response.status_code == 401:
sys.stderr.write(
"ERROR: 401 Unauthorized.\n"
" - The token is wrong/expired, OR\n"
" - Your user doesn't have the 'API' permission enabled.\n"
" Fix: User menu -> Settings -> API tokens (create a new one),\n"
" and ask your admin to enable 'API access' on your user.\n"
)
return 1
sys.stderr.write(
f"ERROR: HTTP {exc.response.status_code if exc.response else '?'} "
f"from Part-DB\n"
)
return 1
except requests.RequestException as exc:
sys.stderr.write(f"ERROR: cannot reach Part-DB at {args.url}: {exc}\n")
return 1
att_path = Path(args.attachments)
if not att_path.is_file():
sys.stderr.write(f"ERROR: attachments file not found: {att_path}\n")
return 1
try:
entries = json.loads(att_path.read_text(encoding="utf-8"))
except json.JSONDecodeError as exc:
sys.stderr.write(f"ERROR: could not parse {att_path}: {exc}\n")
return 1
# --list-attachment-types: print all types and exit
if args.list_attachment_types:
types = fetch_attachment_types(session, args.url)
if not types:
sys.stderr.write("No attachment types returned by Part-DB.\n")
return 1
sys.stderr.write(f"Available attachment types ({len(types)}):\n")
for t in types:
sys.stderr.write(
f" id={t.get('id'):<4} iri={type_iri(t):<40} "
f"name={t.get('name')!r}\n"
)
return 0
# Resolve attachment_type IRI (explicit > auto-pick)
if args.attachment_type:
attachment_type_iri = args.attachment_type
sys.stderr.write(f"Using attachment_type (from --attachment-type): {attachment_type_iri}\n")
else:
types = fetch_attachment_types(session, args.url)
if not types:
sys.stderr.write(
"ERROR: Part-DB returned no attachment types. Define at least "
"one in the admin panel (Admin -> Attachment types) or pass "
"--attachment-type /api/attachment_types/<id> explicitly.\n"
)
return 1
chosen = pick_default_attachment_type(types)
attachment_type_iri = type_iri(chosen)
sys.stderr.write(
f"Using attachment_type (auto-picked): id={chosen.get('id')} "
f"name={chosen.get('name')!r} iri={attachment_type_iri}\n"
)
if not attachment_type_iri:
sys.stderr.write("ERROR: could not derive attachment_type IRI.\n")
return 1
if args.limit > 0:
entries = entries[: args.limit]
sys.stderr.write(f"Loaded {len(entries)} attachment entries from {att_path}\n")
sys.stderr.write(f"Mode: {'DRY-RUN' if args.dry_run else 'LIVE'}\n")
sys.stderr.write("-" * 70 + "\n")
succeeded = skipped = failed = no_match = 0
name_cache: dict[str, list[dict]] = {}
for idx, entry in enumerate(entries, start=1):
part_name = (entry.get("part_name") or "").strip()
files = entry.get("files") or []
if not part_name or not files:
continue
if part_name in name_cache:
parts = name_cache[part_name]
else:
try:
parts = find_part_by_name(session, args.url, part_name)
except requests.RequestException as exc:
sys.stderr.write(
f"[{idx}/{len(entries)}] {part_name[:50]!r}... "
f"API error: {exc}\n"
)
failed += len(files)
continue
name_cache[part_name] = parts
if not parts:
sys.stderr.write(
f"[{idx:>4}/{len(entries)}] {part_name[:55]:<55} no match in Part-DB\n"
)
no_match += len(files)
continue
for part in parts:
part_id = part.get("id") or part_id_from_iri(part.get("@id") or "")
if part_id is None:
continue
for image_path_str in files:
image_path = Path(image_path_str)
if already_attached(part, image_path.name):
sys.stderr.write(
f"[{idx:>4}/{len(entries)}] {part_name[:40]:<40} "
f"part {part_id}: already has {image_path.name}\n"
)
skipped += 1
continue
ok, msg = upload_attachment(
session, args.url, part_id, image_path,
attachment_type_iri, dry_run=args.dry_run,
)
tag = "OK " if ok else "ERR"
sys.stderr.write(
f"[{idx:>4}/{len(entries)}] {part_name[:40]:<40} "
f"{tag} part {part_id}: {msg}\n"
)
if ok:
succeeded += 1
else:
failed += 1
time.sleep(args.delay)
sys.stderr.write("-" * 70 + "\n")
sys.stderr.write("Done.\n")
sys.stderr.write(f" Uploaded: {succeeded}\n")
sys.stderr.write(f" Skipped: {skipped} (already attached)\n")
sys.stderr.write(f" No match: {no_match} (part not found in Part-DB)\n")
sys.stderr.write(f" Failed: {failed}\n")
if failed:
sys.stderr.write(
"\nTip: re-run the same command to retry only the failed ones - "
"successful uploads are skipped on re-run.\n"
)
return 0 if failed == 0 else 2
if __name__ == "__main__":
sys.exit(main())