Files
Netbox-Documentation/netbox_documentation/archive.py
T

315 lines
15 KiB
Python

import json
from pathlib import Path, PurePosixPath
from tempfile import SpooledTemporaryFile
from zipfile import ZIP_DEFLATED, BadZipFile, ZipFile
from django.conf import settings
from django.contrib.contenttypes.models import ContentType
from django.core.exceptions import ValidationError
from django.db import transaction
from django.utils.text import slugify
from tenancy.models import Tenant, TenantGroup
from .models import Document, DocumentAssignment, DocumentAttachment, DocumentCategory
ARCHIVE_FORMAT = "netbox-documentation"
ARCHIVE_VERSION = 1
MAX_ENTRIES = 5000
class ArchiveFailure(ValueError):
pass
def _save_imported_object(instance, context, **save_kwargs):
try:
instance.save(**save_kwargs)
except ValidationError as exc:
messages = getattr(exc, "messages", None) or [str(exc)]
raise ArchiveFailure(f"{context}: {'; '.join(messages)}") from exc
def documents_for_categories(categories, include_subfolders=True, user=None):
"""Return distinct permitted documents contained in the selected category trees."""
category_ids = set(categories.values_list("pk", flat=True))
frontier = set(category_ids)
while include_subfolders and frontier:
children = set(DocumentCategory.objects.filter(parent_id__in=frontier).values_list("pk", flat=True))
frontier = children - category_ids
category_ids.update(children)
documents = Document.objects.filter(category_id__in=category_ids).distinct().order_by("title")
if user is not None:
documents = documents.restrict(user, "view")
return documents
def _category_path(category):
path, seen = [], set()
while category and category.pk not in seen:
path.append(category.name)
seen.add(category.pk)
category = category.parent
return list(reversed(path))
def _category_tenancy_path(category):
path, seen = [], set()
while category and category.pk not in seen:
path.append({
"tenant": category.tenant.slug if category.tenant else None,
"tenant_group": category.tenant_group.slug if category.tenant_group else None,
})
seen.add(category.pk)
category = category.parent
return list(reversed(path))
def export_documents(documents):
stream = SpooledTemporaryFile(max_size=10 * 1024 * 1024, mode="w+b")
manifest = {"format": ARCHIVE_FORMAT, "version": ARCHIVE_VERSION, "documents": []}
with ZipFile(stream, "w", compression=ZIP_DEFLATED, compresslevel=6) as archive:
for document in documents.prefetch_related("assignments__assigned_object_type", "attachments").select_related(
"category", "tenant", "tenant_group", "category__tenant", "category__tenant_group",
):
root = f"documents/{document.pk}"
extension = "html" if document.body_format == "html" else "md"
body_path = f"{root}/content.{extension}"
archive.writestr(body_path, document.body.encode("utf-8"))
attachments = []
for attachment in document.attachments.all():
safe_name = Path(attachment.original_name).name or f"attachment-{attachment.pk}"
member = f"{root}/attachments/{attachment.pk}-{safe_name}"
try:
with attachment.file.open("rb") as source:
archive.writestr(member, source.read())
except (FileNotFoundError, OSError):
continue
attachments.append({
"path": member, "original_name": safe_name,
"content_type": attachment.content_type, "size": attachment.size,
"source_url": attachment.file.url,
})
assignments = [{
"app_label": item.assigned_object_type.app_label,
"model": item.assigned_object_type.model,
"object_id": item.assigned_object_id,
"note": item.note,
} for item in document.assignments.all()]
manifest["documents"].append({
"title": document.title, "slug": document.slug, "summary": document.summary,
"body_format": document.body_format, "is_published": document.is_published,
"category_path": _category_path(document.category), "body_path": body_path,
"category_tenancy": _category_tenancy_path(document.category),
"tenant": document.tenant.slug if document.tenant else None,
"tenant_group": document.tenant_group.slug if document.tenant_group else None,
"assignments": assignments, "attachments": attachments,
})
archive.writestr("manifest.json", json.dumps(manifest, ensure_ascii=False, indent=2).encode("utf-8"))
stream.seek(0)
return stream
def _validate_archive(archive):
infos = archive.infolist()
if len(infos) > MAX_ENTRIES:
raise ArchiveFailure(f"Das Archiv enthält mehr als {MAX_ENTRIES} Dateien.")
limit = settings.PLUGINS_CONFIG.get("netbox_documentation", {}).get("max_archive_size_mb", 250) * 1024 * 1024
if sum(info.file_size for info in infos) > limit:
raise ArchiveFailure("Die entpackte Gesamtgröße überschreitet das konfigurierte Limit.")
for info in infos:
path = PurePosixPath(info.filename)
if path.is_absolute() or ".." in path.parts:
raise ArchiveFailure("Das Archiv enthält einen unsicheren Dateipfad.")
if info.compress_size and info.file_size / info.compress_size > 200:
raise ArchiveFailure("Das Archiv enthält eine verdächtig stark komprimierte Datei.")
def _read_json(archive, name):
try:
return json.loads(archive.read(name).decode("utf-8"))
except (KeyError, UnicodeDecodeError, json.JSONDecodeError) as exc:
raise ArchiveFailure("Das Archiv enthält kein gültiges manifest.json.") from exc
def _resolve_tenant_group(slug, context):
"""Look up a tenant group by slug for an imported record.
Returns None if no slug was recorded (i.e. the object genuinely had no
tenant group). Raises ArchiveFailure if a slug *was* recorded but no
matching TenantGroup exists on this instance, instead of silently
dropping the assignment - a missing tenant group later tripping over
an unrelated validation rule further downstream is far more confusing
than failing here with a precise message.
"""
if not slug:
return None
group = TenantGroup.objects.filter(slug=slug).first()
if group is None:
raise ArchiveFailure(
f"{context}: Die Mandantengruppe „{slug}“ existiert auf dieser Instanz nicht. "
"Bitte die Mandantengruppe vorher anlegen oder den Import mit einem angepassten "
"manifest.json wiederholen."
)
return group
def _resolve_tenant(slug, context):
"""Look up a tenant by slug for an imported record.
Same reasoning as _resolve_tenant_group(): fail loudly and specifically
instead of leaving the field empty and letting a generic "a tenant is
required" validation error surface later with no indication of which
tenant or which record was actually the problem.
"""
if not slug:
return None
tenant = Tenant.objects.filter(slug=slug).first()
if tenant is None:
raise ArchiveFailure(
f"{context}: Der Mandant „{slug}“ existiert auf dieser Instanz nicht. "
"Bitte den Mandanten vorher anlegen oder den Import mit einem angepassten "
"manifest.json wiederholen."
)
return tenant
def _category_from_path(names, tenancy_path=None):
parent = None
tenancy_path = tenancy_path or []
for index, name in enumerate(names):
name = str(name).strip()[:100]
if not name:
continue
slug = slugify(name)[:100] or "ordner"
tenancy = tenancy_path[index] if index < len(tenancy_path) and isinstance(tenancy_path[index], dict) else {}
context = f"Ordner „{name}“"
tenant_group = _resolve_tenant_group(tenancy.get("tenant_group"), context)
tenant = _resolve_tenant(tenancy.get("tenant"), context)
if tenant and tenant_group and tenant.group_id != tenant_group.pk:
raise ArchiveFailure(
f"{context}: Der Mandant „{tenant.slug}“ gehört nicht zur "
f"Mandantengruppe „{tenant_group.slug}“ auf dieser Instanz."
)
category = DocumentCategory.objects.filter(parent=parent, slug=slug).first()
if not category:
# Set tenancy before the first save. NetBox installations may
# enforce a tenant via custom validation rules, so creating the
# category first and assigning its tenant in a second save fails.
category = DocumentCategory(
name=name, slug=slug, parent=parent,
tenant_group=tenant_group, tenant=tenant,
)
_save_imported_object(category, context)
elif tenancy:
category.tenant_group = tenant_group
category.tenant = tenant
_save_imported_object(
category, context,
update_fields=("tenant_group", "tenant", "last_updated"),
)
parent = category
return parent
def _unique_slug(value):
base = (slugify(value) or "dokumentation")[:180]
candidate, number = base, 2
while Document.objects.filter(slug=candidate).exists():
candidate = f"{base[:190-len(str(number))]}-{number}"
number += 1
return candidate
@transaction.atomic
def import_archive(upload, update_existing=False):
try:
archive = ZipFile(upload)
except BadZipFile as exc:
raise ArchiveFailure("Die hochgeladene Datei ist kein gültiges ZIP-Archiv.") from exc
with archive:
_validate_archive(archive)
manifest = _read_json(archive, "manifest.json")
if manifest.get("format") != ARCHIVE_FORMAT or manifest.get("version") != ARCHIVE_VERSION:
raise ArchiveFailure("Archivformat oder Version wird nicht unterstützt.")
records = manifest.get("documents")
if not isinstance(records, list):
raise ArchiveFailure("Die Dokumentliste im Archiv ist ungültig.")
allowed = set(settings.PLUGINS_CONFIG.get("netbox_documentation", {}).get("allowed_object_types", []))
created, updated, skipped_assignments = 0, 0, 0
for record in records:
if not isinstance(record, dict) or not record.get("title") or not record.get("body_path"):
raise ArchiveFailure("Das Archiv enthält einen unvollständigen Dokumenteintrag.")
try:
body = archive.read(record["body_path"]).decode("utf-8")
except (KeyError, UnicodeDecodeError) as exc:
raise ArchiveFailure("Ein Dokumentinhalt fehlt oder ist nicht UTF-8-kodiert.") from exc
source_slug = str(record.get("slug") or record["title"])
document = Document.objects.filter(slug=source_slug).first() if update_existing else None
category = _category_from_path(
record.get("category_path") or [], record.get("category_tenancy") or [],
)
if document:
updated += 1
else:
document = Document(slug=_unique_slug(source_slug))
created += 1
document.title = str(record["title"])[:200]
document.summary = str(record.get("summary") or "")[:500]
document.body = body
document.body_format = record.get("body_format") if record.get("body_format") in {"html", "markdown"} else "html"
document.is_published = bool(record.get("is_published", True))
document.category = category
context = f"Dokument „{document.title}“"
document.tenant_group = _resolve_tenant_group(record.get("tenant_group"), context)
document.tenant = _resolve_tenant(record.get("tenant"), context)
if document.tenant and document.tenant_group and document.tenant.group_id != document.tenant_group_id:
raise ArchiveFailure(
f"{context}: Der Mandant „{document.tenant.slug}“ gehört nicht zur "
f"Mandantengruppe „{document.tenant_group.slug}“ auf dieser Instanz."
)
_save_imported_object(document, context)
for assignment in record.get("assignments") or []:
label = f"{assignment.get('app_label')}.{assignment.get('model')}"
if label not in allowed:
skipped_assignments += 1
continue
content_type = ContentType.objects.filter(app_label=assignment.get("app_label"), model=assignment.get("model")).first()
model = content_type.model_class() if content_type else None
if not model or not model.objects.filter(pk=assignment.get("object_id")).exists():
skipped_assignments += 1
continue
DocumentAssignment.objects.get_or_create(
document=document, assigned_object_type=content_type,
assigned_object_id=assignment["object_id"],
defaults={"note": str(assignment.get("note") or "")[:200]},
)
for item in record.get("attachments") or []:
member = item.get("path")
if not member:
continue
try:
content = archive.read(member)
except KeyError as exc:
raise ArchiveFailure(f"Anhang {member} fehlt im Archiv.") from exc
from django.core.files.base import ContentFile
original_name = Path(str(item.get("original_name") or "attachment")).name[:255]
allowed_extensions = {".docx", ".xlsx", ".xlsm", ".pdf", ".md", ".txt", ".jpg", ".jpeg", ".png", ".gif", ".webp"}
if Path(original_name).suffix.lower() not in allowed_extensions:
raise ArchiveFailure(f"Der Anhang {original_name} verwendet einen nicht erlaubten Dateityp.")
existing = document.attachments.filter(original_name=original_name, size=len(content)).first()
if existing:
if item.get("source_url"):
document.body = document.body.replace(str(item["source_url"]), existing.file.url)
continue
attachment = DocumentAttachment(
document=document, original_name=original_name,
content_type=str(item.get("content_type") or "")[:100], size=len(content),
)
attachment.file.save(original_name, ContentFile(content), save=False)
attachment.save()
if item.get("source_url"):
document.body = document.body.replace(str(item["source_url"]), attachment.file.url)
document.save(update_fields=("body", "last_updated"))
return {"created": created, "updated": updated, "skipped_assignments": skipped_assignments}