Files
bambuddy/backend/app/services/git_providers/gitlab.py
T
jmoore-skild 6a239314dc feat(backup): restore selected categories from a Git backup commit (#2656)
The Git backup feature was push-only: there was no equivalent of the local
backup's Restore button, so recovering meant hand-downloading JSON files from
the repository. This adds the read side.

Providers gain list_commits / list_tree / fetch_files on the GitProviderBackend
ABC. GitHub implements them against the Git Data API and Gitea/Forgejo inherit
that unchanged; GitLab overrides for its own REST shape, including tree
pagination and subgroup path encoding. fetch_files is batched so the path ->
blob SHA lookup happens once per restore rather than once per file, and uses the
blobs API rather than contents because contents silently inlines only the first
1 MB.

The new GitHubRestoreService resolves HEAD to a concrete SHA up front, so a
preview and the restore that follows act on the same commit even if a scheduled
backup lands in between. Categories are applied archives -> spools -> settings
-> kprofiles: archives first because spool usage history references archive_id,
K-profiles last because they leave the database and publish over MQTT.

Restores never reuse the backup's primary keys. spool.id and print_archives.id
are bare autoincrement columns, so ids from an old backup very likely belong to
unrelated rows today; rows are matched on natural keys (tag_uid, then
tray_uuid, then a descriptive composite for spools; content_hash or filename
plus started_at for archives), inserted without an explicit id, and an
old_id -> new_id map rewrites the foreign keys in spool usage history.
created_at is carried across on insert so restoring the same backup twice
matches instead of duplicating. Dangling printer/project links are cleared and
reported rather than failing the row.

Settings restore re-applies the collector's credential denylist on the read
side, plus a pattern guard, because a backup taken before that denylist existed
can still contain secrets. Restored archives are metadata-only: the 3MF and
thumbnail bytes are not in a Git backup and print_archives.file_path is NOT
NULL, so inserted rows get an empty path and the UI says so.

Backup and restore take a mutex against each other; both write the same tables
and talk to the same printers. Restores are logged as GitHubBackupLog rows with
trigger="restore", which needs no migration and surfaces them in the existing
History card.

Cloud profiles are deliberately not a restore category. The collector never
actually writes cloud_profiles/*.json - it reads a "setting" list key the Bambu
Cloud API does not return - and the preset list it would write carries no
setting payload. Filed separately.

Permission github:restore already existed and is granted to Administrators, so
no permission changes were needed.

Tests: 125 new backend tests (provider reads across all four providers, the
per-category appliers, the API endpoints) and 13 frontend tests. Full suites
pass with no regressions; the 35 backend failures on Windows are byte-identical
with and without this branch.
2026-08-04 08:22:57 -04:00

462 lines
20 KiB
Python

"""GitLab backend — implements GitProviderBackend using the GitLab REST API v4."""
import base64
import json
import logging
import re
import urllib.parse
from datetime import datetime, timezone
import httpx
from backend.app.services.git_providers.base import GitProviderBackend
logger = logging.getLogger(__name__)
class GitLabBackend(GitProviderBackend):
"""Backend for gitlab.com and self-hosted GitLab instances."""
def get_api_base(self, repo_url: str) -> str:
match = re.match(r"(https?://[\w.\-]+(:\d+)?)/", repo_url)
if not match:
raise ValueError(f"Cannot derive API base from URL: {repo_url}")
return f"{match.group(1)}/api/v4"
def get_headers(self, token: str) -> dict:
return {"Authorization": f"Bearer {token}", "Content-Type": "application/json"}
def parse_repo_url(self, url: str) -> tuple[str, str]:
"""Return (namespace, repo) from HTTPS or SSH URL.
namespace may include subgroups, e.g. 'group/subgroup' for
gitlab.com/group/subgroup/project. Callers join them with '/' and
URL-encode the result for /api/v4/projects/{encoded_path}.
"""
if not url or len(url) > 500:
raise ValueError("Invalid Git URL: URL too long or empty")
match = re.match(r"https?://[\w.\-]+(:\d+)?/(.+?)(?:\.git)?/?$", url)
if match:
full_path = match.group(2)
if "/" not in full_path:
raise ValueError(f"Cannot parse repository URL: {url}")
namespace, _, repo = full_path.rpartition("/")
return namespace, repo
match = re.match(r"git@[\w.\-]+:(.+?)(?:\.git)?$", url)
if match:
full_path = match.group(1)
if "/" not in full_path:
raise ValueError(f"Cannot parse repository URL: {url}")
namespace, _, repo = full_path.rpartition("/")
return namespace, repo
raise ValueError(f"Cannot parse repository URL: {url}")
async def test_connection(self, repo_url: str, token: str, client: httpx.AsyncClient) -> dict:
try:
owner, repo = self.parse_repo_url(repo_url)
api_base = self.get_api_base(repo_url)
headers = self.get_headers(token)
encoded_path = urllib.parse.quote(f"{owner}/{repo}", safe="")
response = await client.get(f"{api_base}/projects/{encoded_path}", headers=headers)
if response.status_code == 401:
return {"success": False, "message": "Invalid access token", "repo_name": None, "permissions": None}
if response.status_code == 404:
return {
"success": False,
"message": "Repository not found. Check URL and token permissions.",
"repo_name": None,
"permissions": None,
}
if response.status_code != 200:
return {
"success": False,
"message": f"API error: {response.status_code}",
"repo_name": None,
"permissions": None,
}
data = response.json()
perms = data.get("permissions") or {}
project_level = (perms.get("project_access") or {}).get("access_level", 0)
group_level = (perms.get("group_access") or {}).get("access_level", 0)
effective = max(project_level, group_level)
# GitLab uses visibility="private" / "internal" / "public". Both
# "internal" (signed-in users) and "public" are non-private for
# the purposes of this safety check.
visibility = (data.get("visibility") or "").lower()
is_private = visibility == "private"
if effective < 30: # Developer = 30, Maintainer = 40, Owner = 50
return {
"success": False,
"message": "Token requires Developer access or higher to push",
"repo_name": data.get("name_with_namespace"),
"permissions": perms,
"is_private": is_private,
}
return {
"success": True,
"message": "Connection successful",
"repo_name": data.get("name_with_namespace"),
"permissions": perms,
"is_private": is_private,
}
except Exception as e:
logger.error("GitLab connection test failed: %s", e)
return {
"success": False,
"message": f"Connection failed: {type(e).__name__}",
"repo_name": None,
"permissions": None,
"is_private": None,
}
def _encoded_project(self, repo_url: str) -> str:
"""Return the URL-encoded ``namespace/project`` path for /api/v4/projects/."""
owner, repo = self.parse_repo_url(repo_url)
return urllib.parse.quote(f"{owner}/{repo}", safe="")
async def list_commits(
self,
repo_url: str,
token: str,
branch: str,
client: httpx.AsyncClient,
limit: int = 20,
) -> dict:
"""List recent commits on ``branch`` via /repository/commits."""
try:
api_base = self.get_api_base(repo_url)
headers = self.get_headers(token)
encoded_path = self._encoded_project(repo_url)
response = await client.get(
f"{api_base}/projects/{encoded_path}/repository/commits",
headers=headers,
params={"ref_name": branch, "per_page": limit},
)
if response.status_code == 404:
return {
"success": False,
"message": (
f"Branch '{branch}' not found, or the repository has no commits yet. "
"Run a backup before restoring."
),
"commits": [],
}
if response.status_code != 200:
msg = f"Failed to list commits (HTTP {response.status_code}): {self._truncated_response_text(response)}"
logger.warning("list_commits %s: %s", repo_url, msg)
return {"success": False, "message": msg, "commits": []}
try:
data = response.json()
except ValueError:
return {"success": False, "message": "Non-JSON response listing commits", "commits": []}
if not isinstance(data, list):
return {"success": False, "message": "Unexpected shape listing commits", "commits": []}
commits = []
for entry in data[:limit]:
if not isinstance(entry, dict):
continue
sha = entry.get("id")
if not isinstance(sha, str) or not sha:
continue
# GitLab flattens author/date onto the commit itself rather than
# nesting them under "commit" the way GitHub does.
commits.append(
{
"sha": sha,
"message": entry.get("message") or "",
"author": entry.get("author_name") or "",
"date": entry.get("committed_date") or entry.get("created_at") or "",
}
)
return {"success": True, "message": "OK", "commits": commits}
except Exception as e:
logger.exception("list_commits failed for %s branch=%s", repo_url, branch)
return {"success": False, "message": f"{type(e).__name__}: {str(e)[:200]}", "commits": []}
async def list_tree(
self,
repo_url: str,
token: str,
ref: str,
client: httpx.AsyncClient,
) -> dict:
"""List blob paths at ``ref`` via /repository/tree, following pagination."""
try:
api_base = self.get_api_base(repo_url)
headers = self.get_headers(token)
encoded_path = self._encoded_project(repo_url)
paths: list[str] = []
page = 1
# GitLab's tree endpoint paginates instead of exposing a "truncated"
# flag, so walk pages until one comes back short. The page cap stops
# a malformed X-Next-Page loop from spinning forever.
while page <= 50:
response = await client.get(
f"{api_base}/projects/{encoded_path}/repository/tree",
headers=headers,
params={"ref": ref, "recursive": "true", "per_page": 100, "page": page},
)
if response.status_code == 404:
return {
"success": False,
"message": f"Commit or tree '{ref}' not found in the repository",
"paths": [],
}
if response.status_code != 200:
msg = (
f"Failed to list tree (HTTP {response.status_code}): {self._truncated_response_text(response)}"
)
logger.warning("list_tree %s ref=%s: %s", repo_url, ref, msg)
return {"success": False, "message": msg, "paths": []}
try:
data = response.json()
except ValueError:
return {"success": False, "message": "Non-JSON response listing tree", "paths": []}
if not isinstance(data, list):
return {"success": False, "message": "Unexpected shape listing tree", "paths": []}
for item in data:
if isinstance(item, dict) and item.get("type") == "blob":
path = item.get("path")
if isinstance(path, str) and path:
paths.append(path)
if len(data) < 100:
break
page += 1
return {"success": True, "message": "OK", "paths": sorted(paths)}
except Exception as e:
logger.exception("list_tree failed for %s ref=%s", repo_url, ref)
return {"success": False, "message": f"{type(e).__name__}: {str(e)[:200]}", "paths": []}
async def fetch_files(
self,
repo_url: str,
token: str,
ref: str,
paths: list[str],
client: httpx.AsyncClient,
) -> dict:
"""Read ``paths`` at ``ref`` via /repository/files/{path}."""
try:
api_base = self.get_api_base(repo_url)
headers = self.get_headers(token)
encoded_path = self._encoded_project(repo_url)
files: dict[str, str] = {}
for path in paths:
encoded_file = urllib.parse.quote(path, safe="")
response = await client.get(
f"{api_base}/projects/{encoded_path}/repository/files/{encoded_file}",
headers=headers,
params={"ref": ref},
)
# A path absent from this commit is expected — which categories a
# backup contains varies by config — so skip rather than fail.
if response.status_code == 404:
continue
if response.status_code != 200:
msg = (
f"Failed to read {path} (HTTP {response.status_code}): "
f"{self._truncated_response_text(response)}"
)
logger.warning("fetch_files %s: %s", repo_url, msg)
return {"success": False, "message": msg, "files": {}}
try:
data = response.json()
except ValueError:
return {"success": False, "message": f"Non-JSON response reading {path}", "files": {}}
if not isinstance(data, dict):
return {"success": False, "message": f"Unexpected shape reading {path}", "files": {}}
content = data.get("content")
if not isinstance(content, str):
return {"success": False, "message": f"Missing content reading {path}", "files": {}}
encoding = data.get("encoding", "base64")
try:
if encoding == "base64":
files[path] = base64.b64decode(content).decode("utf-8")
elif encoding in ("text", "utf-8", "plain"):
files[path] = content
else:
return {
"success": False,
"message": f"Unsupported encoding {encoding!r} reading {path}",
"files": {},
}
except (ValueError, UnicodeDecodeError) as e:
return {"success": False, "message": f"Could not decode {path}: {type(e).__name__}", "files": {}}
return {"success": True, "message": "OK", "files": files}
except Exception as e:
logger.exception("fetch_files failed for %s ref=%s", repo_url, ref)
return {"success": False, "message": f"{type(e).__name__}: {str(e)[:200]}", "files": {}}
async def push_files(
self,
repo_url: str,
token: str,
branch: str,
files: dict,
client: httpx.AsyncClient,
) -> dict:
try:
owner, repo = self.parse_repo_url(repo_url)
api_base = self.get_api_base(repo_url)
headers = self.get_headers(token)
encoded_path = urllib.parse.quote(f"{owner}/{repo}", safe="")
encoded_branch = urllib.parse.quote(branch, safe="")
branch_response = await client.get(
f"{api_base}/projects/{encoded_path}/repository/branches/{encoded_branch}",
headers=headers,
)
if branch_response.status_code == 404:
proj_response = await client.get(f"{api_base}/projects/{encoded_path}", headers=headers)
if proj_response.status_code != 200:
return {"status": "failed", "message": "Failed to get project info"}
default_branch = proj_response.json().get("default_branch", "main")
default_encoded = urllib.parse.quote(default_branch, safe="")
default_response = await client.get(
f"{api_base}/projects/{encoded_path}/repository/branches/{default_encoded}",
headers=headers,
)
if default_response.status_code != 200:
return await self._create_initial_commit(client, headers, api_base, encoded_path, branch, files)
create_response = await client.post(
f"{api_base}/projects/{encoded_path}/repository/branches",
headers=headers,
json={"branch": branch, "ref": default_branch},
)
if create_response.status_code not in (200, 201):
return {"status": "failed", "message": f"Failed to create branch: {create_response.status_code}"}
elif branch_response.status_code != 200:
return {"status": "failed", "message": f"Failed to check branch: {branch_response.status_code}"}
existing_blobs: dict[str, str] = {}
page = 1
while True:
tree_response = await client.get(
f"{api_base}/projects/{encoded_path}/repository/tree",
headers=headers,
params={"recursive": "true", "ref": branch, "per_page": 100, "page": page},
)
if tree_response.status_code != 200:
break
items = tree_response.json()
if not items:
break
for item in items:
if item.get("type") == "blob":
existing_blobs[item["path"]] = item["id"]
page += 1
actions = []
for path, content in files.items():
content_str = json.dumps(content, indent=2, default=str)
content_bytes = content_str.encode("utf-8")
content_sha = self._blob_sha(content_bytes)
if path in existing_blobs and existing_blobs[path] == content_sha:
continue
actions.append(
{
"action": "update" if path in existing_blobs else "create",
"file_path": path,
"content": base64.b64encode(content_bytes).decode(),
"encoding": "base64",
}
)
if not actions:
return {"status": "skipped", "message": "No changes to commit", "commit_sha": None, "files_changed": 0}
commit_message = f"Bambuddy backup - {datetime.now(timezone.utc).strftime('%Y-%m-%d %H:%M:%S UTC')}"
commit_response = await client.post(
f"{api_base}/projects/{encoded_path}/repository/commits",
headers=headers,
json={"branch": branch, "commit_message": commit_message, "actions": actions},
)
if commit_response.status_code not in (200, 201):
return {
"status": "failed",
"message": f"Failed to create commit: {self._truncated_response_text(commit_response)}",
}
return {
"status": "success",
"message": f"Backup successful - {len(actions)} files updated",
"commit_sha": commit_response.json().get("id"),
"files_changed": len(actions),
}
except Exception as e:
logger.error("Push to GitLab failed: %s", e)
return {"status": "failed", "message": str(e), "error": str(e)}
async def _create_initial_commit(
self,
client: httpx.AsyncClient,
headers: dict,
api_base: str,
encoded_path: str,
branch: str,
files: dict,
) -> dict:
"""Create the first commit in an empty repository."""
try:
actions = []
for path, content in files.items():
content_str = json.dumps(content, indent=2, default=str)
actions.append(
{
"action": "create",
"file_path": path,
"content": base64.b64encode(content_str.encode()).decode(),
"encoding": "base64",
}
)
commit_message = f"Initial Bambuddy backup - {datetime.now(timezone.utc).strftime('%Y-%m-%d %H:%M:%S UTC')}"
commit_response = await client.post(
f"{api_base}/projects/{encoded_path}/repository/commits",
headers=headers,
json={"branch": branch, "commit_message": commit_message, "actions": actions, "start_branch": branch},
)
if commit_response.status_code not in (200, 201):
return {
"status": "failed",
"message": f"Failed to create initial commit: {self._truncated_response_text(commit_response)}",
}
return {
"status": "success",
"message": f"Initial backup created - {len(files)} files",
"commit_sha": commit_response.json().get("id"),
"files_changed": len(files),
}
except Exception as e:
return {"status": "failed", "message": str(e)}