mirror of
https://github.com/maziggy/bambuddy.git
synced 2026-10-01 03:31:25 +02:00
`if not isinstance(total, int) or seen >= total or not entries: return blobs, ""` — the first arm short-circuited the page loop into a **success** holding page 1 only. Gitea clamps `per_page` to `MAX_RESPONSE_ITEMS` (default 50), so that is 50 entries of an arbitrarily large tree returned as a complete listing. The restore then reports genuinely-present categories as "Not present in this backup commit". That silent skip is the exact failure this override exists to prevent, and the same class as E7 and G2 — G2 fixed the arithmetic here and left the shape. GitHub and GitLab both hard-fail in the equivalent spot; only Gitea guessed, and it guessed in the one direction that loses data quietly. Whether Gitea always sends `total_count` on this route is beside the point: the code was defending against a response shape it did not trust, and then trusting it. Now a missing or non-int `total_count` means "page until a short or empty page". A page shorter than the first one is the last one, floored at Gitea's default clamp so a genuinely small tree still costs exactly one request — the reason G2 rejected paging-until-short in the `total_count`-present case, which is unchanged and still stops on the count. The existing `page <= 50` ceiling gives the correct hard failure for a tree that really is over cap, so this cannot truncate. Residual, and deliberately not widened into a `return None` on the first ambiguous response — that would break single-page trees, the common case: an instance whose `MAX_RESPONSE_ITEMS` is set *below* 50 *and* which omits `total_count` would still stop at page 1. Both halves have to be true. Tests: +6 (paged to the end with no count, on both Gitea and Forgejo; a short page ends it; a non-int count is treated as no count; the page ceiling still fails). Fail-pre-fix 5, control that passes either way 1 (a small tree is one request). These don't match `-k github`, so 274 -> 280 across the three restore files but `-k github` is unmoved.
479 lines
22 KiB
Python
479 lines
22 KiB
Python
"""Gitea backend — overrides GitHubBackend where Gitea's API diverges."""
|
|
|
|
import base64
|
|
import json
|
|
import logging
|
|
import re
|
|
from datetime import datetime, timezone
|
|
|
|
import httpx
|
|
|
|
from backend.app.services.git_providers.github import GitHubBackend
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# Gitea clamps per_page to MAX_RESPONSE_ITEMS, which defaults to 50. Consulted
|
|
# only when a tree response carries no usable total_count: a page at least this
|
|
# long may be a clamped full page and cannot be assumed to be the last one.
|
|
_ASSUMED_MIN_PAGE_SIZE = 50
|
|
|
|
|
|
class GiteaBackend(GitHubBackend):
|
|
"""Backend for Gitea instances.
|
|
|
|
Gitea's Git Data API (/api/v1/repos/{owner}/{repo}/git/...) is *mostly*
|
|
compatible with GitHub's, but diverges on three points that broke real-world
|
|
backups (#1224, #1225, #1239):
|
|
|
|
1. ``GET /git/refs/heads/{branch}`` returns a *list* of matching refs even
|
|
when only one matches; GitHub returns a single object. The push paths
|
|
below extract the SHA via ``_ref_sha()`` instead of the GitHub-style
|
|
``["object"]["sha"]`` chain.
|
|
|
|
2. The Git Data API (blobs/trees/commits/refs) refuses writes against an
|
|
empty repository — every blob POST returns 404 until the repo has at
|
|
least one commit. ``_create_initial_commit()`` is overridden to use the
|
|
Contents API, which seeds the branch + initial commit in a single call.
|
|
|
|
3. The Git Data API does not support atomic multi-file commits — each file
|
|
requires a separate blob POST followed by a tree/commit/ref sequence.
|
|
``push_files()`` is overridden to use the Contents API
|
|
(``POST /repos/.../contents`` with a ``files`` array), which commits all
|
|
changed files in a single round-trip and avoids partial-commit failures.
|
|
"""
|
|
|
|
@staticmethod
|
|
def _ref_sha(ref_data) -> str:
|
|
"""Extract the commit SHA from Gitea's list-shaped ref response."""
|
|
if isinstance(ref_data, list):
|
|
if not ref_data:
|
|
raise ValueError("Empty refs list returned by Gitea API")
|
|
return ref_data[0]["object"]["sha"]
|
|
return ref_data["object"]["sha"]
|
|
|
|
@staticmethod
|
|
def _commit_tree_sha(commit_data: dict) -> str | None:
|
|
"""Extract the tree SHA from a commit response.
|
|
|
|
GitHub's ``GET /git/commits/{sha}`` returns the GitCommit schema with
|
|
``tree`` at the top level. Gitea's same-named endpoint may return the
|
|
wrapped Commit schema where ``tree`` lives under ``commit``. Try the
|
|
flat shape first (GitHub-compatible deployments and some Gitea/Forgejo
|
|
versions) then fall back to the wrapped shape.
|
|
"""
|
|
tree_node = commit_data.get("tree")
|
|
if not isinstance(tree_node, dict):
|
|
tree_node = (commit_data.get("commit") or {}).get("tree")
|
|
if isinstance(tree_node, dict):
|
|
return tree_node.get("sha")
|
|
return None
|
|
|
|
# Gitea/Forgejo can be hosted under a URL path prefix (ROOT_URL like
|
|
# https://host/gitea), so the repo lives at /<prefix...>/<owner>/<repo>
|
|
# rather than at the host root (#2642). Capture the scheme+host+prefix as
|
|
# one group and the final two path segments as owner/repo; the lazy prefix
|
|
# group is empty for a root-hosted instance. One shared pattern keeps
|
|
# parse_repo_url() and get_api_base() from drifting.
|
|
_HTTPS_REPO_RE = re.compile(
|
|
r"(https?://[\w.\-]+(?::\d+)?(?:/[\w.\-]+)*?)/([\w.\-]{1,100})/([\w.\-]{1,100})(?:\.git)?/?$"
|
|
)
|
|
|
|
def parse_repo_url(self, url: str) -> tuple[str, str]:
|
|
"""Return (owner, repo) — accepts both https:// and http:// for self-hosted instances."""
|
|
if not url or len(url) > 500:
|
|
raise ValueError("Invalid Git URL: URL too long or empty")
|
|
match = self._HTTPS_REPO_RE.match(url)
|
|
if match:
|
|
return match.group(2), match.group(3).removesuffix(".git")
|
|
match = re.match(
|
|
r"git@[\w.\-]+:([\w.\-]{1,100})/([\w.\-]{1,100})(?:\.git)?$",
|
|
url,
|
|
)
|
|
if match:
|
|
return match.group(1), match.group(2).removesuffix(".git")
|
|
raise ValueError(f"Cannot parse repository URL: {url}")
|
|
|
|
def get_api_base(self, repo_url: str) -> str:
|
|
"""Derive API base from the repository URL's scheme, host and any path prefix."""
|
|
match = self._HTTPS_REPO_RE.match(repo_url)
|
|
if match:
|
|
return f"{match.group(1)}/api/v1"
|
|
raise ValueError(f"Cannot derive API base from URL: {repo_url}")
|
|
|
|
def get_headers(self, token: str) -> dict:
|
|
headers = super().get_headers(token)
|
|
headers["Accept"] = "application/json"
|
|
return headers
|
|
|
|
async def _blob_shas_at(
|
|
self,
|
|
client: httpx.AsyncClient,
|
|
headers: dict,
|
|
api_base: str,
|
|
owner: str,
|
|
repo: str,
|
|
ref: str,
|
|
) -> tuple[dict[str, str] | None, str]:
|
|
"""Paged override of GitHub's single-GET tree read (#2656).
|
|
|
|
Divergence four, alongside the three in the class docstring. GitHub's
|
|
recursive trees endpoint is not paginated and signals overflow with
|
|
``truncated: true``, which the inherited implementation hard-fails on.
|
|
Gitea and Forgejo *do* page the same endpoint — ``page``/``per_page``,
|
|
with ``total_count`` alongside the tree — so the inherited version would
|
|
read only the first page and then report every category beyond it as
|
|
absent from the commit. A restore that silently skips categories is the
|
|
exact failure the GitHub version refuses to allow, so this pages instead.
|
|
|
|
The cap mirrors GitLab's: reaching it means there are more pages, and
|
|
that is a failure rather than a partial result. Because the page size is
|
|
the server's choice rather than ours (see below), the cap is a page count
|
|
and not a file count.
|
|
"""
|
|
blobs: dict[str, str] = {}
|
|
seen = 0
|
|
page = 1
|
|
page_size: int | None = None
|
|
while page <= 50:
|
|
response = await client.get(
|
|
f"{api_base}/repos/{owner}/{repo}/git/trees/{ref}",
|
|
headers=headers,
|
|
params={"recursive": "true", "page": page, "per_page": 1000},
|
|
)
|
|
if response.status_code == 404:
|
|
return None, f"Commit or tree '{ref}' not found in the repository"
|
|
if response.status_code != 200:
|
|
return None, (
|
|
f"Failed to list tree (HTTP {response.status_code}): {self._truncated_response_text(response)}"
|
|
)
|
|
try:
|
|
data = response.json()
|
|
except ValueError:
|
|
return None, "Non-JSON response listing tree"
|
|
if not isinstance(data, dict):
|
|
return None, "Unexpected shape listing tree"
|
|
|
|
entries = data.get("tree")
|
|
if not isinstance(entries, list):
|
|
entries = []
|
|
for item in entries:
|
|
if not isinstance(item, dict) or item.get("type") != "blob":
|
|
continue
|
|
path, sha = item.get("path"), item.get("sha")
|
|
if isinstance(path, str) and isinstance(sha, str) and path and sha:
|
|
blobs[path] = sha
|
|
|
|
# total_count counts every entry, trees included, so compare against
|
|
# what came back rather than against len(blobs).
|
|
#
|
|
# Count what the server actually returned, never the per_page we
|
|
# asked for: Gitea clamps per_page to MAX_RESPONSE_ITEMS, which
|
|
# defaults to 50. Deriving the offset from the requested 1000 made
|
|
# page 2 report 1050 entries seen, which clears any total_count below
|
|
# that — so the loop stopped and returned the first two pages of a
|
|
# much larger tree as a success. The restore then read every missing
|
|
# path as "category not present in this commit" and skipped it
|
|
# silently, the exact failure this override exists to prevent.
|
|
total = data.get("total_count")
|
|
seen += len(entries)
|
|
if page_size is None:
|
|
page_size = max(len(entries), _ASSUMED_MIN_PAGE_SIZE)
|
|
|
|
if not entries:
|
|
return blobs, ""
|
|
if isinstance(total, int):
|
|
if seen >= total:
|
|
return blobs, ""
|
|
elif len(entries) < page_size:
|
|
# No usable total_count. This used to return here on the *first*
|
|
# page, i.e. fail open into a success holding whatever one page
|
|
# happened to be — 50 entries of an arbitrarily large tree under
|
|
# the default clamp — and the restore then reported every
|
|
# category beyond it as absent from the commit. Page until a
|
|
# short or empty page instead; the page-count ceiling below
|
|
# still gives the correct hard failure for a tree that really is
|
|
# too large. A page shorter than the first one (or than Gitea's
|
|
# default clamp, so a genuinely small tree stays one request)
|
|
# cannot be followed by another. The residual case is an
|
|
# instance whose MAX_RESPONSE_ITEMS is set *below* 50 and which
|
|
# also omits total_count; real Gitea and Forgejo always send it
|
|
# on this route.
|
|
return blobs, ""
|
|
page += 1
|
|
|
|
return None, (
|
|
"Repository tree exceeds the listing limit, so the backup contents cannot be "
|
|
"enumerated reliably. Rotate the backup repository."
|
|
)
|
|
|
|
async def push_files(
|
|
self,
|
|
repo_url: str,
|
|
token: str,
|
|
branch: str,
|
|
files: dict,
|
|
client: httpx.AsyncClient,
|
|
_allow_branch_create: bool = True,
|
|
) -> dict:
|
|
"""Push files via the Git Data API, normalising Gitea's list-shaped ref response."""
|
|
try:
|
|
owner, repo = self.parse_repo_url(repo_url)
|
|
api_base = self.get_api_base(repo_url)
|
|
headers = self.get_headers(token)
|
|
|
|
ref_response = await client.get(f"{api_base}/repos/{owner}/{repo}/git/refs/heads/{branch}", headers=headers)
|
|
|
|
if ref_response.status_code == 404:
|
|
if not _allow_branch_create:
|
|
return {
|
|
"status": "failed",
|
|
"message": (
|
|
f"Branch '{branch}' not found after creation — possible replication lag. "
|
|
"The next scheduled backup will retry."
|
|
),
|
|
}
|
|
return await self._create_branch_and_push(
|
|
client, headers, api_base, owner, repo, branch, files, repo_url, token
|
|
)
|
|
|
|
if ref_response.status_code != 200:
|
|
return {
|
|
"status": "failed",
|
|
"message": f"Failed to get branch ref: {ref_response.status_code}",
|
|
"error": self._truncated_response_text(ref_response),
|
|
}
|
|
|
|
current_commit_sha = self._ref_sha(ref_response.json())
|
|
|
|
commit_response = await client.get(
|
|
f"{api_base}/repos/{owner}/{repo}/git/commits/{current_commit_sha}", headers=headers
|
|
)
|
|
if commit_response.status_code != 200:
|
|
msg = f"Failed to get current commit (HTTP {commit_response.status_code}): {self._truncated_response_text(commit_response)}"
|
|
logger.warning("push_files %s/%s: %s", owner, repo, msg)
|
|
return {"status": "failed", "message": msg}
|
|
|
|
current_tree_sha = self._commit_tree_sha(commit_response.json())
|
|
if not current_tree_sha:
|
|
msg = (
|
|
f"Failed to extract tree SHA from commit response: {self._truncated_response_text(commit_response)}"
|
|
)
|
|
logger.warning("push_files %s/%s: %s", owner, repo, msg)
|
|
return {"status": "failed", "message": msg}
|
|
|
|
tree_response = await client.get(
|
|
f"{api_base}/repos/{owner}/{repo}/git/trees/{current_tree_sha}?recursive=1", headers=headers
|
|
)
|
|
if tree_response.status_code != 200:
|
|
msg = f"Failed to list existing tree (HTTP {tree_response.status_code}): {self._truncated_response_text(tree_response)}"
|
|
logger.warning("push_files %s/%s: %s", owner, repo, msg)
|
|
return {"status": "failed", "message": msg, "error": self._truncated_response_text(tree_response)}
|
|
tree_data = tree_response.json()
|
|
# Gitea's tree API can report ``truncated: true`` for large
|
|
# listings; if we honour the partial map, the dedup check misses
|
|
# and every file gets re-uploaded each run.
|
|
if tree_data.get("truncated"):
|
|
msg = (
|
|
"Repository tree exceeds the Gitea API listing limit (truncated=true). "
|
|
"Rotate the backup repository to avoid silent file-by-file churn on every backup."
|
|
)
|
|
logger.warning("push_files %s/%s: %s", owner, repo, msg)
|
|
return {"status": "failed", "message": msg}
|
|
existing_files: dict[str, str] = {}
|
|
for item in tree_data.get("tree", []):
|
|
if item.get("type") != "blob":
|
|
continue
|
|
path, sha = item.get("path"), item.get("sha")
|
|
if not path or not sha:
|
|
logger.warning("push_files: skipping malformed tree entry: %s", item)
|
|
continue
|
|
existing_files[path] = sha
|
|
|
|
api_files = []
|
|
files_changed = 0
|
|
|
|
for path, content in files.items():
|
|
content_str = json.dumps(content, indent=2, default=str)
|
|
content_bytes = content_str.encode("utf-8")
|
|
content_b64 = base64.b64encode(content_bytes).decode()
|
|
content_sha = self._blob_sha(content_bytes)
|
|
|
|
if path in existing_files:
|
|
if existing_files[path] == content_sha:
|
|
continue
|
|
api_files.append(
|
|
{"operation": "update", "path": path, "content": content_b64, "sha": existing_files[path]}
|
|
)
|
|
else:
|
|
api_files.append({"operation": "create", "path": path, "content": content_b64})
|
|
files_changed += 1
|
|
|
|
if not api_files:
|
|
return {"status": "skipped", "message": "No changes to commit", "commit_sha": None, "files_changed": 0}
|
|
|
|
commit_message = f"Bambuddy backup - {datetime.now(timezone.utc).strftime('%Y-%m-%d %H:%M:%S UTC')}"
|
|
response = await client.post(
|
|
f"{api_base}/repos/{owner}/{repo}/contents",
|
|
headers=headers,
|
|
json={"branch": branch, "message": commit_message, "files": api_files},
|
|
)
|
|
|
|
if response.status_code == 404:
|
|
return {
|
|
"status": "failed",
|
|
"message": "Contents API endpoint not found — your Gitea instance may be older than v1.18 or the API may be disabled by an administrator (POST /contents returned 404)",
|
|
}
|
|
if response.status_code == 409:
|
|
return {
|
|
"status": "failed",
|
|
"message": (
|
|
"Conflict committing files — the branch likely advanced concurrently "
|
|
"(web-UI edit, another backup run, or path-vs-tree collision). "
|
|
"The next scheduled backup will re-read the current tree and resolve this."
|
|
),
|
|
}
|
|
if response.status_code not in (200, 201):
|
|
return {
|
|
"status": "failed",
|
|
"message": f"Backup commit failed: {self._truncated_response_text(response)}",
|
|
}
|
|
|
|
commit_sha = (response.json().get("commit") or {}).get("sha")
|
|
message = (
|
|
f"Backup successful - {files_changed} files updated"
|
|
if commit_sha
|
|
else f"Backup successful - {files_changed} files updated (commit SHA not reported by server)"
|
|
)
|
|
return {
|
|
"status": "success",
|
|
"message": message,
|
|
"commit_sha": commit_sha,
|
|
"files_changed": files_changed,
|
|
}
|
|
|
|
except Exception as e:
|
|
logger.exception("push_files failed for %s branch=%s", repo_url, branch)
|
|
return {"status": "failed", "message": str(e), "error": str(e)}
|
|
|
|
async def _create_branch_and_push(
|
|
self,
|
|
client: httpx.AsyncClient,
|
|
headers: dict,
|
|
api_base: str,
|
|
owner: str,
|
|
repo: str,
|
|
branch: str,
|
|
files: dict,
|
|
repo_url: str,
|
|
token: str,
|
|
) -> dict:
|
|
"""Create branch (from default branch or as initial commit) then push."""
|
|
try:
|
|
repo_response = await client.get(f"{api_base}/repos/{owner}/{repo}", headers=headers)
|
|
if repo_response.status_code != 200:
|
|
msg = f"Failed to get repo info (HTTP {repo_response.status_code}): {self._truncated_response_text(repo_response)}"
|
|
logger.warning("_create_branch_and_push %s/%s: %s", owner, repo, msg)
|
|
return {"status": "failed", "message": msg}
|
|
|
|
default_branch = repo_response.json().get("default_branch", "main")
|
|
|
|
# GET the default branch to confirm the repo is non-empty; SHA is intentionally unused —
|
|
# POST /branches takes a branch name, not a SHA.
|
|
ref_response = await client.get(
|
|
f"{api_base}/repos/{owner}/{repo}/git/refs/heads/{default_branch}", headers=headers
|
|
)
|
|
if ref_response.status_code != 200:
|
|
return await self._create_initial_commit(client, headers, api_base, owner, repo, branch, files)
|
|
|
|
create_ref = await client.post(
|
|
f"{api_base}/repos/{owner}/{repo}/branches",
|
|
headers=headers,
|
|
json={"new_branch_name": branch, "old_ref_name": default_branch},
|
|
)
|
|
if create_ref.status_code == 403:
|
|
msg = f"Permission denied creating branch '{branch}' — token may lack write access to this repository"
|
|
logger.warning("_create_branch_and_push %s/%s: 403 %s", owner, repo, msg)
|
|
return {"status": "failed", "message": msg}
|
|
if create_ref.status_code == 409:
|
|
msg = f"Branch '{branch}' already exists (possible race condition)"
|
|
logger.warning("_create_branch_and_push %s/%s: 409 %s", owner, repo, msg)
|
|
return {"status": "failed", "message": msg}
|
|
if create_ref.status_code != 201:
|
|
msg = f"Failed to create branch '{branch}' (HTTP {create_ref.status_code}): {self._truncated_response_text(create_ref)}"
|
|
logger.warning("_create_branch_and_push %s/%s: %s", owner, repo, msg)
|
|
return {"status": "failed", "message": msg}
|
|
|
|
logger.info("Re-entering push_files after branch create %s/%s -> %s", owner, repo, branch)
|
|
return await self.push_files(repo_url, token, branch, files, client, _allow_branch_create=False)
|
|
|
|
except Exception as e:
|
|
logger.exception("_create_branch_and_push failed for %s/%s branch=%s", owner, repo, branch)
|
|
return {"status": "failed", "message": str(e), "error": str(e)}
|
|
|
|
async def _create_initial_commit(
|
|
self,
|
|
client: httpx.AsyncClient,
|
|
headers: dict,
|
|
api_base: str,
|
|
owner: str,
|
|
repo: str,
|
|
branch: str,
|
|
files: dict,
|
|
) -> dict:
|
|
"""Seed an empty Gitea repository via the Contents API.
|
|
|
|
Gitea's Git Data API requires the repository to have at least one
|
|
commit before it accepts blob/tree/commit writes; on an empty repo
|
|
every ``POST /git/blobs`` returns 404. The Contents API is the
|
|
documented bootstrap path: a single ``POST /repos/{owner}/{repo}/contents``
|
|
with a ``files`` array creates the initial commit and the target
|
|
branch in one round-trip (Gitea 1.18+, Forgejo all versions).
|
|
"""
|
|
try:
|
|
if not files:
|
|
return {"status": "skipped", "message": "No files to commit", "commit_sha": None, "files_changed": 0}
|
|
|
|
api_files = []
|
|
for path, content in files.items():
|
|
content_str = json.dumps(content, indent=2, default=str)
|
|
content_b64 = base64.b64encode(content_str.encode("utf-8")).decode()
|
|
api_files.append({"operation": "create", "path": path, "content": content_b64})
|
|
|
|
commit_message = f"Initial Bambuddy backup - {datetime.now(timezone.utc).strftime('%Y-%m-%d %H:%M:%S UTC')}"
|
|
body = {
|
|
"branch": branch,
|
|
"new_branch": branch,
|
|
"message": commit_message,
|
|
"files": api_files,
|
|
}
|
|
|
|
response = await client.post(
|
|
f"{api_base}/repos/{owner}/{repo}/contents",
|
|
headers=headers,
|
|
json=body,
|
|
)
|
|
|
|
if response.status_code not in (200, 201):
|
|
return {
|
|
"status": "failed",
|
|
"message": f"Failed to create initial commit: {self._truncated_response_text(response)}",
|
|
}
|
|
|
|
data = response.json()
|
|
commit_sha = (data.get("commit") or {}).get("sha")
|
|
message = (
|
|
f"Initial backup created - {len(files)} files"
|
|
if commit_sha
|
|
else f"Initial backup created - {len(files)} files (commit SHA not reported by server)"
|
|
)
|
|
return {
|
|
"status": "success",
|
|
"message": message,
|
|
"commit_sha": commit_sha,
|
|
"files_changed": len(files),
|
|
}
|
|
|
|
except Exception as e:
|
|
logger.exception("_create_initial_commit failed for %s/%s branch=%s", owner, repo, branch)
|
|
return {"status": "failed", "message": str(e), "error": str(e)}
|