Files
jmoore-skild 0be6ccd090 fix(backup): stop Gitea's tree pager failing open on a missing total_count (#2656)
`if not isinstance(total, int) or seen >= total or not entries: return blobs, ""`
— the first arm short-circuited the page loop into a **success** holding page 1
only. Gitea clamps `per_page` to `MAX_RESPONSE_ITEMS` (default 50), so that is
50 entries of an arbitrarily large tree returned as a complete listing.

The restore then reports genuinely-present categories as "Not present in this
backup commit". That silent skip is the exact failure this override exists to
prevent, and the same class as E7 and G2 — G2 fixed the arithmetic here and
left the shape. GitHub and GitLab both hard-fail in the equivalent spot; only
Gitea guessed, and it guessed in the one direction that loses data quietly.
Whether Gitea always sends `total_count` on this route is beside the point: the
code was defending against a response shape it did not trust, and then trusting
it.

Now a missing or non-int `total_count` means "page until a short or empty
page". A page shorter than the first one is the last one, floored at Gitea's
default clamp so a genuinely small tree still costs exactly one request — the
reason G2 rejected paging-until-short in the `total_count`-present case, which
is unchanged and still stops on the count. The existing `page <= 50` ceiling
gives the correct hard failure for a tree that really is over cap, so this
cannot truncate.

Residual, and deliberately not widened into a `return None` on the first
ambiguous response — that would break single-page trees, the common case: an
instance whose `MAX_RESPONSE_ITEMS` is set *below* 50 *and* which omits
`total_count` would still stop at page 1. Both halves have to be true.

Tests: +6 (paged to the end with no count, on both Gitea and Forgejo; a short
page ends it; a non-int count is treated as no count; the page ceiling still
fails). Fail-pre-fix 5, control that passes either way 1 (a small tree is one
request). These don't match `-k github`, so 274 -> 280 across the three restore
files but `-k github` is unmoved.
2026-08-04 08:57:34 -04:00

479 lines
22 KiB
Python

"""Gitea backend — overrides GitHubBackend where Gitea's API diverges."""
import base64
import json
import logging
import re
from datetime import datetime, timezone
import httpx
from backend.app.services.git_providers.github import GitHubBackend
logger = logging.getLogger(__name__)
# Gitea clamps per_page to MAX_RESPONSE_ITEMS, which defaults to 50. Consulted
# only when a tree response carries no usable total_count: a page at least this
# long may be a clamped full page and cannot be assumed to be the last one.
_ASSUMED_MIN_PAGE_SIZE = 50
class GiteaBackend(GitHubBackend):
"""Backend for Gitea instances.
Gitea's Git Data API (/api/v1/repos/{owner}/{repo}/git/...) is *mostly*
compatible with GitHub's, but diverges on three points that broke real-world
backups (#1224, #1225, #1239):
1. ``GET /git/refs/heads/{branch}`` returns a *list* of matching refs even
when only one matches; GitHub returns a single object. The push paths
below extract the SHA via ``_ref_sha()`` instead of the GitHub-style
``["object"]["sha"]`` chain.
2. The Git Data API (blobs/trees/commits/refs) refuses writes against an
empty repository — every blob POST returns 404 until the repo has at
least one commit. ``_create_initial_commit()`` is overridden to use the
Contents API, which seeds the branch + initial commit in a single call.
3. The Git Data API does not support atomic multi-file commits — each file
requires a separate blob POST followed by a tree/commit/ref sequence.
``push_files()`` is overridden to use the Contents API
(``POST /repos/.../contents`` with a ``files`` array), which commits all
changed files in a single round-trip and avoids partial-commit failures.
"""
@staticmethod
def _ref_sha(ref_data) -> str:
"""Extract the commit SHA from Gitea's list-shaped ref response."""
if isinstance(ref_data, list):
if not ref_data:
raise ValueError("Empty refs list returned by Gitea API")
return ref_data[0]["object"]["sha"]
return ref_data["object"]["sha"]
@staticmethod
def _commit_tree_sha(commit_data: dict) -> str | None:
"""Extract the tree SHA from a commit response.
GitHub's ``GET /git/commits/{sha}`` returns the GitCommit schema with
``tree`` at the top level. Gitea's same-named endpoint may return the
wrapped Commit schema where ``tree`` lives under ``commit``. Try the
flat shape first (GitHub-compatible deployments and some Gitea/Forgejo
versions) then fall back to the wrapped shape.
"""
tree_node = commit_data.get("tree")
if not isinstance(tree_node, dict):
tree_node = (commit_data.get("commit") or {}).get("tree")
if isinstance(tree_node, dict):
return tree_node.get("sha")
return None
# Gitea/Forgejo can be hosted under a URL path prefix (ROOT_URL like
# https://host/gitea), so the repo lives at /<prefix...>/<owner>/<repo>
# rather than at the host root (#2642). Capture the scheme+host+prefix as
# one group and the final two path segments as owner/repo; the lazy prefix
# group is empty for a root-hosted instance. One shared pattern keeps
# parse_repo_url() and get_api_base() from drifting.
_HTTPS_REPO_RE = re.compile(
r"(https?://[\w.\-]+(?::\d+)?(?:/[\w.\-]+)*?)/([\w.\-]{1,100})/([\w.\-]{1,100})(?:\.git)?/?$"
)
def parse_repo_url(self, url: str) -> tuple[str, str]:
"""Return (owner, repo) — accepts both https:// and http:// for self-hosted instances."""
if not url or len(url) > 500:
raise ValueError("Invalid Git URL: URL too long or empty")
match = self._HTTPS_REPO_RE.match(url)
if match:
return match.group(2), match.group(3).removesuffix(".git")
match = re.match(
r"git@[\w.\-]+:([\w.\-]{1,100})/([\w.\-]{1,100})(?:\.git)?$",
url,
)
if match:
return match.group(1), match.group(2).removesuffix(".git")
raise ValueError(f"Cannot parse repository URL: {url}")
def get_api_base(self, repo_url: str) -> str:
"""Derive API base from the repository URL's scheme, host and any path prefix."""
match = self._HTTPS_REPO_RE.match(repo_url)
if match:
return f"{match.group(1)}/api/v1"
raise ValueError(f"Cannot derive API base from URL: {repo_url}")
def get_headers(self, token: str) -> dict:
headers = super().get_headers(token)
headers["Accept"] = "application/json"
return headers
async def _blob_shas_at(
self,
client: httpx.AsyncClient,
headers: dict,
api_base: str,
owner: str,
repo: str,
ref: str,
) -> tuple[dict[str, str] | None, str]:
"""Paged override of GitHub's single-GET tree read (#2656).
Divergence four, alongside the three in the class docstring. GitHub's
recursive trees endpoint is not paginated and signals overflow with
``truncated: true``, which the inherited implementation hard-fails on.
Gitea and Forgejo *do* page the same endpoint — ``page``/``per_page``,
with ``total_count`` alongside the tree — so the inherited version would
read only the first page and then report every category beyond it as
absent from the commit. A restore that silently skips categories is the
exact failure the GitHub version refuses to allow, so this pages instead.
The cap mirrors GitLab's: reaching it means there are more pages, and
that is a failure rather than a partial result. Because the page size is
the server's choice rather than ours (see below), the cap is a page count
and not a file count.
"""
blobs: dict[str, str] = {}
seen = 0
page = 1
page_size: int | None = None
while page <= 50:
response = await client.get(
f"{api_base}/repos/{owner}/{repo}/git/trees/{ref}",
headers=headers,
params={"recursive": "true", "page": page, "per_page": 1000},
)
if response.status_code == 404:
return None, f"Commit or tree '{ref}' not found in the repository"
if response.status_code != 200:
return None, (
f"Failed to list tree (HTTP {response.status_code}): {self._truncated_response_text(response)}"
)
try:
data = response.json()
except ValueError:
return None, "Non-JSON response listing tree"
if not isinstance(data, dict):
return None, "Unexpected shape listing tree"
entries = data.get("tree")
if not isinstance(entries, list):
entries = []
for item in entries:
if not isinstance(item, dict) or item.get("type") != "blob":
continue
path, sha = item.get("path"), item.get("sha")
if isinstance(path, str) and isinstance(sha, str) and path and sha:
blobs[path] = sha
# total_count counts every entry, trees included, so compare against
# what came back rather than against len(blobs).
#
# Count what the server actually returned, never the per_page we
# asked for: Gitea clamps per_page to MAX_RESPONSE_ITEMS, which
# defaults to 50. Deriving the offset from the requested 1000 made
# page 2 report 1050 entries seen, which clears any total_count below
# that — so the loop stopped and returned the first two pages of a
# much larger tree as a success. The restore then read every missing
# path as "category not present in this commit" and skipped it
# silently, the exact failure this override exists to prevent.
total = data.get("total_count")
seen += len(entries)
if page_size is None:
page_size = max(len(entries), _ASSUMED_MIN_PAGE_SIZE)
if not entries:
return blobs, ""
if isinstance(total, int):
if seen >= total:
return blobs, ""
elif len(entries) < page_size:
# No usable total_count. This used to return here on the *first*
# page, i.e. fail open into a success holding whatever one page
# happened to be — 50 entries of an arbitrarily large tree under
# the default clamp — and the restore then reported every
# category beyond it as absent from the commit. Page until a
# short or empty page instead; the page-count ceiling below
# still gives the correct hard failure for a tree that really is
# too large. A page shorter than the first one (or than Gitea's
# default clamp, so a genuinely small tree stays one request)
# cannot be followed by another. The residual case is an
# instance whose MAX_RESPONSE_ITEMS is set *below* 50 and which
# also omits total_count; real Gitea and Forgejo always send it
# on this route.
return blobs, ""
page += 1
return None, (
"Repository tree exceeds the listing limit, so the backup contents cannot be "
"enumerated reliably. Rotate the backup repository."
)
async def push_files(
self,
repo_url: str,
token: str,
branch: str,
files: dict,
client: httpx.AsyncClient,
_allow_branch_create: bool = True,
) -> dict:
"""Push files via the Git Data API, normalising Gitea's list-shaped ref response."""
try:
owner, repo = self.parse_repo_url(repo_url)
api_base = self.get_api_base(repo_url)
headers = self.get_headers(token)
ref_response = await client.get(f"{api_base}/repos/{owner}/{repo}/git/refs/heads/{branch}", headers=headers)
if ref_response.status_code == 404:
if not _allow_branch_create:
return {
"status": "failed",
"message": (
f"Branch '{branch}' not found after creation — possible replication lag. "
"The next scheduled backup will retry."
),
}
return await self._create_branch_and_push(
client, headers, api_base, owner, repo, branch, files, repo_url, token
)
if ref_response.status_code != 200:
return {
"status": "failed",
"message": f"Failed to get branch ref: {ref_response.status_code}",
"error": self._truncated_response_text(ref_response),
}
current_commit_sha = self._ref_sha(ref_response.json())
commit_response = await client.get(
f"{api_base}/repos/{owner}/{repo}/git/commits/{current_commit_sha}", headers=headers
)
if commit_response.status_code != 200:
msg = f"Failed to get current commit (HTTP {commit_response.status_code}): {self._truncated_response_text(commit_response)}"
logger.warning("push_files %s/%s: %s", owner, repo, msg)
return {"status": "failed", "message": msg}
current_tree_sha = self._commit_tree_sha(commit_response.json())
if not current_tree_sha:
msg = (
f"Failed to extract tree SHA from commit response: {self._truncated_response_text(commit_response)}"
)
logger.warning("push_files %s/%s: %s", owner, repo, msg)
return {"status": "failed", "message": msg}
tree_response = await client.get(
f"{api_base}/repos/{owner}/{repo}/git/trees/{current_tree_sha}?recursive=1", headers=headers
)
if tree_response.status_code != 200:
msg = f"Failed to list existing tree (HTTP {tree_response.status_code}): {self._truncated_response_text(tree_response)}"
logger.warning("push_files %s/%s: %s", owner, repo, msg)
return {"status": "failed", "message": msg, "error": self._truncated_response_text(tree_response)}
tree_data = tree_response.json()
# Gitea's tree API can report ``truncated: true`` for large
# listings; if we honour the partial map, the dedup check misses
# and every file gets re-uploaded each run.
if tree_data.get("truncated"):
msg = (
"Repository tree exceeds the Gitea API listing limit (truncated=true). "
"Rotate the backup repository to avoid silent file-by-file churn on every backup."
)
logger.warning("push_files %s/%s: %s", owner, repo, msg)
return {"status": "failed", "message": msg}
existing_files: dict[str, str] = {}
for item in tree_data.get("tree", []):
if item.get("type") != "blob":
continue
path, sha = item.get("path"), item.get("sha")
if not path or not sha:
logger.warning("push_files: skipping malformed tree entry: %s", item)
continue
existing_files[path] = sha
api_files = []
files_changed = 0
for path, content in files.items():
content_str = json.dumps(content, indent=2, default=str)
content_bytes = content_str.encode("utf-8")
content_b64 = base64.b64encode(content_bytes).decode()
content_sha = self._blob_sha(content_bytes)
if path in existing_files:
if existing_files[path] == content_sha:
continue
api_files.append(
{"operation": "update", "path": path, "content": content_b64, "sha": existing_files[path]}
)
else:
api_files.append({"operation": "create", "path": path, "content": content_b64})
files_changed += 1
if not api_files:
return {"status": "skipped", "message": "No changes to commit", "commit_sha": None, "files_changed": 0}
commit_message = f"Bambuddy backup - {datetime.now(timezone.utc).strftime('%Y-%m-%d %H:%M:%S UTC')}"
response = await client.post(
f"{api_base}/repos/{owner}/{repo}/contents",
headers=headers,
json={"branch": branch, "message": commit_message, "files": api_files},
)
if response.status_code == 404:
return {
"status": "failed",
"message": "Contents API endpoint not found — your Gitea instance may be older than v1.18 or the API may be disabled by an administrator (POST /contents returned 404)",
}
if response.status_code == 409:
return {
"status": "failed",
"message": (
"Conflict committing files — the branch likely advanced concurrently "
"(web-UI edit, another backup run, or path-vs-tree collision). "
"The next scheduled backup will re-read the current tree and resolve this."
),
}
if response.status_code not in (200, 201):
return {
"status": "failed",
"message": f"Backup commit failed: {self._truncated_response_text(response)}",
}
commit_sha = (response.json().get("commit") or {}).get("sha")
message = (
f"Backup successful - {files_changed} files updated"
if commit_sha
else f"Backup successful - {files_changed} files updated (commit SHA not reported by server)"
)
return {
"status": "success",
"message": message,
"commit_sha": commit_sha,
"files_changed": files_changed,
}
except Exception as e:
logger.exception("push_files failed for %s branch=%s", repo_url, branch)
return {"status": "failed", "message": str(e), "error": str(e)}
async def _create_branch_and_push(
self,
client: httpx.AsyncClient,
headers: dict,
api_base: str,
owner: str,
repo: str,
branch: str,
files: dict,
repo_url: str,
token: str,
) -> dict:
"""Create branch (from default branch or as initial commit) then push."""
try:
repo_response = await client.get(f"{api_base}/repos/{owner}/{repo}", headers=headers)
if repo_response.status_code != 200:
msg = f"Failed to get repo info (HTTP {repo_response.status_code}): {self._truncated_response_text(repo_response)}"
logger.warning("_create_branch_and_push %s/%s: %s", owner, repo, msg)
return {"status": "failed", "message": msg}
default_branch = repo_response.json().get("default_branch", "main")
# GET the default branch to confirm the repo is non-empty; SHA is intentionally unused —
# POST /branches takes a branch name, not a SHA.
ref_response = await client.get(
f"{api_base}/repos/{owner}/{repo}/git/refs/heads/{default_branch}", headers=headers
)
if ref_response.status_code != 200:
return await self._create_initial_commit(client, headers, api_base, owner, repo, branch, files)
create_ref = await client.post(
f"{api_base}/repos/{owner}/{repo}/branches",
headers=headers,
json={"new_branch_name": branch, "old_ref_name": default_branch},
)
if create_ref.status_code == 403:
msg = f"Permission denied creating branch '{branch}' — token may lack write access to this repository"
logger.warning("_create_branch_and_push %s/%s: 403 %s", owner, repo, msg)
return {"status": "failed", "message": msg}
if create_ref.status_code == 409:
msg = f"Branch '{branch}' already exists (possible race condition)"
logger.warning("_create_branch_and_push %s/%s: 409 %s", owner, repo, msg)
return {"status": "failed", "message": msg}
if create_ref.status_code != 201:
msg = f"Failed to create branch '{branch}' (HTTP {create_ref.status_code}): {self._truncated_response_text(create_ref)}"
logger.warning("_create_branch_and_push %s/%s: %s", owner, repo, msg)
return {"status": "failed", "message": msg}
logger.info("Re-entering push_files after branch create %s/%s -> %s", owner, repo, branch)
return await self.push_files(repo_url, token, branch, files, client, _allow_branch_create=False)
except Exception as e:
logger.exception("_create_branch_and_push failed for %s/%s branch=%s", owner, repo, branch)
return {"status": "failed", "message": str(e), "error": str(e)}
async def _create_initial_commit(
self,
client: httpx.AsyncClient,
headers: dict,
api_base: str,
owner: str,
repo: str,
branch: str,
files: dict,
) -> dict:
"""Seed an empty Gitea repository via the Contents API.
Gitea's Git Data API requires the repository to have at least one
commit before it accepts blob/tree/commit writes; on an empty repo
every ``POST /git/blobs`` returns 404. The Contents API is the
documented bootstrap path: a single ``POST /repos/{owner}/{repo}/contents``
with a ``files`` array creates the initial commit and the target
branch in one round-trip (Gitea 1.18+, Forgejo all versions).
"""
try:
if not files:
return {"status": "skipped", "message": "No files to commit", "commit_sha": None, "files_changed": 0}
api_files = []
for path, content in files.items():
content_str = json.dumps(content, indent=2, default=str)
content_b64 = base64.b64encode(content_str.encode("utf-8")).decode()
api_files.append({"operation": "create", "path": path, "content": content_b64})
commit_message = f"Initial Bambuddy backup - {datetime.now(timezone.utc).strftime('%Y-%m-%d %H:%M:%S UTC')}"
body = {
"branch": branch,
"new_branch": branch,
"message": commit_message,
"files": api_files,
}
response = await client.post(
f"{api_base}/repos/{owner}/{repo}/contents",
headers=headers,
json=body,
)
if response.status_code not in (200, 201):
return {
"status": "failed",
"message": f"Failed to create initial commit: {self._truncated_response_text(response)}",
}
data = response.json()
commit_sha = (data.get("commit") or {}).get("sha")
message = (
f"Initial backup created - {len(files)} files"
if commit_sha
else f"Initial backup created - {len(files)} files (commit SHA not reported by server)"
)
return {
"status": "success",
"message": message,
"commit_sha": commit_sha,
"files_changed": len(files),
}
except Exception as e:
logger.exception("_create_initial_commit failed for %s/%s branch=%s", owner, repo, branch)
return {"status": "failed", "message": str(e), "error": str(e)}