expand language_stats.py to include check of gitea repos

This commit is contained in:
Mark Eaton
2026-01-10 20:23:15 -05:00
parent bee54f56b6
commit 478efb72b7
+184 -58
View File
@@ -6,29 +6,37 @@ in all repositories of a given account and prints a markdown
table to stdout, delimited by the
markers <!--LANG_STATS_START--> … <!--LANG_STATS_END-->.
When run as a systemd job, it checks for new commits since the last run
and only regenerates if there are changes. Use --force to always regenerate.
Required environment variables
--------------------------------
GITEA_BASE_URL e.g. https://gitea.example.com
GITEA_TOKEN personal access-token with `repo` scope
GITEA_USERNAME the account to analyse
Optional environment variables
--------------------------------
STATE_FILE path to store last-run state (default: ~/.local/state/language_stats.json)
"""
import argparse
import json
import os
import sys
import requests
from collections import defaultdict
from pathlib import Path
import requests
from dotenv import load_dotenv
load_dotenv()
BASE_URL = os.getenv("GITEA_BASE_URL", "").rstrip("/")
TOKEN = os.getenv("GITEA_TOKEN")
USERNAME = os.getenv("GITEA_USERNAME")
BASE_URL = os.getenv("GITEA_BASE_URL", "").rstrip("/")
TOKEN = os.getenv("GITEA_TOKEN")
USERNAME = os.getenv("GITEA_USERNAME")
if not BASE_URL or not TOKEN or not USERNAME:
sys.exit("GITEA_BASE_URL, GITEA_TOKEN and GITEA_USERNAME must be set")
session = requests.Session()
session.headers["Authorization"] = f"token {TOKEN}"
DEFAULT_STATE_DIR = Path.home() / ".local" / "state"
STATE_FILE = Path(os.getenv("STATE_FILE", DEFAULT_STATE_DIR / "language_stats.json"))
# List of files whose byte-counts should be excluded from the totals.
# Each tuple: (repository_name, file_path, language_name)
@@ -40,63 +48,181 @@ EXCLUDE_FILES: list[tuple[str, str, str]] = [
("Kingsborough-LibGuide", "groups/home/bootstrap3/Vue3-v.5-css.html", "HTML"),
("Kingsborough-LibGuide", "groups/home/bootstrap3/Vue3-v.6.html", "HTML"),
("Kingsborough-LibGuide", "groups/home/bootstrap3/Vue3-v.6-css.html", "HTML"),
("Kingsborough-LibGuide", "groups/home/bootstrap3/Vue3-v.7.html", "HTML"),
("Kingsborough-LibGuide", "groups/home/bootstrap3/Vue3-v.7-css.html", "HTML"),
]
# ---------- collect repositories ----------
page = 1
repos = []
while True:
resp = session.get(
f"{BASE_URL}/api/v1/users/{USERNAME}/repos",
params={"page": page, "limit": 50},
timeout=30,
)
resp.raise_for_status()
batch = resp.json()
if not batch:
break
repos.extend(batch)
page += 1
# ---------- aggregate language bytes ----------
totals: dict[str, int] = defaultdict(int)
def get_session() -> requests.Session:
"""Create an authenticated requests session."""
session = requests.Session()
session.headers["Authorization"] = f"token {TOKEN}"
return session
for repo in repos:
langs = session.get(
f"{BASE_URL}/api/v1/repos/{USERNAME}/{repo['name']}/languages",
timeout=30,
).json()
for lang, bytes_ in langs.items():
totals[lang] += bytes_
# ---------- subtract excluded files ----------
for repo_name, file_path, language in EXCLUDE_FILES:
try:
meta = session.get(
f"{BASE_URL}/api/v1/repos/{USERNAME}/{repo_name}/contents/{file_path}",
def fetch_repos(session: requests.Session) -> list[dict]:
"""Fetch all repositories for the configured user."""
page = 1
repos = []
while True:
resp = session.get(
f"{BASE_URL}/api/v1/users/{USERNAME}/repos",
params={"page": page, "limit": 50},
timeout=30,
)
if meta.ok:
size = meta.json().get("size", 0)
totals[language] = max(0, totals[language] - size)
except requests.RequestException:
# Ignore failures continue with best-effort stats
pass
resp.raise_for_status()
batch = resp.json()
if not batch:
break
repos.extend(batch)
page += 1
return repos
total_bytes = sum(totals.values())
if total_bytes == 0:
sys.exit("No language data returned from API.")
# ---------- build markdown table ----------
pairs = sorted(totals.items(), key=lambda x: x[1], reverse=True)
def get_repo_push_timestamps(repos: list[dict]) -> dict[str, str]:
"""Extract repository names and their pushed_at timestamps."""
return {repo["name"]: repo.get("pushed_at", "") for repo in repos}
lines = [
"## 📊 Languages\n",
]
for lang, nbytes in pairs:
pct = nbytes * 100 / total_bytes
bar = "" * int(pct / 2) # up to 50 chars
lines.append(f"{bar} {lang}{pct:5.1f}%\\")
for line in lines:
print(line)
def load_state() -> dict:
"""Load the previous state from the state file."""
if STATE_FILE.exists():
try:
return json.loads(STATE_FILE.read_text())
except (json.JSONDecodeError, OSError):
return {}
return {}
def save_state(state: dict) -> None:
"""Save the current state to the state file."""
STATE_FILE.parent.mkdir(parents=True, exist_ok=True)
STATE_FILE.write_text(json.dumps(state, indent=2))
def has_new_commits(current_timestamps: dict[str, str], saved_state: dict) -> bool:
"""Check if any repository has new commits since last run."""
saved_timestamps = saved_state.get("repo_timestamps", {})
# Check for new or updated repos
for repo_name, pushed_at in current_timestamps.items():
if repo_name not in saved_timestamps:
return True
if pushed_at != saved_timestamps[repo_name]:
return True
# Check for deleted repos (also counts as a change)
for repo_name in saved_timestamps:
if repo_name not in current_timestamps:
return True
return False
def aggregate_language_bytes(session: requests.Session, repos: list[dict]) -> dict[str, int]:
"""Aggregate language byte counts across all repositories."""
totals: dict[str, int] = defaultdict(int)
for repo in repos:
langs = session.get(
f"{BASE_URL}/api/v1/repos/{USERNAME}/{repo['name']}/languages",
timeout=30,
).json()
for lang, bytes_ in langs.items():
totals[lang] += bytes_
return dict(totals)
def subtract_excluded_files(session: requests.Session, totals: dict[str, int]) -> dict[str, int]:
"""Subtract byte counts for explicitly excluded files."""
for repo_name, file_path, language in EXCLUDE_FILES:
try:
meta = session.get(
f"{BASE_URL}/api/v1/repos/{USERNAME}/{repo_name}/contents/{file_path}",
timeout=30,
)
if meta.ok:
size = meta.json().get("size", 0)
totals[language] = max(0, totals.get(language, 0) - size)
except requests.RequestException:
# Ignore failures continue with best-effort stats
pass
return totals
def generate_markdown(totals: dict[str, int]) -> list[str]:
"""Generate markdown output for language statistics."""
total_bytes = sum(totals.values())
if total_bytes == 0:
return []
pairs = sorted(totals.items(), key=lambda x: x[1], reverse=True)
lines = ["## 📊 Languages\n"]
for lang, nbytes in pairs:
pct = nbytes * 100 / total_bytes
bar = "" * int(pct / 2) # up to 50 chars
lines.append(f"{bar} {lang}{pct:5.1f}%\\")
return lines
def main() -> int:
"""Main entry point for the script."""
parser = argparse.ArgumentParser(
description="Generate language statistics from Gitea repositories"
)
parser.add_argument(
"--force",
action="store_true",
help="Force regeneration even if no new commits detected",
)
parser.add_argument(
"--check-only",
action="store_true",
help="Only check for new commits, don't generate output",
)
args = parser.parse_args()
# Validate environment
if not BASE_URL or not TOKEN or not USERNAME:
sys.exit("GITEA_BASE_URL, GITEA_TOKEN and GITEA_USERNAME must be set")
session = get_session()
# Fetch current repository state
repos = fetch_repos(session)
current_timestamps = get_repo_push_timestamps(repos)
# Load previous state and check for changes
saved_state = load_state()
if not args.force and not has_new_commits(current_timestamps, saved_state):
print("No new commits detected. Skipping regeneration.", file=sys.stderr)
return 0
if args.check_only:
print("New commits detected.", file=sys.stderr)
return 0
# Generate the stats
totals = aggregate_language_bytes(session, repos)
totals = subtract_excluded_files(session, totals)
if sum(totals.values()) == 0:
sys.exit("No language data returned from API.")
lines = generate_markdown(totals)
for line in lines:
print(line)
# Save current state for next run
new_state = {"repo_timestamps": current_timestamps}
save_state(new_state)
return 0
if __name__ == "__main__":
sys.exit(main())