expand language_stats.py to include check of gitea repos
This commit is contained in:
+184
-58
@@ -6,29 +6,37 @@ in all repositories of a given account and prints a markdown
|
|||||||
table to stdout, delimited by the
|
table to stdout, delimited by the
|
||||||
markers <!--LANG_STATS_START--> … <!--LANG_STATS_END-->.
|
markers <!--LANG_STATS_START--> … <!--LANG_STATS_END-->.
|
||||||
|
|
||||||
|
When run as a systemd job, it checks for new commits since the last run
|
||||||
|
and only regenerates if there are changes. Use --force to always regenerate.
|
||||||
|
|
||||||
Required environment variables
|
Required environment variables
|
||||||
--------------------------------
|
--------------------------------
|
||||||
GITEA_BASE_URL e.g. https://gitea.example.com
|
GITEA_BASE_URL e.g. https://gitea.example.com
|
||||||
GITEA_TOKEN personal access-token with `repo` scope
|
GITEA_TOKEN personal access-token with `repo` scope
|
||||||
GITEA_USERNAME the account to analyse
|
GITEA_USERNAME the account to analyse
|
||||||
|
|
||||||
|
Optional environment variables
|
||||||
|
--------------------------------
|
||||||
|
STATE_FILE path to store last-run state (default: ~/.local/state/language_stats.json)
|
||||||
"""
|
"""
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
import os
|
import os
|
||||||
import sys
|
import sys
|
||||||
import requests
|
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import requests
|
||||||
from dotenv import load_dotenv
|
from dotenv import load_dotenv
|
||||||
|
|
||||||
load_dotenv()
|
load_dotenv()
|
||||||
|
|
||||||
BASE_URL = os.getenv("GITEA_BASE_URL", "").rstrip("/")
|
BASE_URL = os.getenv("GITEA_BASE_URL", "").rstrip("/")
|
||||||
TOKEN = os.getenv("GITEA_TOKEN")
|
TOKEN = os.getenv("GITEA_TOKEN")
|
||||||
USERNAME = os.getenv("GITEA_USERNAME")
|
USERNAME = os.getenv("GITEA_USERNAME")
|
||||||
|
|
||||||
if not BASE_URL or not TOKEN or not USERNAME:
|
DEFAULT_STATE_DIR = Path.home() / ".local" / "state"
|
||||||
sys.exit("GITEA_BASE_URL, GITEA_TOKEN and GITEA_USERNAME must be set")
|
STATE_FILE = Path(os.getenv("STATE_FILE", DEFAULT_STATE_DIR / "language_stats.json"))
|
||||||
|
|
||||||
session = requests.Session()
|
|
||||||
session.headers["Authorization"] = f"token {TOKEN}"
|
|
||||||
|
|
||||||
# List of files whose byte-counts should be excluded from the totals.
|
# List of files whose byte-counts should be excluded from the totals.
|
||||||
# Each tuple: (repository_name, file_path, language_name)
|
# Each tuple: (repository_name, file_path, language_name)
|
||||||
@@ -40,63 +48,181 @@ EXCLUDE_FILES: list[tuple[str, str, str]] = [
|
|||||||
("Kingsborough-LibGuide", "groups/home/bootstrap3/Vue3-v.5-css.html", "HTML"),
|
("Kingsborough-LibGuide", "groups/home/bootstrap3/Vue3-v.5-css.html", "HTML"),
|
||||||
("Kingsborough-LibGuide", "groups/home/bootstrap3/Vue3-v.6.html", "HTML"),
|
("Kingsborough-LibGuide", "groups/home/bootstrap3/Vue3-v.6.html", "HTML"),
|
||||||
("Kingsborough-LibGuide", "groups/home/bootstrap3/Vue3-v.6-css.html", "HTML"),
|
("Kingsborough-LibGuide", "groups/home/bootstrap3/Vue3-v.6-css.html", "HTML"),
|
||||||
|
("Kingsborough-LibGuide", "groups/home/bootstrap3/Vue3-v.7.html", "HTML"),
|
||||||
|
("Kingsborough-LibGuide", "groups/home/bootstrap3/Vue3-v.7-css.html", "HTML"),
|
||||||
]
|
]
|
||||||
|
|
||||||
# ---------- collect repositories ----------
|
|
||||||
page = 1
|
|
||||||
repos = []
|
|
||||||
while True:
|
|
||||||
resp = session.get(
|
|
||||||
f"{BASE_URL}/api/v1/users/{USERNAME}/repos",
|
|
||||||
params={"page": page, "limit": 50},
|
|
||||||
timeout=30,
|
|
||||||
)
|
|
||||||
resp.raise_for_status()
|
|
||||||
batch = resp.json()
|
|
||||||
if not batch:
|
|
||||||
break
|
|
||||||
repos.extend(batch)
|
|
||||||
page += 1
|
|
||||||
|
|
||||||
# ---------- aggregate language bytes ----------
|
def get_session() -> requests.Session:
|
||||||
totals: dict[str, int] = defaultdict(int)
|
"""Create an authenticated requests session."""
|
||||||
|
session = requests.Session()
|
||||||
|
session.headers["Authorization"] = f"token {TOKEN}"
|
||||||
|
return session
|
||||||
|
|
||||||
for repo in repos:
|
|
||||||
langs = session.get(
|
|
||||||
f"{BASE_URL}/api/v1/repos/{USERNAME}/{repo['name']}/languages",
|
|
||||||
timeout=30,
|
|
||||||
).json()
|
|
||||||
for lang, bytes_ in langs.items():
|
|
||||||
totals[lang] += bytes_
|
|
||||||
|
|
||||||
# ---------- subtract excluded files ----------
|
def fetch_repos(session: requests.Session) -> list[dict]:
|
||||||
for repo_name, file_path, language in EXCLUDE_FILES:
|
"""Fetch all repositories for the configured user."""
|
||||||
try:
|
page = 1
|
||||||
meta = session.get(
|
repos = []
|
||||||
f"{BASE_URL}/api/v1/repos/{USERNAME}/{repo_name}/contents/{file_path}",
|
while True:
|
||||||
|
resp = session.get(
|
||||||
|
f"{BASE_URL}/api/v1/users/{USERNAME}/repos",
|
||||||
|
params={"page": page, "limit": 50},
|
||||||
timeout=30,
|
timeout=30,
|
||||||
)
|
)
|
||||||
if meta.ok:
|
resp.raise_for_status()
|
||||||
size = meta.json().get("size", 0)
|
batch = resp.json()
|
||||||
totals[language] = max(0, totals[language] - size)
|
if not batch:
|
||||||
except requests.RequestException:
|
break
|
||||||
# Ignore failures – continue with best-effort stats
|
repos.extend(batch)
|
||||||
pass
|
page += 1
|
||||||
|
return repos
|
||||||
|
|
||||||
total_bytes = sum(totals.values())
|
|
||||||
if total_bytes == 0:
|
|
||||||
sys.exit("No language data returned from API.")
|
|
||||||
|
|
||||||
# ---------- build markdown table ----------
|
def get_repo_push_timestamps(repos: list[dict]) -> dict[str, str]:
|
||||||
pairs = sorted(totals.items(), key=lambda x: x[1], reverse=True)
|
"""Extract repository names and their pushed_at timestamps."""
|
||||||
|
return {repo["name"]: repo.get("pushed_at", "") for repo in repos}
|
||||||
|
|
||||||
lines = [
|
|
||||||
"## 📊 Languages\n",
|
|
||||||
]
|
|
||||||
for lang, nbytes in pairs:
|
|
||||||
pct = nbytes * 100 / total_bytes
|
|
||||||
bar = "█" * int(pct / 2) # up to 50 chars
|
|
||||||
lines.append(f"{bar} {lang}{pct:5.1f}%\\")
|
|
||||||
|
|
||||||
for line in lines:
|
def load_state() -> dict:
|
||||||
print(line)
|
"""Load the previous state from the state file."""
|
||||||
|
if STATE_FILE.exists():
|
||||||
|
try:
|
||||||
|
return json.loads(STATE_FILE.read_text())
|
||||||
|
except (json.JSONDecodeError, OSError):
|
||||||
|
return {}
|
||||||
|
return {}
|
||||||
|
|
||||||
|
|
||||||
|
def save_state(state: dict) -> None:
|
||||||
|
"""Save the current state to the state file."""
|
||||||
|
STATE_FILE.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
STATE_FILE.write_text(json.dumps(state, indent=2))
|
||||||
|
|
||||||
|
|
||||||
|
def has_new_commits(current_timestamps: dict[str, str], saved_state: dict) -> bool:
|
||||||
|
"""Check if any repository has new commits since last run."""
|
||||||
|
saved_timestamps = saved_state.get("repo_timestamps", {})
|
||||||
|
|
||||||
|
# Check for new or updated repos
|
||||||
|
for repo_name, pushed_at in current_timestamps.items():
|
||||||
|
if repo_name not in saved_timestamps:
|
||||||
|
return True
|
||||||
|
if pushed_at != saved_timestamps[repo_name]:
|
||||||
|
return True
|
||||||
|
|
||||||
|
# Check for deleted repos (also counts as a change)
|
||||||
|
for repo_name in saved_timestamps:
|
||||||
|
if repo_name not in current_timestamps:
|
||||||
|
return True
|
||||||
|
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def aggregate_language_bytes(session: requests.Session, repos: list[dict]) -> dict[str, int]:
|
||||||
|
"""Aggregate language byte counts across all repositories."""
|
||||||
|
totals: dict[str, int] = defaultdict(int)
|
||||||
|
|
||||||
|
for repo in repos:
|
||||||
|
langs = session.get(
|
||||||
|
f"{BASE_URL}/api/v1/repos/{USERNAME}/{repo['name']}/languages",
|
||||||
|
timeout=30,
|
||||||
|
).json()
|
||||||
|
for lang, bytes_ in langs.items():
|
||||||
|
totals[lang] += bytes_
|
||||||
|
|
||||||
|
return dict(totals)
|
||||||
|
|
||||||
|
|
||||||
|
def subtract_excluded_files(session: requests.Session, totals: dict[str, int]) -> dict[str, int]:
|
||||||
|
"""Subtract byte counts for explicitly excluded files."""
|
||||||
|
for repo_name, file_path, language in EXCLUDE_FILES:
|
||||||
|
try:
|
||||||
|
meta = session.get(
|
||||||
|
f"{BASE_URL}/api/v1/repos/{USERNAME}/{repo_name}/contents/{file_path}",
|
||||||
|
timeout=30,
|
||||||
|
)
|
||||||
|
if meta.ok:
|
||||||
|
size = meta.json().get("size", 0)
|
||||||
|
totals[language] = max(0, totals.get(language, 0) - size)
|
||||||
|
except requests.RequestException:
|
||||||
|
# Ignore failures – continue with best-effort stats
|
||||||
|
pass
|
||||||
|
return totals
|
||||||
|
|
||||||
|
|
||||||
|
def generate_markdown(totals: dict[str, int]) -> list[str]:
|
||||||
|
"""Generate markdown output for language statistics."""
|
||||||
|
total_bytes = sum(totals.values())
|
||||||
|
if total_bytes == 0:
|
||||||
|
return []
|
||||||
|
|
||||||
|
pairs = sorted(totals.items(), key=lambda x: x[1], reverse=True)
|
||||||
|
|
||||||
|
lines = ["## 📊 Languages\n"]
|
||||||
|
for lang, nbytes in pairs:
|
||||||
|
pct = nbytes * 100 / total_bytes
|
||||||
|
bar = "█" * int(pct / 2) # up to 50 chars
|
||||||
|
lines.append(f"{bar} {lang}{pct:5.1f}%\\")
|
||||||
|
|
||||||
|
return lines
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> int:
|
||||||
|
"""Main entry point for the script."""
|
||||||
|
parser = argparse.ArgumentParser(
|
||||||
|
description="Generate language statistics from Gitea repositories"
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--force",
|
||||||
|
action="store_true",
|
||||||
|
help="Force regeneration even if no new commits detected",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--check-only",
|
||||||
|
action="store_true",
|
||||||
|
help="Only check for new commits, don't generate output",
|
||||||
|
)
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
# Validate environment
|
||||||
|
if not BASE_URL or not TOKEN or not USERNAME:
|
||||||
|
sys.exit("GITEA_BASE_URL, GITEA_TOKEN and GITEA_USERNAME must be set")
|
||||||
|
|
||||||
|
session = get_session()
|
||||||
|
|
||||||
|
# Fetch current repository state
|
||||||
|
repos = fetch_repos(session)
|
||||||
|
current_timestamps = get_repo_push_timestamps(repos)
|
||||||
|
|
||||||
|
# Load previous state and check for changes
|
||||||
|
saved_state = load_state()
|
||||||
|
|
||||||
|
if not args.force and not has_new_commits(current_timestamps, saved_state):
|
||||||
|
print("No new commits detected. Skipping regeneration.", file=sys.stderr)
|
||||||
|
return 0
|
||||||
|
|
||||||
|
if args.check_only:
|
||||||
|
print("New commits detected.", file=sys.stderr)
|
||||||
|
return 0
|
||||||
|
|
||||||
|
# Generate the stats
|
||||||
|
totals = aggregate_language_bytes(session, repos)
|
||||||
|
totals = subtract_excluded_files(session, totals)
|
||||||
|
|
||||||
|
if sum(totals.values()) == 0:
|
||||||
|
sys.exit("No language data returned from API.")
|
||||||
|
|
||||||
|
lines = generate_markdown(totals)
|
||||||
|
for line in lines:
|
||||||
|
print(line)
|
||||||
|
|
||||||
|
# Save current state for next run
|
||||||
|
new_state = {"repo_timestamps": current_timestamps}
|
||||||
|
save_state(new_state)
|
||||||
|
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main())
|
||||||
|
|||||||
Reference in New Issue
Block a user