"""Git repository parsing and commit candidate selection.""" from dataclasses import dataclass from pathlib import Path from typing import List, Optional from git import Repo @dataclass class CommitCandidate: repo: str sha: str message: str author: str date: str stats: dict files: List[dict] def list_commits( repo_path: str | Path, max_count: Optional[int] = None, reverse: bool = True, ) -> List[CommitCandidate]: """List commits from a Git repository.""" repo = Repo(str(repo_path)) repo_name = Path(repo_path).name commits = [] iterator = list(repo.iter_commits()) if reverse: iterator = reversed(iterator) for commit in iterator: if max_count and len(commits) >= max_count: break stats = commit.stats.total files = [] for item in commit.stats.files.items(): filename, file_stats = item files.append( { "path": filename, "insertions": file_stats["insertions"], "deletions": file_stats["deletions"], "lines": file_stats["lines"], } ) commits.append( CommitCandidate( repo=repo_name, sha=commit.hexsha, message=commit.message.strip(), author=str(commit.author), date=commit.committed_datetime.isoformat(), stats=stats, files=files, ) ) return commits def score_commit(commit: CommitCandidate) -> float: """Score a commit by size, message quality, and language diversity.""" total_lines = commit.stats.get("lines", 0) # Prefer moderate size: ~50-200 lines ideal size_score = 1.0 - abs(total_lines - 125) / 200.0 size_score = max(0.0, min(1.0, size_score)) # Message quality: length and presence of verb/noun clues msg = commit.message.lower() msg_score = min(1.0, len(commit.message) / 40.0) if any(k in msg for k in ("fix", "bug", "refactor", "feature", "add", "update")): msg_score = min(1.0, msg_score + 0.2) # Language diversity bonus based on file extensions exts = {Path(f["path"]).suffix.lower() for f in commit.files if Path(f["path"]).suffix} diversity_score = min(1.0, len(exts) / 3.0) return size_score * 0.5 + msg_score * 0.3 + diversity_score * 0.2 def select_candidates( repo_path: str | Path, count: int = 12, languages: Optional[List[str]] = None, scan_limit: int = 300, ) -> List[CommitCandidate]: """Select top-scoring commits, optionally balanced by language. Only the most recent ``scan_limit`` commits are scanned: computing per-commit stats spawns a git subprocess each time, so scanning the full history of a large repository is prohibitively slow. """ languages = languages or ["python", "java", "javascript"] commits = list_commits(repo_path, max_count=scan_limit, reverse=False) scored = [(c, score_commit(c)) for c in commits] scored.sort(key=lambda x: x[1], reverse=True) # Simple balancing: prefer at least one commit per target language when detectable by_lang = {lang: [] for lang in languages} others = [] for commit, score in scored: ext_set = {Path(f["path"]).suffix.lower() for f in commit.files} placed = False for lang in languages: hint = ".py" if lang == "python" else ".java" if lang == "java" else ".js" if hint in ext_set: by_lang[lang].append((commit, score)) placed = True break if not placed: others.append((commit, score)) result = [] per_lang = max(1, count // len(languages)) for lang in languages: result.extend(by_lang[lang][:per_lang]) result.extend(others) result = result[:count] result.sort(key=lambda x: x[1], reverse=True) return [c for c, _ in result]