first commit
This commit is contained in:
@@ -0,0 +1,121 @@
|
||||
"""Git repository parsing and commit candidate selection."""
|
||||
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import List, Optional
|
||||
|
||||
from git import Repo
|
||||
|
||||
|
||||
@dataclass
|
||||
class CommitCandidate:
|
||||
repo: str
|
||||
sha: str
|
||||
message: str
|
||||
author: str
|
||||
date: str
|
||||
stats: dict
|
||||
files: List[dict]
|
||||
|
||||
|
||||
def list_commits(
|
||||
repo_path: str | Path,
|
||||
max_count: Optional[int] = None,
|
||||
reverse: bool = True,
|
||||
) -> List[CommitCandidate]:
|
||||
"""List commits from a Git repository."""
|
||||
repo = Repo(str(repo_path))
|
||||
repo_name = Path(repo_path).name
|
||||
commits = []
|
||||
iterator = list(repo.iter_commits())
|
||||
if reverse:
|
||||
iterator = reversed(iterator)
|
||||
for commit in iterator:
|
||||
if max_count and len(commits) >= max_count:
|
||||
break
|
||||
stats = commit.stats.total
|
||||
files = []
|
||||
for item in commit.stats.files.items():
|
||||
filename, file_stats = item
|
||||
files.append(
|
||||
{
|
||||
"path": filename,
|
||||
"insertions": file_stats["insertions"],
|
||||
"deletions": file_stats["deletions"],
|
||||
"lines": file_stats["lines"],
|
||||
}
|
||||
)
|
||||
commits.append(
|
||||
CommitCandidate(
|
||||
repo=repo_name,
|
||||
sha=commit.hexsha,
|
||||
message=commit.message.strip(),
|
||||
author=str(commit.author),
|
||||
date=commit.committed_datetime.isoformat(),
|
||||
stats=stats,
|
||||
files=files,
|
||||
)
|
||||
)
|
||||
return commits
|
||||
|
||||
|
||||
def score_commit(commit: CommitCandidate) -> float:
|
||||
"""Score a commit by size, message quality, and language diversity."""
|
||||
total_lines = commit.stats.get("lines", 0)
|
||||
# Prefer moderate size: ~50-200 lines ideal
|
||||
size_score = 1.0 - abs(total_lines - 125) / 200.0
|
||||
size_score = max(0.0, min(1.0, size_score))
|
||||
|
||||
# Message quality: length and presence of verb/noun clues
|
||||
msg = commit.message.lower()
|
||||
msg_score = min(1.0, len(commit.message) / 40.0)
|
||||
if any(k in msg for k in ("fix", "bug", "refactor", "feature", "add", "update")):
|
||||
msg_score = min(1.0, msg_score + 0.2)
|
||||
|
||||
# Language diversity bonus based on file extensions
|
||||
exts = {Path(f["path"]).suffix.lower() for f in commit.files if Path(f["path"]).suffix}
|
||||
diversity_score = min(1.0, len(exts) / 3.0)
|
||||
|
||||
return size_score * 0.5 + msg_score * 0.3 + diversity_score * 0.2
|
||||
|
||||
|
||||
def select_candidates(
|
||||
repo_path: str | Path,
|
||||
count: int = 12,
|
||||
languages: Optional[List[str]] = None,
|
||||
scan_limit: int = 300,
|
||||
) -> List[CommitCandidate]:
|
||||
"""Select top-scoring commits, optionally balanced by language.
|
||||
|
||||
Only the most recent ``scan_limit`` commits are scanned: computing
|
||||
per-commit stats spawns a git subprocess each time, so scanning the full
|
||||
history of a large repository is prohibitively slow.
|
||||
"""
|
||||
languages = languages or ["python", "java", "javascript"]
|
||||
commits = list_commits(repo_path, max_count=scan_limit, reverse=False)
|
||||
scored = [(c, score_commit(c)) for c in commits]
|
||||
scored.sort(key=lambda x: x[1], reverse=True)
|
||||
|
||||
# Simple balancing: prefer at least one commit per target language when detectable
|
||||
by_lang = {lang: [] for lang in languages}
|
||||
others = []
|
||||
for commit, score in scored:
|
||||
ext_set = {Path(f["path"]).suffix.lower() for f in commit.files}
|
||||
placed = False
|
||||
for lang in languages:
|
||||
hint = ".py" if lang == "python" else ".java" if lang == "java" else ".js"
|
||||
if hint in ext_set:
|
||||
by_lang[lang].append((commit, score))
|
||||
placed = True
|
||||
break
|
||||
if not placed:
|
||||
others.append((commit, score))
|
||||
|
||||
result = []
|
||||
per_lang = max(1, count // len(languages))
|
||||
for lang in languages:
|
||||
result.extend(by_lang[lang][:per_lang])
|
||||
result.extend(others)
|
||||
result = result[:count]
|
||||
result.sort(key=lambda x: x[1], reverse=True)
|
||||
return [c for c, _ in result]
|
||||
Reference in New Issue
Block a user