first commit
This commit is contained in:
@@ -0,0 +1,80 @@
|
||||
"""Matplotlib-based chart generation for paper figures."""
|
||||
|
||||
import base64
|
||||
from io import BytesIO
|
||||
from typing import Dict, List
|
||||
|
||||
import matplotlib
|
||||
import matplotlib.pyplot as plt
|
||||
import numpy as np
|
||||
|
||||
matplotlib.rcParams["font.sans-serif"] = ["DejaVu Sans"]
|
||||
matplotlib.rcParams["axes.unicode_minus"] = False
|
||||
|
||||
|
||||
def _to_base64(fig: matplotlib.figure.Figure) -> str:
|
||||
buf = BytesIO()
|
||||
fig.savefig(buf, format="png", dpi=150, bbox_inches="tight")
|
||||
buf.seek(0)
|
||||
return base64.b64encode(buf.read()).decode("utf-8")
|
||||
|
||||
|
||||
def heatmap(data: Dict[str, Dict[str, float]], title: str = "Heatmap") -> str:
|
||||
"""Generate a heatmap from a nested dict (rows × columns)."""
|
||||
rows = list(data.keys())
|
||||
cols = sorted({c for row in data.values() for c in row.keys()})
|
||||
matrix = np.array([[data[row].get(col, 0.0) for col in cols] for row in rows])
|
||||
|
||||
fig, ax = plt.subplots(figsize=(8, 6))
|
||||
im = ax.imshow(matrix, cmap="YlOrRd", aspect="auto")
|
||||
ax.set_xticks(np.arange(len(cols)))
|
||||
ax.set_yticks(np.arange(len(rows)))
|
||||
ax.set_xticklabels(cols)
|
||||
ax.set_yticklabels(rows)
|
||||
ax.set_title(title)
|
||||
|
||||
for i in range(len(rows)):
|
||||
for j in range(len(cols)):
|
||||
text = ax.text(j, i, f"{matrix[i, j]:.2f}", ha="center", va="center", color="black")
|
||||
|
||||
fig.colorbar(im, ax=ax)
|
||||
encoded = _to_base64(fig)
|
||||
plt.close(fig)
|
||||
return encoded
|
||||
|
||||
|
||||
def boxplot(groups: Dict[str, List[float]], title: str = "Boxplot") -> str:
|
||||
fig, ax = plt.subplots(figsize=(8, 6))
|
||||
labels = list(groups.keys())
|
||||
values = [groups[label] for label in labels]
|
||||
ax.boxplot(values)
|
||||
ax.set_xticklabels(labels)
|
||||
ax.set_title(title)
|
||||
ax.set_ylabel("Score")
|
||||
encoded = _to_base64(fig)
|
||||
plt.close(fig)
|
||||
return encoded
|
||||
|
||||
|
||||
def grouped_bar(
|
||||
data: Dict[str, Dict[str, float]],
|
||||
title: str = "Grouped Bar Chart",
|
||||
) -> str:
|
||||
fig, ax = plt.subplots(figsize=(10, 6))
|
||||
categories = list(data.keys())
|
||||
subcategories = sorted({sc for row in data.values() for sc in row.keys()})
|
||||
x = np.arange(len(categories))
|
||||
width = 0.8 / len(subcategories)
|
||||
|
||||
for idx, subcat in enumerate(subcategories):
|
||||
values = [data[cat].get(subcat, 0.0) for cat in categories]
|
||||
ax.bar(x + idx * width, values, width, label=subcat)
|
||||
|
||||
ax.set_xticks(x + width * (len(subcategories) - 1) / 2)
|
||||
ax.set_xticklabels(categories)
|
||||
ax.set_ylabel("Score")
|
||||
ax.set_title(title)
|
||||
ax.legend()
|
||||
encoded = _to_base64(fig)
|
||||
plt.close(fig)
|
||||
return encoded
|
||||
@@ -0,0 +1,111 @@
|
||||
"""LLM-as-judge evaluation of review outputs against Ground Truth.
|
||||
|
||||
Rule-based matching (position hit + type match) cannot score L1 outputs,
|
||||
which deliberately contain no line numbers or defect-type labels. Following
|
||||
the cross-evaluation methodology of Liang et al., a fixed judge model
|
||||
(temperature 0, reasoning disabled) decides whether a review output
|
||||
semantically identifies each injected defect, and counts false alarms.
|
||||
|
||||
The judge verdict is stored alongside rule-based metrics so both remain
|
||||
auditable.
|
||||
"""
|
||||
|
||||
import json
|
||||
import re
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
JUDGE_PROMPT = """You are the judge in an automated code-review experiment. Exactly one defect was deliberately injected into the code under review.
|
||||
|
||||
## Injected defect (Ground Truth)
|
||||
- Type: {defect_type}
|
||||
- Location: lines {line_start}-{line_end} of the presented diff
|
||||
- Description: {description}
|
||||
- Reference fix: {reference_fix}
|
||||
|
||||
## Review output under evaluation
|
||||
\"\"\"
|
||||
{raw_output}
|
||||
\"\"\"
|
||||
|
||||
Tasks:
|
||||
1. Decide whether the review output correctly identifies the injected defect — i.e. it points out essentially the same problem, even if the wording, defect label, or line numbers differ or are absent.
|
||||
2. Count how many DISTINCT additional problems the review claims that are clearly NOT the injected defect (false alarms). Ignore stylistic nits that are part of describing the injected defect.
|
||||
3. If the review output mentions any line number(s) for the defect it identifies, list them; otherwise use null.
|
||||
|
||||
Answer with JSON only, no other text:
|
||||
{{"detected": true or false, "false_alarms": <integer>, "lines_reported": [<integer>, ...] or null, "reason": "<one short sentence>"}}"""
|
||||
|
||||
|
||||
def build_judge_prompt(raw_output: str, gt: Dict[str, Any]) -> str:
|
||||
return JUDGE_PROMPT.format(
|
||||
defect_type=gt["defect_type"],
|
||||
line_start=gt.get("line_start"),
|
||||
line_end=gt.get("line_end"),
|
||||
description=gt.get("description") or "",
|
||||
reference_fix=gt.get("reference_fix") or "",
|
||||
raw_output=(raw_output or "").strip()[:12000],
|
||||
)
|
||||
|
||||
|
||||
def parse_verdict(text: str) -> Optional[Dict[str, Any]]:
|
||||
"""Extract the JSON verdict from the judge's reply."""
|
||||
if not text:
|
||||
return None
|
||||
match = re.search(r"\{.*\}", text, re.DOTALL)
|
||||
if not match:
|
||||
return None
|
||||
try:
|
||||
data = json.loads(match.group(0))
|
||||
except json.JSONDecodeError:
|
||||
return None
|
||||
if "detected" not in data:
|
||||
return None
|
||||
lines = data.get("lines_reported")
|
||||
if isinstance(lines, list):
|
||||
lines = [int(x) for x in lines if isinstance(x, (int, float))]
|
||||
else:
|
||||
lines = None
|
||||
return {
|
||||
"detected": bool(data["detected"]),
|
||||
"false_alarms": int(data.get("false_alarms") or 0),
|
||||
"lines_reported": lines or None,
|
||||
"reason": str(data.get("reason") or ""),
|
||||
}
|
||||
|
||||
|
||||
def verdict_to_metrics(verdict: Dict[str, Any], n_ground_truth: int = 1) -> Dict[str, float]:
|
||||
"""Convert a judge verdict into detection/false-positive rates.
|
||||
|
||||
detection_rate: share of injected defects identified (0 or 1 per run).
|
||||
false_positive_rate: false alarms / (false alarms + hits), matching the
|
||||
thesis definition "proportion of reported problems that hit no injected
|
||||
defect".
|
||||
"""
|
||||
hits = 1 if verdict["detected"] else 0
|
||||
fa = max(0, verdict["false_alarms"])
|
||||
detection_rate = hits / n_ground_truth if n_ground_truth else 0.0
|
||||
total_reported = hits + fa
|
||||
fpr = fa / total_reported if total_reported else 0.0
|
||||
return {
|
||||
"detection_rate": round(detection_rate, 4),
|
||||
"false_positive_rate": round(fpr, 4),
|
||||
}
|
||||
|
||||
|
||||
def coverage_from_verdict(verdict: Dict[str, Any], ground_truth: List[Dict[str, Any]], tolerance: int = 3) -> Optional[float]:
|
||||
"""Line coverage from judge-extracted line numbers.
|
||||
|
||||
Returns None when the review reported no line numbers (expected for L1),
|
||||
so coverage stays NULL instead of polluting aggregates with zeros.
|
||||
"""
|
||||
lines = verdict.get("lines_reported")
|
||||
if not lines:
|
||||
return None
|
||||
targets = [(gt["line_start"], gt.get("line_end") or gt["line_start"]) for gt in ground_truth if gt.get("line_start") is not None]
|
||||
if not targets:
|
||||
return None
|
||||
hits = 0
|
||||
for gt_start, gt_end in targets:
|
||||
if any(gt_start - tolerance <= ln <= gt_end + tolerance for ln in lines):
|
||||
hits += 1
|
||||
return round(hits / len(targets), 4)
|
||||
@@ -0,0 +1,54 @@
|
||||
"""Likert scale scoring storage and aggregation."""
|
||||
|
||||
from typing import Dict, List
|
||||
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from app.models import Result
|
||||
|
||||
|
||||
def save_likert_score(db: Session, run_id: str, score: int) -> Result:
|
||||
if not 1 <= score <= 5:
|
||||
raise ValueError("Likert score must be between 1 and 5")
|
||||
result = db.query(Result).filter_by(run_id=run_id).first()
|
||||
if not result:
|
||||
raise ValueError(f"Result not found for run {run_id}")
|
||||
result.likert_score = score
|
||||
db.commit()
|
||||
db.refresh(result)
|
||||
return result
|
||||
|
||||
|
||||
def aggregate_likert_by_model_and_level(db: Session) -> Dict[str, Dict[str, Dict[str, float]]]:
|
||||
"""Aggregate Likert scores by model_id and level.
|
||||
|
||||
Returns mean and frequency distribution per (model, level).
|
||||
"""
|
||||
rows = (
|
||||
db.query(Result, ExperimentRun)
|
||||
.join(ExperimentRun, Result.run_id == ExperimentRun.id)
|
||||
.all()
|
||||
)
|
||||
|
||||
grouped: Dict[str, Dict[str, List[int]]] = {}
|
||||
for result, run in rows:
|
||||
if result.likert_score is None:
|
||||
continue
|
||||
key_model = run.model_id
|
||||
key_level = run.template_version.template.level
|
||||
grouped.setdefault(key_model, {}).setdefault(key_level, []).append(result.likert_score)
|
||||
|
||||
output = {}
|
||||
for model, levels in grouped.items():
|
||||
output[model] = {}
|
||||
for level, scores in levels.items():
|
||||
total = len(scores)
|
||||
output[model][level] = {
|
||||
"mean": round(sum(scores) / total, 2) if total else 0.0,
|
||||
"count": total,
|
||||
"distribution": {i: scores.count(i) for i in range(1, 6)},
|
||||
}
|
||||
return output
|
||||
|
||||
|
||||
from app.models import ExperimentRun # noqa: E402
|
||||
@@ -0,0 +1,129 @@
|
||||
"""Parse model outputs and compare against Ground Truth."""
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
|
||||
@dataclass
|
||||
class Finding:
|
||||
defect_type: str
|
||||
line_start: Optional[int]
|
||||
line_end: Optional[int]
|
||||
description: str
|
||||
|
||||
|
||||
def parse_output(output: str, level: str) -> List[Finding]:
|
||||
"""Parse model output into structured findings.
|
||||
|
||||
Does not guess: if output is empty or unparseable, returns empty list.
|
||||
"""
|
||||
if not output or not output.strip():
|
||||
return []
|
||||
|
||||
findings = []
|
||||
if level == "L1":
|
||||
# Expect bullet list of defect types or short descriptions
|
||||
for line in output.splitlines():
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
if line.startswith(("-", "*", "•", "1.", "2.", "3.")):
|
||||
item = re.sub(r"^[-*•0-9.\s]+", "", line)
|
||||
findings.append(
|
||||
Finding(
|
||||
defect_type=item.split(":", 1)[0].strip(),
|
||||
line_start=None,
|
||||
line_end=None,
|
||||
description=item,
|
||||
)
|
||||
)
|
||||
else:
|
||||
# Parse L2/L3 structured blocks
|
||||
current: Dict[str, str] = {}
|
||||
for raw in output.splitlines():
|
||||
line = raw.strip()
|
||||
if line.startswith(("-", "*", "•")):
|
||||
if current:
|
||||
findings.append(_build_finding(current))
|
||||
current = {}
|
||||
key, _, value = line.lstrip("-*• ").partition(":")
|
||||
current[key.strip().lower()] = value.strip()
|
||||
elif line and current:
|
||||
key, _, value = line.partition(":")
|
||||
current[key.strip().lower()] = value.strip()
|
||||
if current:
|
||||
findings.append(_build_finding(current))
|
||||
|
||||
return findings
|
||||
|
||||
|
||||
def _build_finding(fields: Dict[str, str]) -> Finding:
|
||||
defect_type = fields.get("type", "unknown")
|
||||
lines = fields.get("lines", "")
|
||||
line_start, line_end = None, None
|
||||
if lines:
|
||||
parts = re.split(r"[-,\s]+", lines)
|
||||
try:
|
||||
line_start = int(parts[0])
|
||||
line_end = int(parts[-1]) if len(parts) > 1 else line_start
|
||||
except ValueError:
|
||||
pass
|
||||
description = fields.get("explanation", fields.get("fix", ""))
|
||||
return Finding(defect_type, line_start, line_end, description)
|
||||
|
||||
|
||||
def compare_findings(
|
||||
findings: List[Finding],
|
||||
ground_truth: List[dict],
|
||||
line_tolerance: int = 3,
|
||||
) -> Dict[str, float]:
|
||||
"""Compare parsed findings to Ground Truth defects.
|
||||
|
||||
Returns detection_rate, false_positive_rate, coverage_rate.
|
||||
"""
|
||||
if not ground_truth:
|
||||
return {"detection_rate": 0.0, "false_positive_rate": 0.0, "coverage_rate": 0.0}
|
||||
|
||||
detected = set()
|
||||
false_positives = 0
|
||||
|
||||
for finding in findings:
|
||||
matched = False
|
||||
for gt in ground_truth:
|
||||
type_match = finding.defect_type.lower() in gt["defect_type"].lower() or gt[
|
||||
"defect_type"
|
||||
].lower() in finding.defect_type.lower()
|
||||
line_match = False
|
||||
if finding.line_start is not None and gt.get("line_start") is not None:
|
||||
gt_start = gt["line_start"]
|
||||
gt_end = gt.get("line_end", gt_start)
|
||||
if (
|
||||
min(finding.line_start, finding.line_end or finding.line_start) - line_tolerance
|
||||
<= gt_end
|
||||
and max(finding.line_start, finding.line_end or finding.line_start)
|
||||
+ line_tolerance
|
||||
>= gt_start
|
||||
):
|
||||
line_match = True
|
||||
if type_match or line_match:
|
||||
matched = True
|
||||
detected.add(gt.get("id", id(gt)))
|
||||
break
|
||||
if not matched:
|
||||
false_positives += 1
|
||||
|
||||
tp = len(detected)
|
||||
fp = false_positives
|
||||
fn = len(ground_truth) - tp
|
||||
|
||||
detection_rate = tp / len(ground_truth)
|
||||
precision = tp / (tp + fp) if (tp + fp) > 0 else 0.0
|
||||
false_positive_rate = 1.0 - precision
|
||||
coverage_rate = tp / len(ground_truth)
|
||||
|
||||
return {
|
||||
"detection_rate": round(detection_rate, 4),
|
||||
"false_positive_rate": round(false_positive_rate, 4),
|
||||
"coverage_rate": round(coverage_rate, 4),
|
||||
}
|
||||
@@ -0,0 +1,185 @@
|
||||
"""Analysis service: compute metrics and produce charts/JSON for frontend."""
|
||||
|
||||
from typing import Any, Dict, List
|
||||
from uuid import UUID
|
||||
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from app.analysis.charts import boxplot, grouped_bar, heatmap
|
||||
from app.analysis.likert import aggregate_likert_by_model_and_level
|
||||
from app.analysis.parser import compare_findings, parse_output
|
||||
from app.analysis.statistics import anova, descriptive_stats, paired_t_test
|
||||
from app.models import Experiment, ExperimentRun, Result
|
||||
|
||||
|
||||
class AnalysisService:
|
||||
def __init__(self, db: Session):
|
||||
self.db = db
|
||||
|
||||
def get_experiment_results(self, experiment_id: str) -> List[Dict[str, Any]]:
|
||||
runs = (
|
||||
self.db.query(ExperimentRun)
|
||||
.filter_by(experiment_id=UUID(experiment_id))
|
||||
.all()
|
||||
)
|
||||
return [
|
||||
{
|
||||
"run_id": str(run.id),
|
||||
"model_id": run.model_id,
|
||||
"level": run.template_version.template.level,
|
||||
"sample_id": str(run.sample_id),
|
||||
"repeat_index": run.repeat_index,
|
||||
"status": run.status,
|
||||
"result": self._serialize_result(run.result) if run.result else None,
|
||||
}
|
||||
for run in runs
|
||||
]
|
||||
|
||||
def _serialize_result(self, result: Result) -> Dict[str, Any]:
|
||||
return {
|
||||
"raw_output": result.raw_output,
|
||||
"token_usage": result.token_usage,
|
||||
"latency_ms": result.latency_ms,
|
||||
"parsed_findings": result.parsed_findings,
|
||||
"detection_rate": result.detection_rate,
|
||||
"false_positive_rate": result.false_positive_rate,
|
||||
"coverage_rate": result.coverage_rate,
|
||||
"stability_score": result.stability_score,
|
||||
"likert_score": result.likert_score,
|
||||
}
|
||||
|
||||
def compute_metrics_for_run(self, run_id: str) -> Dict[str, Any]:
|
||||
run = self.db.query(ExperimentRun).filter_by(id=UUID(run_id)).first()
|
||||
if not run or not run.result:
|
||||
return {"error": "Run or result not found"}
|
||||
|
||||
level = run.template_version.template.level
|
||||
findings = parse_output(run.result.raw_output or "", level)
|
||||
gt = [
|
||||
{
|
||||
"id": str(d.id),
|
||||
"defect_type": d.defect_type,
|
||||
"line_start": d.line_start,
|
||||
"line_end": d.line_end,
|
||||
}
|
||||
for d in run.sample.defects
|
||||
]
|
||||
metrics = compare_findings(findings, gt)
|
||||
|
||||
run.result.parsed_findings = [self._finding_to_dict(f) for f in findings]
|
||||
run.result.detection_rate = metrics["detection_rate"]
|
||||
run.result.false_positive_rate = metrics["false_positive_rate"]
|
||||
run.result.coverage_rate = metrics["coverage_rate"]
|
||||
self.db.commit()
|
||||
|
||||
return {
|
||||
"run_id": run_id,
|
||||
"findings": [self._finding_to_dict(f) for f in findings],
|
||||
**metrics,
|
||||
}
|
||||
|
||||
def _finding_to_dict(self, finding) -> Dict[str, Any]:
|
||||
return {
|
||||
"defect_type": finding.defect_type,
|
||||
"line_start": finding.line_start,
|
||||
"line_end": finding.line_end,
|
||||
"description": finding.description,
|
||||
}
|
||||
|
||||
def aggregate_metrics(self, experiment_id: str) -> Dict[str, Any]:
|
||||
runs = (
|
||||
self.db.query(ExperimentRun)
|
||||
.filter_by(experiment_id=UUID(experiment_id), status="done")
|
||||
.all()
|
||||
)
|
||||
detection_rates = []
|
||||
fp_rates = []
|
||||
coverage_rates = []
|
||||
for run in runs:
|
||||
if not run.result:
|
||||
continue
|
||||
if run.result.detection_rate is not None:
|
||||
detection_rates.append(run.result.detection_rate)
|
||||
if run.result.false_positive_rate is not None:
|
||||
fp_rates.append(run.result.false_positive_rate)
|
||||
if run.result.coverage_rate is not None:
|
||||
coverage_rates.append(run.result.coverage_rate)
|
||||
|
||||
return {
|
||||
"detection_rate": descriptive_stats(detection_rates),
|
||||
"false_positive_rate": descriptive_stats(fp_rates),
|
||||
"coverage_rate": descriptive_stats(coverage_rates),
|
||||
}
|
||||
|
||||
def likert_aggregation(self) -> Dict[str, Any]:
|
||||
return aggregate_likert_by_model_and_level(self.db)
|
||||
|
||||
def generate_charts(self, experiment_id: str) -> Dict[str, str]:
|
||||
runs = (
|
||||
self.db.query(ExperimentRun)
|
||||
.filter_by(experiment_id=UUID(experiment_id), status="done")
|
||||
.all()
|
||||
)
|
||||
heatmap_data: Dict[str, Dict[str, float]] = {}
|
||||
box_groups: Dict[str, List[float]] = {}
|
||||
bar_data: Dict[str, Dict[str, float]] = {}
|
||||
|
||||
for run in runs:
|
||||
if not run.result:
|
||||
continue
|
||||
model = run.model_id
|
||||
level = run.template_version.template.level
|
||||
dr = run.result.detection_rate or 0.0
|
||||
heatmap_data.setdefault(model, {})
|
||||
bar_data.setdefault(model, {})
|
||||
heatmap_data[model][level] = heatmap_data[model].get(level, 0.0) + dr
|
||||
box_groups.setdefault(f"{model}-{level}", []).append(dr)
|
||||
bar_data[model][level] = bar_data[model].get(level, 0.0) + dr
|
||||
|
||||
# average heatmap and bar values
|
||||
counts: Dict[str, Dict[str, int]] = {}
|
||||
for run in runs:
|
||||
if not run.result:
|
||||
continue
|
||||
model = run.model_id
|
||||
level = run.template_version.template.level
|
||||
counts.setdefault(model, {}).setdefault(level, 0)
|
||||
counts[model][level] += 1
|
||||
for model in heatmap_data:
|
||||
for level in heatmap_data[model]:
|
||||
heatmap_data[model][level] /= counts[model][level]
|
||||
bar_data[model][level] /= counts[model][level]
|
||||
|
||||
return {
|
||||
"heatmap": heatmap(heatmap_data, title="Detection Rate Heatmap"),
|
||||
"boxplot": boxplot(box_groups, title="Detection Rate Distribution"),
|
||||
"grouped_bar": grouped_bar(bar_data, title="Detection Rate by Model and Level"),
|
||||
}
|
||||
|
||||
def run_anova(self, experiment_id: str) -> Dict[str, Any]:
|
||||
runs = (
|
||||
self.db.query(ExperimentRun)
|
||||
.filter_by(experiment_id=UUID(experiment_id), status="done")
|
||||
.all()
|
||||
)
|
||||
groups: Dict[str, List[float]] = {}
|
||||
for run in runs:
|
||||
if not run.result or run.result.detection_rate is None:
|
||||
continue
|
||||
key = f"{run.model_id}-{run.template_version.template.level}"
|
||||
groups.setdefault(key, []).append(run.result.detection_rate)
|
||||
return anova(list(groups.values()))
|
||||
|
||||
def run_paired_t_test(self, group_a_key: str, group_b_key: str, experiment_id: str) -> Dict[str, Any]:
|
||||
runs = (
|
||||
self.db.query(ExperimentRun)
|
||||
.filter_by(experiment_id=UUID(experiment_id), status="done")
|
||||
.all()
|
||||
)
|
||||
groups: Dict[str, List[float]] = {}
|
||||
for run in runs:
|
||||
if not run.result or run.result.detection_rate is None:
|
||||
continue
|
||||
key = f"{run.model_id}-{run.template_version.template.level}"
|
||||
groups.setdefault(key, []).append(run.result.detection_rate)
|
||||
return paired_t_test(groups.get(group_a_key, []), groups.get(group_b_key, []))
|
||||
@@ -0,0 +1,58 @@
|
||||
"""Output stability metric using Jaccard similarity."""
|
||||
|
||||
from typing import Dict, List, Set
|
||||
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from app.analysis.parser import Finding, parse_output
|
||||
from app.models import ExperimentRun
|
||||
|
||||
|
||||
def _finding_key(finding: Finding) -> str:
|
||||
parts = [finding.defect_type.lower()]
|
||||
if finding.line_start is not None:
|
||||
parts.append(str(finding.line_start))
|
||||
if finding.line_end is not None:
|
||||
parts.append(str(finding.line_end))
|
||||
return "|".join(parts)
|
||||
|
||||
|
||||
def compute_stability_score(db: Session, experiment_id: str, model_id: str, level: str) -> float:
|
||||
"""Compute average pairwise Jaccard across three repeats for each sample."""
|
||||
from uuid import UUID
|
||||
|
||||
runs = (
|
||||
db.query(ExperimentRun)
|
||||
.filter_by(experiment_id=UUID(experiment_id), model_id=model_id)
|
||||
.all()
|
||||
)
|
||||
|
||||
by_sample: Dict[str, List[Set[str]]] = {}
|
||||
for run in runs:
|
||||
if run.template_version.template.level != level:
|
||||
continue
|
||||
if not run.result or not run.result.raw_output:
|
||||
continue
|
||||
sample_id = str(run.sample_id)
|
||||
findings = set(_finding_key(f) for f in parse_output(run.result.raw_output, level))
|
||||
by_sample.setdefault(sample_id, []).append(findings)
|
||||
|
||||
scores = []
|
||||
for sample_id, repeats in by_sample.items():
|
||||
if len(repeats) < 2:
|
||||
continue
|
||||
# pairwise Jaccard for up to 3 repeats
|
||||
pairs = [(0, 1), (0, 2), (1, 2)]
|
||||
pair_scores = []
|
||||
for i, j in pairs:
|
||||
if i < len(repeats) and j < len(repeats):
|
||||
a, b = repeats[i], repeats[j]
|
||||
union = a | b
|
||||
if not union:
|
||||
pair_scores.append(1.0)
|
||||
else:
|
||||
pair_scores.append(len(a & b) / len(union))
|
||||
if pair_scores:
|
||||
scores.append(sum(pair_scores) / len(pair_scores))
|
||||
|
||||
return round(sum(scores) / len(scores), 4) if scores else 0.0
|
||||
@@ -0,0 +1,36 @@
|
||||
"""Statistical analysis helpers."""
|
||||
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
from scipy import stats
|
||||
|
||||
|
||||
def descriptive_stats(values: List[float]) -> Dict[str, float]:
|
||||
if not values:
|
||||
return {"mean": 0.0, "std": 0.0, "min": 0.0, "max": 0.0, "median": 0.0}
|
||||
arr = np.array(values, dtype=float)
|
||||
return {
|
||||
"mean": round(float(np.mean(arr)), 4),
|
||||
"std": round(float(np.std(arr, ddof=1)), 4),
|
||||
"min": round(float(np.min(arr)), 4),
|
||||
"max": round(float(np.max(arr)), 4),
|
||||
"median": round(float(np.median(arr)), 4),
|
||||
}
|
||||
|
||||
|
||||
def anova(groups: List[List[float]]) -> Dict[str, Optional[float]]:
|
||||
"""One-way ANOVA across groups."""
|
||||
if len(groups) < 2 or any(len(g) < 2 for g in groups):
|
||||
return {"f_statistic": None, "p_value": None}
|
||||
f_stat, p_value = stats.f_oneway(*groups)
|
||||
return {"f_statistic": round(float(f_stat), 4), "p_value": round(float(p_value), 6)}
|
||||
|
||||
|
||||
def paired_t_test(a: List[float], b: List[float]) -> Dict[str, Optional[float]]:
|
||||
"""Paired t-test between two samples."""
|
||||
if len(a) != len(b) or len(a) < 2:
|
||||
return {"t_statistic": None, "p_value": None}
|
||||
t_stat, p_value = stats.ttest_rel(a, b)
|
||||
return {"t_statistic": round(float(t_stat), 4), "p_value": round(float(p_value), 6)}
|
||||
Reference in New Issue
Block a user