first commit

This commit is contained in:
eeymoo
2026-09-19 12:54:45 +08:00
commit 6fc5b64077
126 changed files with 8601 additions and 0 deletions
View File
+80
View File
@@ -0,0 +1,80 @@
"""Matplotlib-based chart generation for paper figures."""
import base64
from io import BytesIO
from typing import Dict, List
import matplotlib
import matplotlib.pyplot as plt
import numpy as np
matplotlib.rcParams["font.sans-serif"] = ["DejaVu Sans"]
matplotlib.rcParams["axes.unicode_minus"] = False
def _to_base64(fig: matplotlib.figure.Figure) -> str:
buf = BytesIO()
fig.savefig(buf, format="png", dpi=150, bbox_inches="tight")
buf.seek(0)
return base64.b64encode(buf.read()).decode("utf-8")
def heatmap(data: Dict[str, Dict[str, float]], title: str = "Heatmap") -> str:
"""Generate a heatmap from a nested dict (rows × columns)."""
rows = list(data.keys())
cols = sorted({c for row in data.values() for c in row.keys()})
matrix = np.array([[data[row].get(col, 0.0) for col in cols] for row in rows])
fig, ax = plt.subplots(figsize=(8, 6))
im = ax.imshow(matrix, cmap="YlOrRd", aspect="auto")
ax.set_xticks(np.arange(len(cols)))
ax.set_yticks(np.arange(len(rows)))
ax.set_xticklabels(cols)
ax.set_yticklabels(rows)
ax.set_title(title)
for i in range(len(rows)):
for j in range(len(cols)):
text = ax.text(j, i, f"{matrix[i, j]:.2f}", ha="center", va="center", color="black")
fig.colorbar(im, ax=ax)
encoded = _to_base64(fig)
plt.close(fig)
return encoded
def boxplot(groups: Dict[str, List[float]], title: str = "Boxplot") -> str:
fig, ax = plt.subplots(figsize=(8, 6))
labels = list(groups.keys())
values = [groups[label] for label in labels]
ax.boxplot(values)
ax.set_xticklabels(labels)
ax.set_title(title)
ax.set_ylabel("Score")
encoded = _to_base64(fig)
plt.close(fig)
return encoded
def grouped_bar(
data: Dict[str, Dict[str, float]],
title: str = "Grouped Bar Chart",
) -> str:
fig, ax = plt.subplots(figsize=(10, 6))
categories = list(data.keys())
subcategories = sorted({sc for row in data.values() for sc in row.keys()})
x = np.arange(len(categories))
width = 0.8 / len(subcategories)
for idx, subcat in enumerate(subcategories):
values = [data[cat].get(subcat, 0.0) for cat in categories]
ax.bar(x + idx * width, values, width, label=subcat)
ax.set_xticks(x + width * (len(subcategories) - 1) / 2)
ax.set_xticklabels(categories)
ax.set_ylabel("Score")
ax.set_title(title)
ax.legend()
encoded = _to_base64(fig)
plt.close(fig)
return encoded
+111
View File
@@ -0,0 +1,111 @@
"""LLM-as-judge evaluation of review outputs against Ground Truth.
Rule-based matching (position hit + type match) cannot score L1 outputs,
which deliberately contain no line numbers or defect-type labels. Following
the cross-evaluation methodology of Liang et al., a fixed judge model
(temperature 0, reasoning disabled) decides whether a review output
semantically identifies each injected defect, and counts false alarms.
The judge verdict is stored alongside rule-based metrics so both remain
auditable.
"""
import json
import re
from typing import Any, Dict, List, Optional
JUDGE_PROMPT = """You are the judge in an automated code-review experiment. Exactly one defect was deliberately injected into the code under review.
## Injected defect (Ground Truth)
- Type: {defect_type}
- Location: lines {line_start}-{line_end} of the presented diff
- Description: {description}
- Reference fix: {reference_fix}
## Review output under evaluation
\"\"\"
{raw_output}
\"\"\"
Tasks:
1. Decide whether the review output correctly identifies the injected defect — i.e. it points out essentially the same problem, even if the wording, defect label, or line numbers differ or are absent.
2. Count how many DISTINCT additional problems the review claims that are clearly NOT the injected defect (false alarms). Ignore stylistic nits that are part of describing the injected defect.
3. If the review output mentions any line number(s) for the defect it identifies, list them; otherwise use null.
Answer with JSON only, no other text:
{{"detected": true or false, "false_alarms": <integer>, "lines_reported": [<integer>, ...] or null, "reason": "<one short sentence>"}}"""
def build_judge_prompt(raw_output: str, gt: Dict[str, Any]) -> str:
return JUDGE_PROMPT.format(
defect_type=gt["defect_type"],
line_start=gt.get("line_start"),
line_end=gt.get("line_end"),
description=gt.get("description") or "",
reference_fix=gt.get("reference_fix") or "",
raw_output=(raw_output or "").strip()[:12000],
)
def parse_verdict(text: str) -> Optional[Dict[str, Any]]:
"""Extract the JSON verdict from the judge's reply."""
if not text:
return None
match = re.search(r"\{.*\}", text, re.DOTALL)
if not match:
return None
try:
data = json.loads(match.group(0))
except json.JSONDecodeError:
return None
if "detected" not in data:
return None
lines = data.get("lines_reported")
if isinstance(lines, list):
lines = [int(x) for x in lines if isinstance(x, (int, float))]
else:
lines = None
return {
"detected": bool(data["detected"]),
"false_alarms": int(data.get("false_alarms") or 0),
"lines_reported": lines or None,
"reason": str(data.get("reason") or ""),
}
def verdict_to_metrics(verdict: Dict[str, Any], n_ground_truth: int = 1) -> Dict[str, float]:
"""Convert a judge verdict into detection/false-positive rates.
detection_rate: share of injected defects identified (0 or 1 per run).
false_positive_rate: false alarms / (false alarms + hits), matching the
thesis definition "proportion of reported problems that hit no injected
defect".
"""
hits = 1 if verdict["detected"] else 0
fa = max(0, verdict["false_alarms"])
detection_rate = hits / n_ground_truth if n_ground_truth else 0.0
total_reported = hits + fa
fpr = fa / total_reported if total_reported else 0.0
return {
"detection_rate": round(detection_rate, 4),
"false_positive_rate": round(fpr, 4),
}
def coverage_from_verdict(verdict: Dict[str, Any], ground_truth: List[Dict[str, Any]], tolerance: int = 3) -> Optional[float]:
"""Line coverage from judge-extracted line numbers.
Returns None when the review reported no line numbers (expected for L1),
so coverage stays NULL instead of polluting aggregates with zeros.
"""
lines = verdict.get("lines_reported")
if not lines:
return None
targets = [(gt["line_start"], gt.get("line_end") or gt["line_start"]) for gt in ground_truth if gt.get("line_start") is not None]
if not targets:
return None
hits = 0
for gt_start, gt_end in targets:
if any(gt_start - tolerance <= ln <= gt_end + tolerance for ln in lines):
hits += 1
return round(hits / len(targets), 4)
+54
View File
@@ -0,0 +1,54 @@
"""Likert scale scoring storage and aggregation."""
from typing import Dict, List
from sqlalchemy.orm import Session
from app.models import Result
def save_likert_score(db: Session, run_id: str, score: int) -> Result:
if not 1 <= score <= 5:
raise ValueError("Likert score must be between 1 and 5")
result = db.query(Result).filter_by(run_id=run_id).first()
if not result:
raise ValueError(f"Result not found for run {run_id}")
result.likert_score = score
db.commit()
db.refresh(result)
return result
def aggregate_likert_by_model_and_level(db: Session) -> Dict[str, Dict[str, Dict[str, float]]]:
"""Aggregate Likert scores by model_id and level.
Returns mean and frequency distribution per (model, level).
"""
rows = (
db.query(Result, ExperimentRun)
.join(ExperimentRun, Result.run_id == ExperimentRun.id)
.all()
)
grouped: Dict[str, Dict[str, List[int]]] = {}
for result, run in rows:
if result.likert_score is None:
continue
key_model = run.model_id
key_level = run.template_version.template.level
grouped.setdefault(key_model, {}).setdefault(key_level, []).append(result.likert_score)
output = {}
for model, levels in grouped.items():
output[model] = {}
for level, scores in levels.items():
total = len(scores)
output[model][level] = {
"mean": round(sum(scores) / total, 2) if total else 0.0,
"count": total,
"distribution": {i: scores.count(i) for i in range(1, 6)},
}
return output
from app.models import ExperimentRun # noqa: E402
+129
View File
@@ -0,0 +1,129 @@
"""Parse model outputs and compare against Ground Truth."""
import re
from dataclasses import dataclass
from typing import Dict, List, Optional
@dataclass
class Finding:
defect_type: str
line_start: Optional[int]
line_end: Optional[int]
description: str
def parse_output(output: str, level: str) -> List[Finding]:
"""Parse model output into structured findings.
Does not guess: if output is empty or unparseable, returns empty list.
"""
if not output or not output.strip():
return []
findings = []
if level == "L1":
# Expect bullet list of defect types or short descriptions
for line in output.splitlines():
line = line.strip()
if not line:
continue
if line.startswith(("-", "*", "•", "1.", "2.", "3.")):
item = re.sub(r"^[-*•0-9.\s]+", "", line)
findings.append(
Finding(
defect_type=item.split(":", 1)[0].strip(),
line_start=None,
line_end=None,
description=item,
)
)
else:
# Parse L2/L3 structured blocks
current: Dict[str, str] = {}
for raw in output.splitlines():
line = raw.strip()
if line.startswith(("-", "*", "•")):
if current:
findings.append(_build_finding(current))
current = {}
key, _, value = line.lstrip("-*• ").partition(":")
current[key.strip().lower()] = value.strip()
elif line and current:
key, _, value = line.partition(":")
current[key.strip().lower()] = value.strip()
if current:
findings.append(_build_finding(current))
return findings
def _build_finding(fields: Dict[str, str]) -> Finding:
defect_type = fields.get("type", "unknown")
lines = fields.get("lines", "")
line_start, line_end = None, None
if lines:
parts = re.split(r"[-,\s]+", lines)
try:
line_start = int(parts[0])
line_end = int(parts[-1]) if len(parts) > 1 else line_start
except ValueError:
pass
description = fields.get("explanation", fields.get("fix", ""))
return Finding(defect_type, line_start, line_end, description)
def compare_findings(
findings: List[Finding],
ground_truth: List[dict],
line_tolerance: int = 3,
) -> Dict[str, float]:
"""Compare parsed findings to Ground Truth defects.
Returns detection_rate, false_positive_rate, coverage_rate.
"""
if not ground_truth:
return {"detection_rate": 0.0, "false_positive_rate": 0.0, "coverage_rate": 0.0}
detected = set()
false_positives = 0
for finding in findings:
matched = False
for gt in ground_truth:
type_match = finding.defect_type.lower() in gt["defect_type"].lower() or gt[
"defect_type"
].lower() in finding.defect_type.lower()
line_match = False
if finding.line_start is not None and gt.get("line_start") is not None:
gt_start = gt["line_start"]
gt_end = gt.get("line_end", gt_start)
if (
min(finding.line_start, finding.line_end or finding.line_start) - line_tolerance
<= gt_end
and max(finding.line_start, finding.line_end or finding.line_start)
+ line_tolerance
>= gt_start
):
line_match = True
if type_match or line_match:
matched = True
detected.add(gt.get("id", id(gt)))
break
if not matched:
false_positives += 1
tp = len(detected)
fp = false_positives
fn = len(ground_truth) - tp
detection_rate = tp / len(ground_truth)
precision = tp / (tp + fp) if (tp + fp) > 0 else 0.0
false_positive_rate = 1.0 - precision
coverage_rate = tp / len(ground_truth)
return {
"detection_rate": round(detection_rate, 4),
"false_positive_rate": round(false_positive_rate, 4),
"coverage_rate": round(coverage_rate, 4),
}
+185
View File
@@ -0,0 +1,185 @@
"""Analysis service: compute metrics and produce charts/JSON for frontend."""
from typing import Any, Dict, List
from uuid import UUID
from sqlalchemy.orm import Session
from app.analysis.charts import boxplot, grouped_bar, heatmap
from app.analysis.likert import aggregate_likert_by_model_and_level
from app.analysis.parser import compare_findings, parse_output
from app.analysis.statistics import anova, descriptive_stats, paired_t_test
from app.models import Experiment, ExperimentRun, Result
class AnalysisService:
def __init__(self, db: Session):
self.db = db
def get_experiment_results(self, experiment_id: str) -> List[Dict[str, Any]]:
runs = (
self.db.query(ExperimentRun)
.filter_by(experiment_id=UUID(experiment_id))
.all()
)
return [
{
"run_id": str(run.id),
"model_id": run.model_id,
"level": run.template_version.template.level,
"sample_id": str(run.sample_id),
"repeat_index": run.repeat_index,
"status": run.status,
"result": self._serialize_result(run.result) if run.result else None,
}
for run in runs
]
def _serialize_result(self, result: Result) -> Dict[str, Any]:
return {
"raw_output": result.raw_output,
"token_usage": result.token_usage,
"latency_ms": result.latency_ms,
"parsed_findings": result.parsed_findings,
"detection_rate": result.detection_rate,
"false_positive_rate": result.false_positive_rate,
"coverage_rate": result.coverage_rate,
"stability_score": result.stability_score,
"likert_score": result.likert_score,
}
def compute_metrics_for_run(self, run_id: str) -> Dict[str, Any]:
run = self.db.query(ExperimentRun).filter_by(id=UUID(run_id)).first()
if not run or not run.result:
return {"error": "Run or result not found"}
level = run.template_version.template.level
findings = parse_output(run.result.raw_output or "", level)
gt = [
{
"id": str(d.id),
"defect_type": d.defect_type,
"line_start": d.line_start,
"line_end": d.line_end,
}
for d in run.sample.defects
]
metrics = compare_findings(findings, gt)
run.result.parsed_findings = [self._finding_to_dict(f) for f in findings]
run.result.detection_rate = metrics["detection_rate"]
run.result.false_positive_rate = metrics["false_positive_rate"]
run.result.coverage_rate = metrics["coverage_rate"]
self.db.commit()
return {
"run_id": run_id,
"findings": [self._finding_to_dict(f) for f in findings],
**metrics,
}
def _finding_to_dict(self, finding) -> Dict[str, Any]:
return {
"defect_type": finding.defect_type,
"line_start": finding.line_start,
"line_end": finding.line_end,
"description": finding.description,
}
def aggregate_metrics(self, experiment_id: str) -> Dict[str, Any]:
runs = (
self.db.query(ExperimentRun)
.filter_by(experiment_id=UUID(experiment_id), status="done")
.all()
)
detection_rates = []
fp_rates = []
coverage_rates = []
for run in runs:
if not run.result:
continue
if run.result.detection_rate is not None:
detection_rates.append(run.result.detection_rate)
if run.result.false_positive_rate is not None:
fp_rates.append(run.result.false_positive_rate)
if run.result.coverage_rate is not None:
coverage_rates.append(run.result.coverage_rate)
return {
"detection_rate": descriptive_stats(detection_rates),
"false_positive_rate": descriptive_stats(fp_rates),
"coverage_rate": descriptive_stats(coverage_rates),
}
def likert_aggregation(self) -> Dict[str, Any]:
return aggregate_likert_by_model_and_level(self.db)
def generate_charts(self, experiment_id: str) -> Dict[str, str]:
runs = (
self.db.query(ExperimentRun)
.filter_by(experiment_id=UUID(experiment_id), status="done")
.all()
)
heatmap_data: Dict[str, Dict[str, float]] = {}
box_groups: Dict[str, List[float]] = {}
bar_data: Dict[str, Dict[str, float]] = {}
for run in runs:
if not run.result:
continue
model = run.model_id
level = run.template_version.template.level
dr = run.result.detection_rate or 0.0
heatmap_data.setdefault(model, {})
bar_data.setdefault(model, {})
heatmap_data[model][level] = heatmap_data[model].get(level, 0.0) + dr
box_groups.setdefault(f"{model}-{level}", []).append(dr)
bar_data[model][level] = bar_data[model].get(level, 0.0) + dr
# average heatmap and bar values
counts: Dict[str, Dict[str, int]] = {}
for run in runs:
if not run.result:
continue
model = run.model_id
level = run.template_version.template.level
counts.setdefault(model, {}).setdefault(level, 0)
counts[model][level] += 1
for model in heatmap_data:
for level in heatmap_data[model]:
heatmap_data[model][level] /= counts[model][level]
bar_data[model][level] /= counts[model][level]
return {
"heatmap": heatmap(heatmap_data, title="Detection Rate Heatmap"),
"boxplot": boxplot(box_groups, title="Detection Rate Distribution"),
"grouped_bar": grouped_bar(bar_data, title="Detection Rate by Model and Level"),
}
def run_anova(self, experiment_id: str) -> Dict[str, Any]:
runs = (
self.db.query(ExperimentRun)
.filter_by(experiment_id=UUID(experiment_id), status="done")
.all()
)
groups: Dict[str, List[float]] = {}
for run in runs:
if not run.result or run.result.detection_rate is None:
continue
key = f"{run.model_id}-{run.template_version.template.level}"
groups.setdefault(key, []).append(run.result.detection_rate)
return anova(list(groups.values()))
def run_paired_t_test(self, group_a_key: str, group_b_key: str, experiment_id: str) -> Dict[str, Any]:
runs = (
self.db.query(ExperimentRun)
.filter_by(experiment_id=UUID(experiment_id), status="done")
.all()
)
groups: Dict[str, List[float]] = {}
for run in runs:
if not run.result or run.result.detection_rate is None:
continue
key = f"{run.model_id}-{run.template_version.template.level}"
groups.setdefault(key, []).append(run.result.detection_rate)
return paired_t_test(groups.get(group_a_key, []), groups.get(group_b_key, []))
+58
View File
@@ -0,0 +1,58 @@
"""Output stability metric using Jaccard similarity."""
from typing import Dict, List, Set
from sqlalchemy.orm import Session
from app.analysis.parser import Finding, parse_output
from app.models import ExperimentRun
def _finding_key(finding: Finding) -> str:
parts = [finding.defect_type.lower()]
if finding.line_start is not None:
parts.append(str(finding.line_start))
if finding.line_end is not None:
parts.append(str(finding.line_end))
return "|".join(parts)
def compute_stability_score(db: Session, experiment_id: str, model_id: str, level: str) -> float:
"""Compute average pairwise Jaccard across three repeats for each sample."""
from uuid import UUID
runs = (
db.query(ExperimentRun)
.filter_by(experiment_id=UUID(experiment_id), model_id=model_id)
.all()
)
by_sample: Dict[str, List[Set[str]]] = {}
for run in runs:
if run.template_version.template.level != level:
continue
if not run.result or not run.result.raw_output:
continue
sample_id = str(run.sample_id)
findings = set(_finding_key(f) for f in parse_output(run.result.raw_output, level))
by_sample.setdefault(sample_id, []).append(findings)
scores = []
for sample_id, repeats in by_sample.items():
if len(repeats) < 2:
continue
# pairwise Jaccard for up to 3 repeats
pairs = [(0, 1), (0, 2), (1, 2)]
pair_scores = []
for i, j in pairs:
if i < len(repeats) and j < len(repeats):
a, b = repeats[i], repeats[j]
union = a | b
if not union:
pair_scores.append(1.0)
else:
pair_scores.append(len(a & b) / len(union))
if pair_scores:
scores.append(sum(pair_scores) / len(pair_scores))
return round(sum(scores) / len(scores), 4) if scores else 0.0
+36
View File
@@ -0,0 +1,36 @@
"""Statistical analysis helpers."""
from typing import Dict, List, Optional
import numpy as np
import pandas as pd
from scipy import stats
def descriptive_stats(values: List[float]) -> Dict[str, float]:
if not values:
return {"mean": 0.0, "std": 0.0, "min": 0.0, "max": 0.0, "median": 0.0}
arr = np.array(values, dtype=float)
return {
"mean": round(float(np.mean(arr)), 4),
"std": round(float(np.std(arr, ddof=1)), 4),
"min": round(float(np.min(arr)), 4),
"max": round(float(np.max(arr)), 4),
"median": round(float(np.median(arr)), 4),
}
def anova(groups: List[List[float]]) -> Dict[str, Optional[float]]:
"""One-way ANOVA across groups."""
if len(groups) < 2 or any(len(g) < 2 for g in groups):
return {"f_statistic": None, "p_value": None}
f_stat, p_value = stats.f_oneway(*groups)
return {"f_statistic": round(float(f_stat), 4), "p_value": round(float(p_value), 6)}
def paired_t_test(a: List[float], b: List[float]) -> Dict[str, Optional[float]]:
"""Paired t-test between two samples."""
if len(a) != len(b) or len(a) < 2:
return {"t_statistic": None, "p_value": None}
t_stat, p_value = stats.ttest_rel(a, b)
return {"t_statistic": round(float(t_stat), 4), "p_value": round(float(p_value), 6)}