Use strict type and source file scoring
This commit is contained in:
parent
6ff055b9ce
commit
02c0586b43
3 changed files with 194 additions and 3 deletions
15
scripts/patches/forgeguard_evmbench_source_path.patch
Normal file
15
scripts/patches/forgeguard_evmbench_source_path.patch
Normal file
|
|
@ -0,0 +1,15 @@
|
||||||
|
diff --git a/scripts/run_evmbench.py b/scripts/run_evmbench.py
|
||||||
|
--- a/scripts/run_evmbench.py
|
||||||
|
+++ b/scripts/run_evmbench.py
|
||||||
|
@@ -598,7 +598,13 @@ def run_benchmark(model, tokenizer, configs_dir, contracts_dir, use_processor=F
|
||||||
|
)
|
||||||
|
all_raw.append(raw)
|
||||||
|
- all_findings.extend(findings)
|
||||||
|
+ # Findings are produced per source file, but the legacy result
|
||||||
|
+ # aggregated them without retaining that association.
|
||||||
|
+ for finding in findings:
|
||||||
|
+ finding_with_source = dict(finding)
|
||||||
|
+ finding_with_source["source_path"] = sol_path
|
||||||
|
+ all_findings.append(finding_with_source)
|
||||||
|
if findings:
|
||||||
|
detected = True
|
||||||
|
|
@ -4,6 +4,7 @@ set -euo pipefail
|
||||||
|
|
||||||
ROOT=/home/daniel/forgeguard
|
ROOT=/home/daniel/forgeguard
|
||||||
PROJECT="$ROOT/project"
|
PROJECT="$ROOT/project"
|
||||||
|
ARTIFEX_ROOT=/home/daniel/Artifex
|
||||||
DATA="$ROOT/artifex_dataset_versions/guard_curated_v03/training.json"
|
DATA="$ROOT/artifex_dataset_versions/guard_curated_v03/training.json"
|
||||||
RUN_ROOT="$PROJECT/artifex_overnight/guard3b_v03_overnight"
|
RUN_ROOT="$PROJECT/artifex_overnight/guard3b_v03_overnight"
|
||||||
CANARY=artifex-guard3b-v03-canary
|
CANARY=artifex-guard3b-v03-canary
|
||||||
|
|
@ -16,24 +17,34 @@ mkdir -p "$RUN_ROOT"
|
||||||
docker_run() {
|
docker_run() {
|
||||||
docker run --rm --gpus all --shm-size=16g \
|
docker run --rm --gpus all --shm-size=16g \
|
||||||
-v "$PROJECT:/workspace" \
|
-v "$PROJECT:/workspace" \
|
||||||
|
-v "$ARTIFEX_ROOT/scripts/score_evmbench_type_source_file.py:/workspace/artifex/score_evmbench_type_source_file.py:ro" \
|
||||||
-v "$ROOT/artifex_dataset_versions:/workspace/artifex_dataset_versions" \
|
-v "$ROOT/artifex_dataset_versions:/workspace/artifex_dataset_versions" \
|
||||||
-v /home/daniel/.cache/huggingface:/root/.cache/huggingface \
|
-v /home/daniel/.cache/huggingface:/root/.cache/huggingface \
|
||||||
"$IMAGE" "$@"
|
"$IMAGE" "$@"
|
||||||
}
|
}
|
||||||
|
|
||||||
|
install_evaluator_contract() {
|
||||||
|
if ! grep -q 'finding_with_source\["source_path"\]' "$PROJECT/scripts/run_evmbench.py"; then
|
||||||
|
patch --batch --forward -d "$PROJECT" -p1 < "$ARTIFEX_ROOT/scripts/patches/forgeguard_evmbench_source_path.patch"
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
evaluate() {
|
evaluate() {
|
||||||
local name=$1 adapter=${2:-}
|
local name=$1 adapter=${2:-}
|
||||||
local result="$RUN_ROOT/${name}_evmbench.json" score="$RUN_ROOT/${name}_strict.json"
|
local result="$RUN_ROOT/${name}_evmbench.json" score="$RUN_ROOT/${name}_strict.json"
|
||||||
local args=(python /workspace/scripts/run_evmbench.py --model "$MODEL" --unsloth --save "/workspace/artifex_overnight/guard3b_v03_overnight/${name}_evmbench.json")
|
local args=(python /workspace/scripts/run_evmbench.py --model "$MODEL" --unsloth --save "/workspace/artifex_overnight/guard3b_v03_overnight/${name}_evmbench.json")
|
||||||
if [[ -n "$adapter" ]]; then args+=(--adapter "/workspace/artifex_overnight/$adapter"); fi
|
if [[ -n "$adapter" ]]; then args+=(--adapter "/workspace/artifex_overnight/$adapter"); fi
|
||||||
docker_run "${args[@]}" > "$RUN_ROOT/${name}_evaluation.log" 2>&1
|
docker_run "${args[@]}" > "$RUN_ROOT/${name}_evaluation.log" 2>&1
|
||||||
docker_run python /workspace/scripts/score_evmbench_strict.py "/workspace/artifex_overnight/guard3b_v03_overnight/${name}_evmbench.json" > "$score" 2>&1
|
docker_run python /workspace/artifex/score_evmbench_type_source_file.py \
|
||||||
|
"/workspace/artifex_overnight/guard3b_v03_overnight/${name}_evmbench.json" \
|
||||||
|
--output "/workspace/artifex_overnight/guard3b_v03_overnight/${name}_strict.json" \
|
||||||
|
> "$RUN_ROOT/${name}_scoring.log" 2>&1
|
||||||
[[ -s "$result" && -s "$score" ]]
|
[[ -s "$result" && -s "$score" ]]
|
||||||
}
|
}
|
||||||
|
|
||||||
qwen_choice() {
|
qwen_choice() {
|
||||||
local summary
|
local summary
|
||||||
summary=$(tail -c 10000 "$RUN_ROOT/base_strict.json" 2>/dev/null; tail -c 10000 "$RUN_ROOT/current_strict.json" 2>/dev/null || true)
|
summary=$(jq -c '{acceptance_metric,summary:{targets,canonical_type_and_target_source_file_match}}' "$RUN_ROOT/base_strict.json"; jq -c '{acceptance_metric,summary:{targets,canonical_type_and_target_source_file_match}}' "$RUN_ROOT/current_strict.json")
|
||||||
local body
|
local body
|
||||||
body=$(jq -n --arg prompt "You are MODEL_DIRECTOR for a bounded Guard 3B overnight run. Same-harness strict-score evidence follows: $summary Choose one JSON decision only: {\"action\":\"CONTINUE\"|\"REPLICATE\"|\"STOP\",\"rationale\":\"...\"}. CONTINUE means resume the current adapter with lower LR; REPLICATE means fresh seed from the immutable base; STOP means evidence/remaining time does not justify another run." '{model:"qwen38",messages:[{role:"user",content:$prompt}],temperature:0,max_tokens:200}')
|
body=$(jq -n --arg prompt "You are MODEL_DIRECTOR for a bounded Guard 3B overnight run. Same-harness strict-score evidence follows: $summary Choose one JSON decision only: {\"action\":\"CONTINUE\"|\"REPLICATE\"|\"STOP\",\"rationale\":\"...\"}. CONTINUE means resume the current adapter with lower LR; REPLICATE means fresh seed from the immutable base; STOP means evidence/remaining time does not justify another run." '{model:"qwen38",messages:[{role:"user",content:$prompt}],temperature:0,max_tokens:200}')
|
||||||
curl -fsS --max-time 180 http://127.0.0.1:8002/v1/chat/completions -H 'Content-Type: application/json' -d "$body" | jq -r '.choices[0].message.content' > "$RUN_ROOT/qwen_plan.json" || true
|
curl -fsS --max-time 180 http://127.0.0.1:8002/v1/chat/completions -H 'Content-Type: application/json' -d "$body" | jq -r '.choices[0].message.content' > "$RUN_ROOT/qwen_plan.json" || true
|
||||||
|
|
@ -55,10 +66,11 @@ if [[ "$(cat "$RUN_ROOT/canary_exit_code.txt")" != "0" ]]; then
|
||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
cp "$PROJECT/artifex_overnight/guard3b_v03_canary/pilot_metrics.json" "$RUN_ROOT/canary_metrics.json"
|
cp "$PROJECT/artifex_overnight/guard3b_v03_canary/pilot_metrics.json" "$RUN_ROOT/canary_metrics.json"
|
||||||
|
install_evaluator_contract
|
||||||
|
|
||||||
# Base and the initial Challenger use the identical fixed harness/version.
|
# Base and the initial Challenger use the identical fixed harness/version.
|
||||||
# Reuse a completed base result when resuming after an infrastructure repair.
|
# Reuse a completed base result when resuming after an infrastructure repair.
|
||||||
if [[ ! -s "$RUN_ROOT/base_evmbench.json" ]]; then evaluate base; else docker_run python /workspace/scripts/score_evmbench_strict.py "/workspace/artifex_overnight/guard3b_v03_overnight/base_evmbench.json" > "$RUN_ROOT/base_strict.json" 2>&1; fi
|
if [[ ! -s "$RUN_ROOT/base_evmbench.json" ]]; then evaluate base; else docker_run python /workspace/artifex/score_evmbench_type_source_file.py "/workspace/artifex_overnight/guard3b_v03_overnight/base_evmbench.json" --output "/workspace/artifex_overnight/guard3b_v03_overnight/base_strict.json" > "$RUN_ROOT/base_scoring.log" 2>&1; fi
|
||||||
cp "$RUN_ROOT/base_strict.json" "$RUN_ROOT/current_strict.json"
|
cp "$RUN_ROOT/base_strict.json" "$RUN_ROOT/current_strict.json"
|
||||||
CURRENT=guard3b_v03_canary
|
CURRENT=guard3b_v03_canary
|
||||||
evaluate canary "$CURRENT"
|
evaluate canary "$CURRENT"
|
||||||
|
|
|
||||||
164
scripts/score_evmbench_type_source_file.py
Normal file
164
scripts/score_evmbench_type_source_file.py
Normal file
|
|
@ -0,0 +1,164 @@
|
||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Score EVMBench by canonical vulnerability type and patch-mapped source file.
|
||||||
|
|
||||||
|
This is the acceptance scorer for the Guard 3B program. It deliberately does
|
||||||
|
not score the legacy "any finding" detection rate. EVMBench's runnable issue
|
||||||
|
inventory has a reviewed patch-mapped source file but no reviewed function or
|
||||||
|
line anchor, so function/line accuracy is reported as unavailable rather than
|
||||||
|
being guessed from model text.
|
||||||
|
"""
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import re
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
|
ROOT = Path(__file__).resolve().parent.parent
|
||||||
|
DEFAULT_INVENTORY = ROOT / "docs" / "evmbench_runnable_issue_inventory.json"
|
||||||
|
|
||||||
|
TYPE_ALIASES = {
|
||||||
|
"reentrancy": "reentrancy",
|
||||||
|
"re-entrancy": "reentrancy",
|
||||||
|
"flash loan": "flash_loan",
|
||||||
|
"flash_loan": "flash_loan",
|
||||||
|
"oracle manipulation": "oracle_manipulation",
|
||||||
|
"oracle_manipulation": "oracle_manipulation",
|
||||||
|
"access control": "access_control",
|
||||||
|
"access_control": "access_control",
|
||||||
|
"integer overflow": "integer_overflow",
|
||||||
|
"integer_overflow": "integer_overflow",
|
||||||
|
"integer underflow": "integer_overflow",
|
||||||
|
"front running": "front_running",
|
||||||
|
"front-running": "front_running",
|
||||||
|
"front_running": "front_running",
|
||||||
|
"delegate call": "delegate_call",
|
||||||
|
"delegatecall": "delegate_call",
|
||||||
|
"delegate_call": "delegate_call",
|
||||||
|
"logic error": "logic_error",
|
||||||
|
"logic_error": "logic_error",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def read_json(path):
|
||||||
|
with path.open(encoding="utf-8") as handle:
|
||||||
|
return json.load(handle)
|
||||||
|
|
||||||
|
|
||||||
|
def normalize_type(value):
|
||||||
|
value = re.sub(r"[ _-]+", " ", str(value or "").strip().lower())
|
||||||
|
return TYPE_ALIASES.get(value, "unknown")
|
||||||
|
|
||||||
|
|
||||||
|
def normalize_path(value):
|
||||||
|
return str(value or "").replace("\\", "/").strip("/").lower()
|
||||||
|
|
||||||
|
|
||||||
|
def source_file_matches(reported_path, expected_paths):
|
||||||
|
reported = normalize_path(reported_path)
|
||||||
|
return bool(reported) and any(
|
||||||
|
reported.endswith(normalize_path(expected)) for expected in expected_paths
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def rate(count, total):
|
||||||
|
return round(100 * count / total, 1) if total else 0
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
parser = argparse.ArgumentParser(
|
||||||
|
description="Score canonical type plus patch-mapped target source file"
|
||||||
|
)
|
||||||
|
parser.add_argument("results", type=Path)
|
||||||
|
parser.add_argument("--inventory", type=Path, default=DEFAULT_INVENTORY)
|
||||||
|
parser.add_argument("--output", type=Path, required=True)
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
result = read_json(args.results)
|
||||||
|
inventory = read_json(args.inventory)
|
||||||
|
records = result.get("vulnerabilities")
|
||||||
|
if not isinstance(records, list):
|
||||||
|
raise ValueError("results lacks per-vulnerability findings")
|
||||||
|
by_key = {
|
||||||
|
f"{record.get('audit_id')}:{record.get('vuln_id')}": record
|
||||||
|
for record in records
|
||||||
|
}
|
||||||
|
rows = []
|
||||||
|
missing_source_path = 0
|
||||||
|
for issue in inventory.get("issues", []):
|
||||||
|
key = f"{issue['audit_id']}:{issue['vulnerability_id']}"
|
||||||
|
record = by_key.get(key, {})
|
||||||
|
expected_paths = [
|
||||||
|
source["path"] for source in issue.get("source_files", [])
|
||||||
|
if source.get("exists")
|
||||||
|
]
|
||||||
|
findings = record.get("findings", [])
|
||||||
|
if not isinstance(findings, list):
|
||||||
|
findings = []
|
||||||
|
annotated = []
|
||||||
|
for finding in findings:
|
||||||
|
source_path = finding.get("source_path", "")
|
||||||
|
if not source_path:
|
||||||
|
missing_source_path += 1
|
||||||
|
type_match = normalize_type(finding.get("type")) == issue["canonical_type"]
|
||||||
|
file_match = source_file_matches(source_path, expected_paths)
|
||||||
|
annotated.append({
|
||||||
|
"reported_type": finding.get("type", ""),
|
||||||
|
"normalized_type": normalize_type(finding.get("type")),
|
||||||
|
"reported_source_path": source_path,
|
||||||
|
"reported_location": finding.get("location", ""),
|
||||||
|
"canonical_type_match": type_match,
|
||||||
|
"target_source_file_match": file_match,
|
||||||
|
"canonical_type_and_target_source_file_match": type_match and file_match,
|
||||||
|
})
|
||||||
|
rows.append({
|
||||||
|
"target": key,
|
||||||
|
"canonical_type": issue["canonical_type"],
|
||||||
|
"expected_source_files": expected_paths,
|
||||||
|
"any_finding": bool(findings),
|
||||||
|
"canonical_type_match": any(item["canonical_type_match"] for item in annotated),
|
||||||
|
"canonical_type_and_target_source_file_match": any(
|
||||||
|
item["canonical_type_and_target_source_file_match"]
|
||||||
|
for item in annotated
|
||||||
|
),
|
||||||
|
"findings": annotated,
|
||||||
|
})
|
||||||
|
|
||||||
|
total = len(rows)
|
||||||
|
canonical_count = sum(row["canonical_type_match"] for row in rows)
|
||||||
|
strict_count = sum(
|
||||||
|
row["canonical_type_and_target_source_file_match"] for row in rows
|
||||||
|
)
|
||||||
|
artifact = {
|
||||||
|
"schema_version": 1,
|
||||||
|
"scorer": "score_evmbench_type_source_file.py",
|
||||||
|
"input_result": str(args.results),
|
||||||
|
"inventory": str(args.inventory),
|
||||||
|
"acceptance_metric": "canonical_type_and_target_source_file_match",
|
||||||
|
"metric_definition": (
|
||||||
|
"A finding must match the inventory canonical type and be produced "
|
||||||
|
"while evaluating an inventory patch-mapped target source file."
|
||||||
|
),
|
||||||
|
"location_label_coverage": "target source file only; reviewed function/line anchors are unavailable",
|
||||||
|
"summary": {
|
||||||
|
"targets": total,
|
||||||
|
"canonical_type_match": {"count": canonical_count, "rate_percent": rate(canonical_count, total)},
|
||||||
|
"canonical_type_and_target_source_file_match": {
|
||||||
|
"count": strict_count,
|
||||||
|
"rate_percent": rate(strict_count, total),
|
||||||
|
},
|
||||||
|
},
|
||||||
|
"validation": {
|
||||||
|
"findings_missing_source_path": missing_source_path,
|
||||||
|
"legacy_any_finding_excluded_from_acceptance": True,
|
||||||
|
},
|
||||||
|
"rows": rows,
|
||||||
|
}
|
||||||
|
args.output.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
with args.output.open("w", encoding="utf-8") as handle:
|
||||||
|
json.dump(artifact, handle, indent=2)
|
||||||
|
handle.write("\n")
|
||||||
|
print(json.dumps(artifact["summary"], sort_keys=True))
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
Loading…
Add table
Reference in a new issue