Merge nucleic/sleek-ember-seal-uady into dev

This commit is contained in:
2026-07-30 18:53:22 -07:00
parent ddbff97191
commit bb6d53a520
10 changed files with 962 additions and 110 deletions
+13 -107
View File
@@ -4,12 +4,9 @@
from __future__ import annotations
import argparse
import csv
import hashlib
import json
import math
import sys
from collections import Counter, defaultdict
from collections import Counter
from dataclasses import dataclass
from pathlib import Path
from typing import Any, Sequence
@@ -31,10 +28,15 @@ from purpose_data import (
exclude_reviewed_duplicates,
load_classifiable_fixtures,
load_sources,
normalize_prompt,
prompt_hash,
write_json,
)
from review_contract import (
DEFAULT_SAMPLE_FRACTION,
DEFAULT_SAMPLE_SEED,
stratified_review_sample,
write_review_csv,
)
from train import (
DEFAULT_MODEL,
DEFAULT_MODEL_REVISION,
@@ -49,10 +51,6 @@ SCRIPT_DIR = Path(__file__).resolve().parent
REPOSITORY_ROOT = SCRIPT_DIR.parent.parent
DEFAULT_REPORT = SCRIPT_DIR / "data" / "semantic-audit-v1.json"
DEFAULT_REVIEW_CSV = SCRIPT_DIR / ".artifacts" / "human-review-v1.csv"
DEFAULT_SAMPLE_SEED = 0xA11D17
DEFAULT_SAMPLE_FRACTION = 0.10
@dataclass(frozen=True)
class AuditRecord:
prompt: str
@@ -77,103 +75,6 @@ def _relative(path: Path) -> str:
return str(path.resolve())
def _stable_rank(seed: int, record: SourceRecord) -> str:
material = f"{seed}\0{prompt_hash(record.value['prompt'])}".encode("utf-8")
return hashlib.sha256(material).hexdigest()
def _review_stratum(record: SourceRecord) -> tuple[str, str, str]:
language = record.value["lang"].split("-", 1)[0].casefold()
return record.value["purpose"], record.value["slice"], language
def stratified_review_sample(
records: Sequence[SourceRecord],
*,
fraction: float = DEFAULT_SAMPLE_FRACTION,
seed: int = DEFAULT_SAMPLE_SEED,
) -> list[SourceRecord]:
"""Choose exactly round(N*fraction), apportioned across purpose/slice/language."""
if not records:
raise DataError("cannot sample an empty review population")
if not 0.0 < fraction <= 1.0:
raise DataError("review fraction must be in (0, 1]")
target = round(len(records) * fraction)
groups: dict[tuple[str, str, str], list[SourceRecord]] = defaultdict(list)
for record in records:
groups[_review_stratum(record)].append(record)
# Hamilton apportionment preserves small language/slice strata while still producing
# the exact requested global sample size.
allocations: dict[tuple[str, str, str], int] = {}
remainders: list[tuple[float, str, tuple[str, str, str]]] = []
allocated = 0
for key in sorted(groups):
quota = len(groups[key]) * target / len(records)
base = math.floor(quota)
allocations[key] = base
allocated += base
tie_break = hashlib.sha256(f"{seed}\0{key}".encode("utf-8")).hexdigest()
remainders.append((quota - base, tie_break, key))
for _, _, key in sorted(remainders, reverse=True)[: target - allocated]:
allocations[key] += 1
selected: list[SourceRecord] = []
for key in sorted(groups):
ordered = sorted(groups[key], key=lambda record: _stable_rank(seed, record))
selected.extend(ordered[: allocations[key]])
return sorted(selected, key=lambda record: _stable_rank(seed + 1, record))
def write_review_csv(path: Path, records: Sequence[SourceRecord]) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
with path.open("w", encoding="utf-8", newline="") as handle:
writer = csv.DictWriter(
handle,
fieldnames=[
"promptHash",
"source",
"line",
"prompt",
"generatedPurpose",
"generatedSecondary",
"generatedMixed",
"generatedDifficulty",
"generatedSlice",
"generatedLanguage",
"reviewedPurpose",
"reviewedSecondary",
"reviewedDifficulty",
"reviewStatus",
"reviewNotes",
],
)
writer.writeheader()
for record in records:
value = record.value
writer.writerow(
{
"promptHash": prompt_hash(value["prompt"]),
"source": _relative(record.source),
"line": record.line,
"prompt": normalize_prompt(value["prompt"]),
"generatedPurpose": value["purpose"],
"generatedSecondary": value["secondary"] or "",
"generatedMixed": str(value["mixed"]).lower(),
"generatedDifficulty": value["difficulty"],
"generatedSlice": value["slice"],
"generatedLanguage": value["lang"],
"reviewedPurpose": "",
"reviewedSecondary": "",
"reviewedDifficulty": "",
"reviewStatus": "",
"reviewNotes": "",
}
)
def semantic_candidates(
embeddings: np.ndarray,
purposes: Sequence[str],
@@ -350,7 +251,11 @@ def audit(args: argparse.Namespace) -> dict[str, Any]:
fraction=args.review_fraction,
seed=args.review_seed,
)
write_review_csv(args.review_csv.resolve(), sample)
write_review_csv(
args.review_csv.resolve(),
sample,
source_formatter=_relative,
)
audit_records = _audit_records(
curated.records, fixtures, args.fixtures.resolve()
@@ -476,6 +381,7 @@ def audit(args: argparse.Namespace) -> dict[str, Any]:
"reviewedPurpose",
"reviewedSecondary",
"reviewedDifficulty",
"reviewedSlice",
"reviewStatus",
],
},