Merge nucleic/sleek-ember-seal-uady into dev
This commit is contained in:
+13
-107
@@ -4,12 +4,9 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import csv
|
||||
import hashlib
|
||||
import json
|
||||
import math
|
||||
import sys
|
||||
from collections import Counter, defaultdict
|
||||
from collections import Counter
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any, Sequence
|
||||
@@ -31,10 +28,15 @@ from purpose_data import (
|
||||
exclude_reviewed_duplicates,
|
||||
load_classifiable_fixtures,
|
||||
load_sources,
|
||||
normalize_prompt,
|
||||
prompt_hash,
|
||||
write_json,
|
||||
)
|
||||
from review_contract import (
|
||||
DEFAULT_SAMPLE_FRACTION,
|
||||
DEFAULT_SAMPLE_SEED,
|
||||
stratified_review_sample,
|
||||
write_review_csv,
|
||||
)
|
||||
from train import (
|
||||
DEFAULT_MODEL,
|
||||
DEFAULT_MODEL_REVISION,
|
||||
@@ -49,10 +51,6 @@ SCRIPT_DIR = Path(__file__).resolve().parent
|
||||
REPOSITORY_ROOT = SCRIPT_DIR.parent.parent
|
||||
DEFAULT_REPORT = SCRIPT_DIR / "data" / "semantic-audit-v1.json"
|
||||
DEFAULT_REVIEW_CSV = SCRIPT_DIR / ".artifacts" / "human-review-v1.csv"
|
||||
DEFAULT_SAMPLE_SEED = 0xA11D17
|
||||
DEFAULT_SAMPLE_FRACTION = 0.10
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class AuditRecord:
|
||||
prompt: str
|
||||
@@ -77,103 +75,6 @@ def _relative(path: Path) -> str:
|
||||
return str(path.resolve())
|
||||
|
||||
|
||||
def _stable_rank(seed: int, record: SourceRecord) -> str:
|
||||
material = f"{seed}\0{prompt_hash(record.value['prompt'])}".encode("utf-8")
|
||||
return hashlib.sha256(material).hexdigest()
|
||||
|
||||
|
||||
def _review_stratum(record: SourceRecord) -> tuple[str, str, str]:
|
||||
language = record.value["lang"].split("-", 1)[0].casefold()
|
||||
return record.value["purpose"], record.value["slice"], language
|
||||
|
||||
|
||||
def stratified_review_sample(
|
||||
records: Sequence[SourceRecord],
|
||||
*,
|
||||
fraction: float = DEFAULT_SAMPLE_FRACTION,
|
||||
seed: int = DEFAULT_SAMPLE_SEED,
|
||||
) -> list[SourceRecord]:
|
||||
"""Choose exactly round(N*fraction), apportioned across purpose/slice/language."""
|
||||
|
||||
if not records:
|
||||
raise DataError("cannot sample an empty review population")
|
||||
if not 0.0 < fraction <= 1.0:
|
||||
raise DataError("review fraction must be in (0, 1]")
|
||||
|
||||
target = round(len(records) * fraction)
|
||||
groups: dict[tuple[str, str, str], list[SourceRecord]] = defaultdict(list)
|
||||
for record in records:
|
||||
groups[_review_stratum(record)].append(record)
|
||||
|
||||
# Hamilton apportionment preserves small language/slice strata while still producing
|
||||
# the exact requested global sample size.
|
||||
allocations: dict[tuple[str, str, str], int] = {}
|
||||
remainders: list[tuple[float, str, tuple[str, str, str]]] = []
|
||||
allocated = 0
|
||||
for key in sorted(groups):
|
||||
quota = len(groups[key]) * target / len(records)
|
||||
base = math.floor(quota)
|
||||
allocations[key] = base
|
||||
allocated += base
|
||||
tie_break = hashlib.sha256(f"{seed}\0{key}".encode("utf-8")).hexdigest()
|
||||
remainders.append((quota - base, tie_break, key))
|
||||
for _, _, key in sorted(remainders, reverse=True)[: target - allocated]:
|
||||
allocations[key] += 1
|
||||
|
||||
selected: list[SourceRecord] = []
|
||||
for key in sorted(groups):
|
||||
ordered = sorted(groups[key], key=lambda record: _stable_rank(seed, record))
|
||||
selected.extend(ordered[: allocations[key]])
|
||||
return sorted(selected, key=lambda record: _stable_rank(seed + 1, record))
|
||||
|
||||
|
||||
def write_review_csv(path: Path, records: Sequence[SourceRecord]) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with path.open("w", encoding="utf-8", newline="") as handle:
|
||||
writer = csv.DictWriter(
|
||||
handle,
|
||||
fieldnames=[
|
||||
"promptHash",
|
||||
"source",
|
||||
"line",
|
||||
"prompt",
|
||||
"generatedPurpose",
|
||||
"generatedSecondary",
|
||||
"generatedMixed",
|
||||
"generatedDifficulty",
|
||||
"generatedSlice",
|
||||
"generatedLanguage",
|
||||
"reviewedPurpose",
|
||||
"reviewedSecondary",
|
||||
"reviewedDifficulty",
|
||||
"reviewStatus",
|
||||
"reviewNotes",
|
||||
],
|
||||
)
|
||||
writer.writeheader()
|
||||
for record in records:
|
||||
value = record.value
|
||||
writer.writerow(
|
||||
{
|
||||
"promptHash": prompt_hash(value["prompt"]),
|
||||
"source": _relative(record.source),
|
||||
"line": record.line,
|
||||
"prompt": normalize_prompt(value["prompt"]),
|
||||
"generatedPurpose": value["purpose"],
|
||||
"generatedSecondary": value["secondary"] or "",
|
||||
"generatedMixed": str(value["mixed"]).lower(),
|
||||
"generatedDifficulty": value["difficulty"],
|
||||
"generatedSlice": value["slice"],
|
||||
"generatedLanguage": value["lang"],
|
||||
"reviewedPurpose": "",
|
||||
"reviewedSecondary": "",
|
||||
"reviewedDifficulty": "",
|
||||
"reviewStatus": "",
|
||||
"reviewNotes": "",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def semantic_candidates(
|
||||
embeddings: np.ndarray,
|
||||
purposes: Sequence[str],
|
||||
@@ -350,7 +251,11 @@ def audit(args: argparse.Namespace) -> dict[str, Any]:
|
||||
fraction=args.review_fraction,
|
||||
seed=args.review_seed,
|
||||
)
|
||||
write_review_csv(args.review_csv.resolve(), sample)
|
||||
write_review_csv(
|
||||
args.review_csv.resolve(),
|
||||
sample,
|
||||
source_formatter=_relative,
|
||||
)
|
||||
|
||||
audit_records = _audit_records(
|
||||
curated.records, fixtures, args.fixtures.resolve()
|
||||
@@ -476,6 +381,7 @@ def audit(args: argparse.Namespace) -> dict[str, Any]:
|
||||
"reviewedPurpose",
|
||||
"reviewedSecondary",
|
||||
"reviewedDifficulty",
|
||||
"reviewedSlice",
|
||||
"reviewStatus",
|
||||
],
|
||||
},
|
||||
|
||||
Reference in New Issue
Block a user