9ff86f9cd0
Trivial P0 fixes (this commit): two docstrings claimed things the wiring
does not do. variants/__init__.py said the package 'never imports agents/'
— false since Wave 2: generator.py holds the module-direct
agents.structured import (the sanctioned D-020 shared defense, same
exception as grading/engine.py); docstring now states the real boundary.
templates.py (header + RubricAnchors) claimed rubric anchors are 'used by
grading context' / 'shipped to the grader' — false in the current wiring:
GradingEngine is variant-blind (variant_seed=None; no variant lookup; no
anchor consumption; render_trace_digest takes only the digest). Docstrings
now tell the truth and name the follow-up.
Verification (four layers, evidence in the phase report):
- Structural: AST boundary audit clean — variants/ has zero api/fastapi
imports; only sanctioned agents.structured + llm/prompts/config/store.
- Behavioral: variants slice 41/41 green; full suite 324/324 green;
pnpm typecheck 7/7 green (forced, no cache). Live app probes confirmed
distinct learners -> distinct statements/seeds/task_ids at the API
level, cache hit = zero LLM calls, deterministic fallback (calls==2),
a-5 fairness envelope test green, sha256(template|learner|milestone)
seed derivation verified byte-exact against the spec formula.
- Security: secrets scan over the P04 diff (b52bef9..4acffac, 17 files
+2100/-1) clean — no key/token/password assignments, no URLs, no key
shapes. Prompt-injection surface bounded: the variant prompt carries
only template skeleton/title/id + seeded slot values — no learner id
or user-controlled content reaches the LLM. Empty learner_id -> 422.
- Quality: ruff clean; AI_MODEL env override verified live; tests are
mock-only (MockProvider family, zero network imports).
Must-Haves: 5 of 6 SATISFIED. NOT satisfied: MH#4 half — anchors are
present per template and a-5-testable, but NOT shipped to the grader
prompt context (non-trivial cross-module wiring: engine + prompt
signature + lifespan ordering — grading engine is built before the
variant store exists; reported as P1, not fixed here).
---ci---
phase: 4
milestone: v0.3
status: verify
requirements:
covered: [REQ-3-005]
partial: [REQ-3-005]
lessons:
- Docstrings that describe a must-have's target state ('shipped to the
grader prompt') read as done in review — verify wiring, not words:
grep the consumer side (grading/) before believing the producer side.
- main.py builds GradingEngine before variant_store exists; any P4/P6
anchor-shipment fix must reorder lifespan construction or inject the
variant store into the engine after the fact.
- Params distinctness is parametric (280-64908 combos per template) —
two learners CAN draw identical params (~15% at 10 learners on the
tightest template); distinctness is proven via seeds/task_ids/statement
embeddings, which is what the must-have actually requires.
---/ci---
306 lines
12 KiB
Python
306 lines
12 KiB
Python
"""Task template library for seeded variant generation (D-029, REQ-3-005).
|
|
|
|
A `TaskTemplate` binds a competency (D-021-aligned corpus ID), a statement
|
|
skeleton with `{slot}` placeholders, typed `ParameterSlot`s, difficulty-
|
|
normalization rubric anchors (the expected feature envelope that bounds
|
|
variant fairness in the a-5 envelope test — grader-prompt shipment is the
|
|
tracked P4 follow-up; grading is variant-blind today), and starter-file
|
|
scaffolds served into the sandbox workdir (wired in P6).
|
|
|
|
Slot sampling is PURE CODE: `random.Random(seed)` over typed slots — fully
|
|
reproducible for a given seed, independent of the LLM. The LLM only renders
|
|
the seeded slot values into the statement skeleton (D-020 defense).
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import random
|
|
import re
|
|
from typing import Literal
|
|
|
|
from pydantic import BaseModel, ConfigDict, Field, field_validator
|
|
|
|
SlotType = Literal["enum", "int_range", "string_set"]
|
|
|
|
|
|
class ParameterSlot(BaseModel):
|
|
"""One typed fill-in for a statement skeleton."""
|
|
|
|
model_config = ConfigDict(frozen=True)
|
|
|
|
name: str = Field(min_length=1)
|
|
type: SlotType
|
|
values: list[str] = Field(default_factory=list) # enum/string_set options
|
|
lo: int | None = None # int_range bounds
|
|
hi: int | None = None
|
|
|
|
@field_validator("values")
|
|
@classmethod
|
|
def _values_nonempty_for_enums(cls, v: list[str], info) -> list[str]:
|
|
if info.data.get("type") in ("enum", "string_set") and not v:
|
|
raise ValueError(f"slot {info.data.get('name')!r} needs values")
|
|
return v
|
|
|
|
def sample(self, rng: random.Random) -> str | int:
|
|
"""Deterministic sample from the seeded RNG. Validated after sampling."""
|
|
if self.type == "enum" or self.type == "string_set":
|
|
return rng.choice(self.values)
|
|
if self.type == "int_range":
|
|
lo = self.lo if self.lo is not None else 0
|
|
hi = self.hi if self.hi is not None else lo
|
|
if hi < lo:
|
|
raise ValueError(f"slot {self.name!r}: hi < lo")
|
|
return rng.randint(lo, hi)
|
|
raise ValueError(f"unsupported slot type: {self.type!r}")
|
|
|
|
def validate_value(self, value: str | int) -> bool:
|
|
"""Is `value` schema-valid for this slot? (params JSON gate, a-5.)"""
|
|
if self.type in ("enum", "string_set"):
|
|
return isinstance(value, str) and value in self.values
|
|
if self.type == "int_range":
|
|
lo = self.lo if self.lo is not None else 0
|
|
hi = self.hi if self.hi is not None else lo
|
|
return isinstance(value, int) and lo <= value <= hi
|
|
return False
|
|
|
|
|
|
class RubricAnchors(BaseModel):
|
|
"""Difficulty-normalization anchors for the grader (a-5).
|
|
|
|
Expected FEATURE ENVELOPE (digest-space): the expected effort band
|
|
for this template, so two variants of one template are held to the
|
|
same bar regardless of which slot values a learner drew. The a-5
|
|
envelope test (tests/variants/test_generator.py) binds variants to
|
|
these bands in code. Shipping them into the grader prompt context is
|
|
the P4 must-have follow-up tracked for final review: grading is
|
|
variant-blind in the current wiring (engine.py stamps
|
|
variant_seed=None), so today the anchors gate variant fairness in
|
|
tests only — not yet in the LLM prompt.
|
|
"""
|
|
|
|
model_config = ConfigDict(frozen=True)
|
|
|
|
expected_edit_count_band: tuple[int, int]
|
|
expected_min_test_runs: int
|
|
expected_error_fix_cycles_band: tuple[int, int]
|
|
notes: str = ""
|
|
|
|
|
|
class TaskTemplate(BaseModel):
|
|
"""A reusable task shape; variants instantiate it per learner."""
|
|
|
|
model_config = ConfigDict(frozen=True)
|
|
|
|
id: str = Field(min_length=1)
|
|
competency_id: str = Field(min_length=1) # D-021 corpus alignment
|
|
title: str
|
|
statement_skeleton: str = Field(min_length=1) # {slot} placeholders
|
|
slots: list[ParameterSlot] = Field(min_length=1)
|
|
rubric_anchors: RubricAnchors
|
|
starter_files: dict[str, str] = Field(default_factory=dict) # path -> content
|
|
test_command: str
|
|
|
|
@field_validator("statement_skeleton")
|
|
@classmethod
|
|
def _skeleton_placeholders(cls, v: str) -> str:
|
|
if "{" not in v or "}" not in v:
|
|
raise ValueError("statement_skeleton needs at least one {slot}")
|
|
return v
|
|
|
|
def render(self, params: dict[str, str | int]) -> str:
|
|
"""Fill the skeleton with validated params."""
|
|
for slot in self.slots:
|
|
if slot.name not in params:
|
|
raise ValueError(f"missing param for slot {slot.name!r}")
|
|
if not slot.validate_value(params[slot.name]):
|
|
raise ValueError(f"invalid value for slot {slot.name!r}: {params[slot.name]!r}")
|
|
return self.statement_skeleton.format(**params)
|
|
|
|
def sample_params(self, seed: int) -> dict[str, str | int]:
|
|
"""Seeded, reproducible, schema-valid slot values (pure code)."""
|
|
rng = random.Random(seed)
|
|
return {slot.name: slot.sample(rng) for slot in self.slots}
|
|
|
|
|
|
# --- Template library (v0.3 initial set) --------------------------------------
|
|
# Competency IDs are D-021-aligned with the Python corpus
|
|
# (ai_service/corpus/learner_context.py) and the TS mock-data layer
|
|
# (packages/mock-data/competency-stacks.ts: deterministic cid() scheme).
|
|
|
|
TEMPLATES: dict[str, TaskTemplate] = {
|
|
"tpl-llm-judge": TaskTemplate(
|
|
id="tpl-llm-judge",
|
|
competency_id="stack-orchestration-c007",
|
|
title="Build an LLM-as-Judge Evaluator",
|
|
statement_skeleton=(
|
|
"Build a small LLM-as-judge evaluator for {domain} answers. "
|
|
"The judge must score each answer on {criterion} using a 0-4 scale, "
|
|
"return structured JSON, and handle at least {edge_cases} edge-case "
|
|
"answer classes (empty, off-topic, adversarial). Include a tiny "
|
|
"repro test set of at least {test_size} examples and print a summary "
|
|
"table of scores."
|
|
),
|
|
slots=[
|
|
ParameterSlot(
|
|
name="domain",
|
|
type="enum",
|
|
values=["customer-support", "code-review", "summarization", "tutoring"],
|
|
),
|
|
ParameterSlot(
|
|
name="criterion",
|
|
type="enum",
|
|
values=["factual-accuracy", "helpfulness", "safety", "completeness"],
|
|
),
|
|
ParameterSlot(name="edge_cases", type="int_range", lo=2, hi=4),
|
|
ParameterSlot(name="test_size", type="int_range", lo=3, hi=8),
|
|
],
|
|
rubric_anchors=RubricAnchors(
|
|
expected_edit_count_band=(3, 25),
|
|
expected_min_test_runs=2,
|
|
expected_error_fix_cycles_band=(0, 4),
|
|
notes="Slot draw changes the SCENARIO, not the engineering depth.",
|
|
),
|
|
starter_files={
|
|
"README.md": (
|
|
"# LLM-as-Judge Evaluator\n\n"
|
|
"Implement `judge.py`:\n"
|
|
"- `score(answer: str) -> dict` — 0-4 on the named criterion\n"
|
|
"- structured JSON output (schema below)\n"
|
|
"- edge-case classes handled explicitly\n"
|
|
"- `pytest` must pass\n"
|
|
),
|
|
"judge.py": "def score(answer: str) -> dict:\n raise NotImplementedError\n",
|
|
"test_judge.py": "def test_placeholder():\n assert True\n",
|
|
},
|
|
test_command="pytest -q",
|
|
),
|
|
"tpl-guardrail-schema": TaskTemplate(
|
|
id="tpl-guardrail-schema",
|
|
competency_id="stack-orchestration-c008",
|
|
title="Schema Guardrail Pipeline",
|
|
statement_skeleton=(
|
|
"Implement an output-validation guardrail for a model returning "
|
|
"{entity} records. Validate against a typed schema with {field_count} "
|
|
"required fields, coerce or reject {failure_mode} failures, and emit "
|
|
"a fallback response for invalid payloads. Cover with at least "
|
|
"{test_size} unit tests including malformed JSON."
|
|
),
|
|
slots=[
|
|
ParameterSlot(
|
|
name="entity",
|
|
type="enum",
|
|
values=["user-profile", "job-posting", "candidate", "invoice"],
|
|
),
|
|
ParameterSlot(
|
|
name="failure_mode",
|
|
type="enum",
|
|
values=["strict-reject", "coerce-when-safe"],
|
|
),
|
|
ParameterSlot(name="field_count", type="int_range", lo=4, hi=8),
|
|
ParameterSlot(name="test_size", type="int_range", lo=4, hi=10),
|
|
],
|
|
rubric_anchors=RubricAnchors(
|
|
expected_edit_count_band=(3, 30),
|
|
expected_min_test_runs=2,
|
|
expected_error_fix_cycles_band=(0, 5),
|
|
notes="All slot draws land in the same engineering band.",
|
|
),
|
|
starter_files={
|
|
"README.md": (
|
|
"# Schema Guardrail\n\nImplement `guardrail.py`:\n"
|
|
"- `validate(payload: dict) -> dict | Fallback`\n"
|
|
"- required-field checks, failure policy, fallback emission\n"
|
|
),
|
|
"guardrail.py": "def validate(payload: dict):\n raise NotImplementedError\n",
|
|
"test_guardrail.py": "def test_placeholder():\n assert True\n",
|
|
},
|
|
test_command="pytest -q",
|
|
),
|
|
"tpl-rag-chunker": TaskTemplate(
|
|
id="tpl-rag-chunker",
|
|
competency_id="stack-orchestration-c005",
|
|
title="RAG Chunking Strategy",
|
|
statement_skeleton=(
|
|
"Implement a document chunker for {doc_type} retrieval. Support "
|
|
"{strategy} chunking with a target size of ~{chunk_size} tokens, "
|
|
"preserve {invariant} across chunk boundaries, and evaluate overlap "
|
|
"quality with at least {test_size} fixture documents."
|
|
),
|
|
slots=[
|
|
ParameterSlot(
|
|
name="doc_type",
|
|
type="enum",
|
|
values=["technical-docs", "legal-contracts", "transcripts"],
|
|
),
|
|
ParameterSlot(
|
|
name="strategy",
|
|
type="enum",
|
|
values=["fixed-window", "semantic-boundary", "hybrid"],
|
|
),
|
|
ParameterSlot(
|
|
name="invariant",
|
|
type="enum",
|
|
values=["code-block-integrity", "section-headers", "sentence-completeness"],
|
|
),
|
|
ParameterSlot(name="chunk_size", type="int_range", lo=200, hi=800),
|
|
ParameterSlot(name="test_size", type="int_range", lo=3, hi=6),
|
|
],
|
|
rubric_anchors=RubricAnchors(
|
|
expected_edit_count_band=(4, 35),
|
|
expected_min_test_runs=2,
|
|
expected_error_fix_cycles_band=(0, 6),
|
|
notes="Strategy draw changes implementation shape, not depth.",
|
|
),
|
|
starter_files={
|
|
"README.md": (
|
|
"# RAG Chunker\n\nImplement `chunker.py`:\n"
|
|
"- `chunk(text: str) -> list[str]`\n- invariant preserved\n- tests green\n"
|
|
),
|
|
"chunker.py": "def chunk(text: str) -> list[str]:\n raise NotImplementedError\n",
|
|
"test_chunker.py": "def test_placeholder():\n assert True\n",
|
|
},
|
|
test_command="pytest -q",
|
|
),
|
|
}
|
|
|
|
|
|
_KNOWN_COMPETENCY_IDS: set[str] = {
|
|
# D-021: mirrored from ai_service/corpus/learner_context.py — the Python
|
|
# source of truth for stack-orchestration competencies used by v0.2 agents.
|
|
"stack-orchestration-c001",
|
|
"stack-orchestration-c002",
|
|
"stack-orchestration-c003",
|
|
"stack-orchestration-c004",
|
|
"stack-orchestration-c005",
|
|
"stack-orchestration-c007",
|
|
"stack-orchestration-c008",
|
|
"stack-orchestration-c011",
|
|
"stack-designer-c001",
|
|
"stack-designer-c002",
|
|
"stack-safety-c021",
|
|
}
|
|
|
|
|
|
def get_template(template_id: str) -> TaskTemplate | None:
|
|
return TEMPLATES.get(template_id)
|
|
|
|
|
|
def template_for_competency(competency_id: str) -> list[TaskTemplate]:
|
|
return [t for t in TEMPLATES.values() if t.competency_id == competency_id]
|
|
|
|
|
|
def validate_competency_binding() -> None:
|
|
"""All templates must bind to known D-021 corpus competency IDs."""
|
|
for t in TEMPLATES.values():
|
|
if t.competency_id not in _KNOWN_COMPETENCY_IDS:
|
|
raise ValueError(
|
|
f"template {t.id!r} binds unknown competency {t.competency_id!r}"
|
|
)
|
|
|
|
|
|
def slots_pattern_ok(skeleton: str, slots: list[ParameterSlot]) -> bool:
|
|
"""Every {placeholder} in the skeleton has a matching slot and vice versa."""
|
|
placeholders = set(re.findall(r"\{([a-z_][a-z0-9_]*)\}", skeleton))
|
|
slot_names = {s.name for s in slots}
|
|
return placeholders == slot_names
|