Files
nextcraft/apps/ai-service/ai_service/variants/templates.py
T
CIAgent 12b2300f6f fix(P07): final review — CORS PUT, WS origin gate, ingest leak+O(n²), symlink escape, retry leak, doc-reality gaps
---ci---
phase: 7
milestone: v0.3
status: review
lessons:
  - P0 CORS: allow_methods lacked PUT while the build surface writes files with PUT — every cross-origin Save failed preflight; pinned with tests/api/test_cors.py
  - P0 ingest leak: queue-overflow flood path returned without the disconnect sentinel, parking the drainer forever (one leaked task-set per flooded trace); sentinel now always enqueued, real-server regression test added
  - P1 perf: flood cap counted rows via len(get_trace(...)) — O(trace) per append, O(n²) per session; TraceStore.count() (COUNT(*)) added and wired
  - P0 security: file routes followed exec-planted symlinks out of the workspace bind; _resolve_in_workspace refuses escapes (422), read/write now 404 on unknown sandboxes (was 500)
  - P1 security: WS ingest accepted any browser Origin (CORS middleware does not cover WS); localhost dev origins + no-Origin (capture agent) allowed, others 1008
  - P1 correctness: use-sandbox-session leaked a created sandbox on any mid-start failure (per-learner cap 1 → all retries 429 forever); failed starts now destroy what they created
  - P2 testing: reconnect-flush test killed mid-burst (nondeterministic under load, reproduced on pre-change code); now waits for server-side observation of the pre-kill burst — the underlying one-line replay-margin/ACK gap is documented for v0.4
  - maintainability: grading-store/templates/grading.ts docstrings claimed grading is variant-blind (stale pre-P4 text) — updated; ARCHITECTURE.md referenced nonexistent voice/openai_audio.py; dead if TYPE_CHECKING: pass blocks removed
---/ci---
2026-09-12 20:02:10 +00:00

306 lines
12 KiB
Python

"""Task template library for seeded variant generation (D-029, REQ-3-005).
A `TaskTemplate` binds a competency (D-021-aligned corpus ID), a statement
skeleton with `{slot}` placeholders, typed `ParameterSlot`s, difficulty-
normalization rubric anchors (the expected feature envelope that bounds
variant fairness in the a-5 envelope test — grader-prompt shipment is the
tracked P4 follow-up; grading is variant-blind today), and starter-file
scaffolds served into the sandbox workdir (wired in P6).
Slot sampling is PURE CODE: `random.Random(seed)` over typed slots — fully
reproducible for a given seed, independent of the LLM. The LLM only renders
the seeded slot values into the statement skeleton (D-020 defense).
"""
from __future__ import annotations
import random
import re
from typing import Literal
from pydantic import BaseModel, ConfigDict, Field, field_validator
SlotType = Literal["enum", "int_range", "string_set"]
class ParameterSlot(BaseModel):
"""One typed fill-in for a statement skeleton."""
model_config = ConfigDict(frozen=True)
name: str = Field(min_length=1)
type: SlotType
values: list[str] = Field(default_factory=list) # enum/string_set options
lo: int | None = None # int_range bounds
hi: int | None = None
@field_validator("values")
@classmethod
def _values_nonempty_for_enums(cls, v: list[str], info) -> list[str]:
if info.data.get("type") in ("enum", "string_set") and not v:
raise ValueError(f"slot {info.data.get('name')!r} needs values")
return v
def sample(self, rng: random.Random) -> str | int:
"""Deterministic sample from the seeded RNG. Validated after sampling."""
if self.type == "enum" or self.type == "string_set":
return rng.choice(self.values)
if self.type == "int_range":
lo = self.lo if self.lo is not None else 0
hi = self.hi if self.hi is not None else lo
if hi < lo:
raise ValueError(f"slot {self.name!r}: hi < lo")
return rng.randint(lo, hi)
raise ValueError(f"unsupported slot type: {self.type!r}")
def validate_value(self, value: str | int) -> bool:
"""Is `value` schema-valid for this slot? (params JSON gate, a-5.)"""
if self.type in ("enum", "string_set"):
return isinstance(value, str) and value in self.values
if self.type == "int_range":
lo = self.lo if self.lo is not None else 0
hi = self.hi if self.hi is not None else lo
return isinstance(value, int) and lo <= value <= hi
return False
class RubricAnchors(BaseModel):
"""Difficulty-normalization anchors for the grader (a-5).
Expected FEATURE ENVELOPE (digest-space): the expected effort band
for this template, so two variants of one template are held to the
same bar regardless of which slot values a learner drew. The a-5
envelope test (tests/variants/test_generator.py) binds variants to
these bands in code, and — since Phase 4 (MH#4) — the grading engine
ships this envelope into the grader prompt
(grading/engine._anchors_context) and stamps the variant seed on the
GradeRecord, so the anchors gate variant fairness in BOTH tests and
the live rubric.
"""
model_config = ConfigDict(frozen=True)
expected_edit_count_band: tuple[int, int]
expected_min_test_runs: int
expected_error_fix_cycles_band: tuple[int, int]
notes: str = ""
class TaskTemplate(BaseModel):
"""A reusable task shape; variants instantiate it per learner."""
model_config = ConfigDict(frozen=True)
id: str = Field(min_length=1)
competency_id: str = Field(min_length=1) # D-021 corpus alignment
title: str
statement_skeleton: str = Field(min_length=1) # {slot} placeholders
slots: list[ParameterSlot] = Field(min_length=1)
rubric_anchors: RubricAnchors
starter_files: dict[str, str] = Field(default_factory=dict) # path -> content
test_command: str
@field_validator("statement_skeleton")
@classmethod
def _skeleton_placeholders(cls, v: str) -> str:
if "{" not in v or "}" not in v:
raise ValueError("statement_skeleton needs at least one {slot}")
return v
def render(self, params: dict[str, str | int]) -> str:
"""Fill the skeleton with validated params."""
for slot in self.slots:
if slot.name not in params:
raise ValueError(f"missing param for slot {slot.name!r}")
if not slot.validate_value(params[slot.name]):
raise ValueError(f"invalid value for slot {slot.name!r}: {params[slot.name]!r}")
return self.statement_skeleton.format(**params)
def sample_params(self, seed: int) -> dict[str, str | int]:
"""Seeded, reproducible, schema-valid slot values (pure code)."""
rng = random.Random(seed)
return {slot.name: slot.sample(rng) for slot in self.slots}
# --- Template library (v0.3 initial set) --------------------------------------
# Competency IDs are D-021-aligned with the Python corpus
# (ai_service/corpus/learner_context.py) and the TS mock-data layer
# (packages/mock-data/competency-stacks.ts: deterministic cid() scheme).
TEMPLATES: dict[str, TaskTemplate] = {
"tpl-llm-judge": TaskTemplate(
id="tpl-llm-judge",
competency_id="stack-orchestration-c007",
title="Build an LLM-as-Judge Evaluator",
statement_skeleton=(
"Build a small LLM-as-judge evaluator for {domain} answers. "
"The judge must score each answer on {criterion} using a 0-4 scale, "
"return structured JSON, and handle at least {edge_cases} edge-case "
"answer classes (empty, off-topic, adversarial). Include a tiny "
"repro test set of at least {test_size} examples and print a summary "
"table of scores."
),
slots=[
ParameterSlot(
name="domain",
type="enum",
values=["customer-support", "code-review", "summarization", "tutoring"],
),
ParameterSlot(
name="criterion",
type="enum",
values=["factual-accuracy", "helpfulness", "safety", "completeness"],
),
ParameterSlot(name="edge_cases", type="int_range", lo=2, hi=4),
ParameterSlot(name="test_size", type="int_range", lo=3, hi=8),
],
rubric_anchors=RubricAnchors(
expected_edit_count_band=(3, 25),
expected_min_test_runs=2,
expected_error_fix_cycles_band=(0, 4),
notes="Slot draw changes the SCENARIO, not the engineering depth.",
),
starter_files={
"README.md": (
"# LLM-as-Judge Evaluator\n\n"
"Implement `judge.py`:\n"
"- `score(answer: str) -> dict` — 0-4 on the named criterion\n"
"- structured JSON output (schema below)\n"
"- edge-case classes handled explicitly\n"
"- `pytest` must pass\n"
),
"judge.py": "def score(answer: str) -> dict:\n raise NotImplementedError\n",
"test_judge.py": "def test_placeholder():\n assert True\n",
},
test_command="pytest -q",
),
"tpl-guardrail-schema": TaskTemplate(
id="tpl-guardrail-schema",
competency_id="stack-orchestration-c008",
title="Schema Guardrail Pipeline",
statement_skeleton=(
"Implement an output-validation guardrail for a model returning "
"{entity} records. Validate against a typed schema with {field_count} "
"required fields, coerce or reject {failure_mode} failures, and emit "
"a fallback response for invalid payloads. Cover with at least "
"{test_size} unit tests including malformed JSON."
),
slots=[
ParameterSlot(
name="entity",
type="enum",
values=["user-profile", "job-posting", "candidate", "invoice"],
),
ParameterSlot(
name="failure_mode",
type="enum",
values=["strict-reject", "coerce-when-safe"],
),
ParameterSlot(name="field_count", type="int_range", lo=4, hi=8),
ParameterSlot(name="test_size", type="int_range", lo=4, hi=10),
],
rubric_anchors=RubricAnchors(
expected_edit_count_band=(3, 30),
expected_min_test_runs=2,
expected_error_fix_cycles_band=(0, 5),
notes="All slot draws land in the same engineering band.",
),
starter_files={
"README.md": (
"# Schema Guardrail\n\nImplement `guardrail.py`:\n"
"- `validate(payload: dict) -> dict | Fallback`\n"
"- required-field checks, failure policy, fallback emission\n"
),
"guardrail.py": "def validate(payload: dict):\n raise NotImplementedError\n",
"test_guardrail.py": "def test_placeholder():\n assert True\n",
},
test_command="pytest -q",
),
"tpl-rag-chunker": TaskTemplate(
id="tpl-rag-chunker",
competency_id="stack-orchestration-c005",
title="RAG Chunking Strategy",
statement_skeleton=(
"Implement a document chunker for {doc_type} retrieval. Support "
"{strategy} chunking with a target size of ~{chunk_size} tokens, "
"preserve {invariant} across chunk boundaries, and evaluate overlap "
"quality with at least {test_size} fixture documents."
),
slots=[
ParameterSlot(
name="doc_type",
type="enum",
values=["technical-docs", "legal-contracts", "transcripts"],
),
ParameterSlot(
name="strategy",
type="enum",
values=["fixed-window", "semantic-boundary", "hybrid"],
),
ParameterSlot(
name="invariant",
type="enum",
values=["code-block-integrity", "section-headers", "sentence-completeness"],
),
ParameterSlot(name="chunk_size", type="int_range", lo=200, hi=800),
ParameterSlot(name="test_size", type="int_range", lo=3, hi=6),
],
rubric_anchors=RubricAnchors(
expected_edit_count_band=(4, 35),
expected_min_test_runs=2,
expected_error_fix_cycles_band=(0, 6),
notes="Strategy draw changes implementation shape, not depth.",
),
starter_files={
"README.md": (
"# RAG Chunker\n\nImplement `chunker.py`:\n"
"- `chunk(text: str) -> list[str]`\n- invariant preserved\n- tests green\n"
),
"chunker.py": "def chunk(text: str) -> list[str]:\n raise NotImplementedError\n",
"test_chunker.py": "def test_placeholder():\n assert True\n",
},
test_command="pytest -q",
),
}
_KNOWN_COMPETENCY_IDS: set[str] = {
# D-021: mirrored from ai_service/corpus/learner_context.py — the Python
# source of truth for stack-orchestration competencies used by v0.2 agents.
"stack-orchestration-c001",
"stack-orchestration-c002",
"stack-orchestration-c003",
"stack-orchestration-c004",
"stack-orchestration-c005",
"stack-orchestration-c007",
"stack-orchestration-c008",
"stack-orchestration-c011",
"stack-designer-c001",
"stack-designer-c002",
"stack-safety-c021",
}
def get_template(template_id: str) -> TaskTemplate | None:
return TEMPLATES.get(template_id)
def template_for_competency(competency_id: str) -> list[TaskTemplate]:
return [t for t in TEMPLATES.values() if t.competency_id == competency_id]
def validate_competency_binding() -> None:
"""All templates must bind to known D-021 corpus competency IDs."""
for t in TEMPLATES.values():
if t.competency_id not in _KNOWN_COMPETENCY_IDS:
raise ValueError(
f"template {t.id!r} binds unknown competency {t.competency_id!r}"
)
def slots_pattern_ok(skeleton: str, slots: list[ParameterSlot]) -> bool:
"""Every {placeholder} in the skeleton has a matching slot and vice versa."""
placeholders = set(re.findall(r"\{([a-z_][a-z0-9_]*)\}", skeleton))
slot_names = {s.name for s in slots}
return placeholders == slot_names