feat(P03): assessment grade endpoint (Wave 3)
Task 3-3-01: POST /v1/assessment/grade (engine over DI; GRADED -> 200 full record;
UNGRADABLE_* gate outcomes -> 200 gate records with missing_seqs/integrity_flag —
never 5xx, they are valid results; persistent D-020 failure -> 502, nothing persisted)
+ GET /v1/assessment/grade/{learner}/{task} (404 unknown). GradeStore wired into
lifespan alongside TraceStore (single db_path, per-connection WAL). Unknown task POST ->
200 UNGRADABLE_EMPTY_TRACE (engine cannot distinguish absent from empty; gate record
persisted so POST->GET round-trips).
11 endpoint tests green; suite 283 green; ruff clean.
---ci---
phase: 3
milestone: v0.3
status: execute
requirements: {covered: [REQ-3-004], partial: []}
---/ci---
This commit is contained in:
@@ -1,22 +1,86 @@
|
||||
"""POST /v1/assessment/evaluate — structured rubric scores (REQ-2-008).
|
||||
"""/v1/assessment — rubric evaluation + trace grading endpoints (REQ-2-008, REQ-3-004).
|
||||
|
||||
JSON response (not SSE): a pydantic-validated RubricScore. Unknown
|
||||
artifact → 404. The Assessor's structured output IS the payload.
|
||||
Two endpoint families share this router:
|
||||
|
||||
POST /v1/assessment/evaluate (v0.2, REQ-2-008) — corpus
|
||||
artifact evaluation through
|
||||
the Assessor agent.
|
||||
POST /v1/assessment/grade (v0.3, REQ-3-004) — grade a
|
||||
REAL process trace through
|
||||
the GradingEngine.
|
||||
GET /v1/assessment/grade/{learner_id}/{task_id} — stored latest grade.
|
||||
|
||||
Grading status-code mapping (the engine's outcomes are CONTRACT, not errors):
|
||||
|
||||
GradeRecord(verdict=GRADED) → 200 — rubric scores +
|
||||
verdict (in scores.verdict)
|
||||
+ digest summary.
|
||||
GradeRecord(UNGRADABLE_TRACE_INCOMPLETE) → 200 — the ungradable
|
||||
record IS a valid result:
|
||||
the trace cannot be graded,
|
||||
and the gate surfaces WHY
|
||||
(scores.missing_seqs +
|
||||
scores.integrity_flag).
|
||||
Persisted like any grade.
|
||||
GradeRecord(UNGRADABLE_EMPTY_TRACE) → 200 — no events stored for
|
||||
the pair. This covers BOTH
|
||||
a known pair whose trace
|
||||
ended up empty AND a task
|
||||
that never had a trace at
|
||||
all: the engine cannot
|
||||
distinguish them (zero
|
||||
stored events is zero
|
||||
events), and grading an
|
||||
absent trace genuinely has
|
||||
the empty-trace outcome —
|
||||
a 404 here would erase the
|
||||
durable gate record the
|
||||
engine persists for the
|
||||
pair. PLAN's 404 applies to
|
||||
GET of a never-graded pair.
|
||||
StructuredOutputError → 502 — the trace was
|
||||
gradable but the provider
|
||||
failed the D-020 budget;
|
||||
provider failure (bad
|
||||
gateway to the model), same
|
||||
mapping as evaluate.
|
||||
GET of an unknown (never-graded) pair → 404.
|
||||
|
||||
DI (D-027/D-032 house pattern): engine + store arrive via deps.get_grading_engine
|
||||
/ get_grade_store from app.state; this module owns all FastAPI wiring — the
|
||||
engine knows nothing of HTTP. UNGRADABLE_* bodies are rendered by the same
|
||||
GradeResponse model as GRADED ones (a gate record's `scores` holds the gate
|
||||
detail instead of rubric scores), so consumers read ONE shape.
|
||||
"""
|
||||
|
||||
from datetime import datetime
|
||||
from typing import Any
|
||||
|
||||
from fastapi import APIRouter, Depends, HTTPException
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from ..agents.assessor import RubricScore
|
||||
from ..agents.registry import AgentRegistry
|
||||
from ..agents.structured import StructuredOutputError
|
||||
from ..config import Settings
|
||||
from ..corpus.artifacts import get_artifact_bundle, get_transcript_for_artifact
|
||||
from ..corpus.learner_context import get_learner_context
|
||||
from .deps import get_agent_registry, get_provider, get_settings
|
||||
from ..grading.engine import GradingEngine
|
||||
from ..grading.store import GradeRecord, GradeStore
|
||||
from .deps import (
|
||||
get_agent_registry,
|
||||
get_grade_store,
|
||||
get_grading_engine,
|
||||
get_provider,
|
||||
get_settings,
|
||||
)
|
||||
|
||||
router = APIRouter(prefix="/v1")
|
||||
|
||||
|
||||
# --- v0.2 artifact evaluation (REQ-2-008) --------------------------------------
|
||||
|
||||
|
||||
class AssessmentRequest(BaseModel):
|
||||
artifact_id: str = Field(min_length=1)
|
||||
learner_id: str | None = None
|
||||
@@ -45,3 +109,72 @@ async def assessment_evaluate(
|
||||
status_code=502,
|
||||
detail=f"assessment evaluation failed: {exc}",
|
||||
) from exc
|
||||
|
||||
|
||||
# --- v0.3 trace grading (REQ-3-004) ---------------------------------------------
|
||||
|
||||
|
||||
class GradeRequest(BaseModel):
|
||||
learner_id: str = Field(min_length=1)
|
||||
task_id: str = Field(min_length=1)
|
||||
|
||||
|
||||
class GradeResponse(BaseModel):
|
||||
"""GradeRecord over HTTP — one shape for GRADED and UNGRADABLE_* alike.
|
||||
|
||||
`scores` holds the validated rubric (criteria 0-4, strengths, gaps,
|
||||
rubric verdict) for a GRADED record, or the gate detail
|
||||
({integrity_flag, missing_seqs}) for an UNGRADABLE_* record — never both.
|
||||
`digest` is the compact trace summary that fed the rubric prompt (empty
|
||||
for gate records: nothing was graded).
|
||||
"""
|
||||
|
||||
learner_id: str
|
||||
task_id: str
|
||||
variant_seed: str | None
|
||||
digest: dict[str, Any]
|
||||
scores: dict[str, Any]
|
||||
verdict: str
|
||||
model: str
|
||||
created_at: datetime
|
||||
|
||||
|
||||
def _grade_response(record: GradeRecord) -> GradeResponse:
|
||||
return GradeResponse.model_validate(record, from_attributes=True)
|
||||
|
||||
|
||||
@router.post("/assessment/grade", response_model=GradeResponse)
|
||||
async def assessment_grade(
|
||||
body: GradeRequest,
|
||||
engine: GradingEngine = Depends(get_grading_engine),
|
||||
) -> GradeResponse:
|
||||
"""Run the grading engine for one (learner_id, task_id) trace.
|
||||
|
||||
Gate outcomes (UNGRADABLE_*) are 200s — they are first-class results the
|
||||
engine persists, not failures. Only a provider that exhausts the D-020
|
||||
budget turns into a 502; nothing is persisted on that path.
|
||||
"""
|
||||
try:
|
||||
record = await engine.grade(body.learner_id, body.task_id)
|
||||
except StructuredOutputError as exc:
|
||||
raise HTTPException(
|
||||
status_code=502,
|
||||
detail=f"grading failed: {exc}",
|
||||
) from exc
|
||||
return _grade_response(record)
|
||||
|
||||
|
||||
@router.get("/assessment/grade/{learner_id}/{task_id}", response_model=GradeResponse)
|
||||
async def assessment_get_grade(
|
||||
learner_id: str,
|
||||
task_id: str,
|
||||
store: GradeStore = Depends(get_grade_store),
|
||||
) -> GradeResponse:
|
||||
"""Latest stored grade for the pair; 404 when none was ever stored."""
|
||||
record = store.get(learner_id, task_id)
|
||||
if record is None:
|
||||
raise HTTPException(
|
||||
status_code=404,
|
||||
detail=f"no stored grade for {learner_id!r}/{task_id!r}",
|
||||
)
|
||||
return _grade_response(record)
|
||||
|
||||
@@ -5,6 +5,8 @@ from fastapi import Request
|
||||
from ..agents.registry import AgentRegistry
|
||||
from ..agents.session import SessionStore
|
||||
from ..config import Settings
|
||||
from ..grading.engine import GradingEngine
|
||||
from ..grading.store import GradeStore
|
||||
from ..llm.base import LLMProvider
|
||||
from ..sandbox.manager import SandboxManager
|
||||
from ..sandbox.workdir import SandboxDir
|
||||
@@ -43,3 +45,11 @@ def get_trace_store(request: Request) -> TraceStore:
|
||||
|
||||
def get_trace_integrity(request: Request) -> TraceIntegrityMap:
|
||||
return request.app.state.trace_integrity
|
||||
|
||||
|
||||
def get_grade_store(request: Request) -> GradeStore:
|
||||
return request.app.state.grade_store
|
||||
|
||||
|
||||
def get_grading_engine(request: Request) -> GradingEngine:
|
||||
return request.app.state.grading_engine
|
||||
|
||||
@@ -21,6 +21,8 @@ from .api import (
|
||||
telemetry_router,
|
||||
)
|
||||
from .config import Settings
|
||||
from .grading.engine import GradingEngine
|
||||
from .grading.store import SQLiteGradeStore
|
||||
from .llm import create_provider
|
||||
from .sandbox import SandboxManager, UnshareBackend
|
||||
from .telemetry.ingest import TraceIntegrityMap
|
||||
@@ -71,6 +73,27 @@ def create_app(settings: Settings | None = None) -> FastAPI:
|
||||
if getattr(app.state, "trace_integrity", None) is None:
|
||||
app.state.trace_integrity = TraceIntegrityMap()
|
||||
|
||||
# Grading persistence + engine (REQ-3-004): GradeStore from the same
|
||||
# SQLite file as traces (D-027), one GradingEngine singleton wired
|
||||
# through app.state — the engine receives its stores via constructor
|
||||
# DI and knows nothing of FastAPI (api/ owns composition). Tests may
|
||||
# pre-set app.state.grade_store / app.state.grading_engine (the same
|
||||
# state-injection override as sandbox_manager/trace_store) to swap
|
||||
# either; the lifespan adopts a pre-set store but NEVER rebuilds a
|
||||
# pre-set engine (its provider binding is part of the test fixture).
|
||||
grade_store = getattr(app.state, "grade_store", None)
|
||||
if grade_store is None:
|
||||
grade_store = SQLiteGradeStore(db_path=settings.db_path)
|
||||
app.state.grade_store = grade_store
|
||||
if getattr(app.state, "grading_engine", None) is None:
|
||||
app.state.grading_engine = GradingEngine(
|
||||
trace_store,
|
||||
grade_store,
|
||||
app.state.trace_integrity,
|
||||
app.state.provider,
|
||||
model=settings.model,
|
||||
)
|
||||
|
||||
async def _reaper_loop() -> None:
|
||||
# Wall-clock timeout + G-2 workdir-size sweep, one pass per tick.
|
||||
while True:
|
||||
@@ -91,6 +114,7 @@ def create_app(settings: Settings | None = None) -> FastAPI:
|
||||
# everything live; workdirs stay on disk for snapshot restore.
|
||||
await manager.destroy_all()
|
||||
trace_store.close()
|
||||
grade_store.close()
|
||||
await app.state.http_client.aclose()
|
||||
|
||||
app = FastAPI(title="Nextcraft AI Service", version="0.3.0", lifespan=lifespan)
|
||||
|
||||
@@ -0,0 +1,397 @@
|
||||
"""Assessment grade endpoint tests — grading engine over HTTP (Task 3-3-01).
|
||||
|
||||
Contract under test (api/assessment.py, REQ-3-004):
|
||||
|
||||
POST /v1/assessment/grade {learner_id, task_id}
|
||||
GRADED → 200, rubric scores + digest summary
|
||||
UNGRADABLE_TRACE_INCOMPLETE → 200, gate record (missing_seqs /
|
||||
integrity_flag surfaced in scores)
|
||||
UNGRADABLE_EMPTY_TRACE → 200, gate record (documented choice: an
|
||||
unknown task is ALSO an empty pair; the
|
||||
engine cannot distinguish, and the gate
|
||||
outcome is a durable first-class result)
|
||||
StructuredOutputError → 502 (provider exhausted the D-020 budget)
|
||||
GET /v1/assessment/grade/{learner_id}/{task_id}
|
||||
stored latest grade → 200 (same scores as the POST)
|
||||
never-graded pair → 404
|
||||
|
||||
Wiring: per-test tmp-path SQLite stores + a pre-set GradingEngine
|
||||
(state-injection override — the lifespan adopts trace_store/grade_store/
|
||||
trace_integrity/grading_engine from app.state instead of constructing
|
||||
them; same pattern as test_telemetry_ingest.py / test_sandboxes.py). The
|
||||
engine binds a scripted provider so each test controls the LLM exactly,
|
||||
and counts calls so gate tests can assert the LLM was never reached
|
||||
(G-4 holds through the whole HTTP stack).
|
||||
|
||||
Zero network: providers are MockProvider subclasses only (conftest rule).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Iterator
|
||||
from datetime import UTC, datetime, timedelta
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
from ai_service.config import Settings
|
||||
from ai_service.corpus.trace_fixtures import STRONG_BUILDER
|
||||
from ai_service.grading.engine import GradingEngine
|
||||
from ai_service.grading.store import SQLiteGradeStore
|
||||
from ai_service.llm.mock import MockProvider, ScriptedJSONProvider
|
||||
from ai_service.main import create_app
|
||||
from ai_service.telemetry.ingest import TraceIntegrityMap
|
||||
from ai_service.telemetry.models import TelemetryEvent
|
||||
from ai_service.telemetry.store import SQLiteTraceStore
|
||||
|
||||
LEARNER = "grade-learner"
|
||||
TASK = "task-grade-1"
|
||||
T0 = datetime(2026, 9, 12, 3, 0, 0, tzinfo=UTC)
|
||||
|
||||
#: Canonical well-formed rubric payload (matches grading.engine.RubricScore).
|
||||
RUBRIC_PAYLOAD: dict = {
|
||||
"criteria": {
|
||||
"process_quality": 4,
|
||||
"correctness": 4,
|
||||
"debugging_discipline": 3,
|
||||
"test_usage": 4,
|
||||
},
|
||||
"strengths": ["tight edit-test loops throughout"],
|
||||
"gaps": ["final commit discipline loose"],
|
||||
"verdict": "mastered",
|
||||
}
|
||||
|
||||
|
||||
class CountingScriptedProvider(ScriptedJSONProvider):
|
||||
"""ScriptedJSONProvider that counts chat() calls.
|
||||
|
||||
Gate tests assert calls == 0 (the LLM is never reached through the
|
||||
whole HTTP stack — G-4); happy-path tests sanity-check calls >= 1.
|
||||
"""
|
||||
|
||||
def __init__(self, payload: dict) -> None:
|
||||
super().__init__(payload)
|
||||
self.calls = 0
|
||||
|
||||
async def chat(self, messages, *, model, temperature=0.7, response_format=None):
|
||||
self.calls += 1
|
||||
return await super().chat(
|
||||
messages, model=model, temperature=temperature, response_format=response_format
|
||||
)
|
||||
|
||||
|
||||
def _event(seq: int, kind: str, payload: dict | None = None, offset_s: float = 0.0):
|
||||
return TelemetryEvent(
|
||||
learner_id=LEARNER,
|
||||
task_id=TASK,
|
||||
seq=seq,
|
||||
kind=kind,
|
||||
payload=payload or {},
|
||||
ts=T0 + timedelta(seconds=offset_s),
|
||||
sandbox_id="sbx-grade",
|
||||
)
|
||||
|
||||
|
||||
def _complete_trace() -> list[TelemetryEvent]:
|
||||
"""Contiguous seq 0..6 — a gradeable trace (gap-free, unflagged)."""
|
||||
return [
|
||||
_event(0, "activity", {"state": "starting"}, 0.0),
|
||||
_event(1, "file_diff", {"path": "a.py", "added": 12}, 10.0),
|
||||
_event(2, "command", {"cmd": "pytest -q"}, 20.0),
|
||||
_event(3, "test_result", {"passed": False, "exit_code": 1}, 25.0),
|
||||
_event(4, "file_diff", {"path": "a.py", "added": 4, "removed": 2}, 40.0),
|
||||
_event(5, "command", {"cmd": "pytest -q"}, 60.0),
|
||||
_event(6, "test_result", {"passed": True, "exit_code": 0}, 65.0),
|
||||
]
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def trace_store(tmp_path: Path) -> Iterator[SQLiteTraceStore]:
|
||||
store = SQLiteTraceStore(db_path=tmp_path / "traces.db")
|
||||
yield store
|
||||
store.close()
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def grade_store(tmp_path: Path) -> Iterator[SQLiteGradeStore]:
|
||||
store = SQLiteGradeStore(db_path=tmp_path / "grades.db")
|
||||
yield store
|
||||
store.close()
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def integrity() -> TraceIntegrityMap:
|
||||
return TraceIntegrityMap()
|
||||
|
||||
|
||||
def _make_client(
|
||||
tmp_path: Path,
|
||||
trace_store: SQLiteTraceStore,
|
||||
grade_store: SQLiteGradeStore,
|
||||
integrity: TraceIntegrityMap,
|
||||
provider: MockProvider,
|
||||
) -> TestClient:
|
||||
"""App + TestClient with stores, integrity map and a pre-set engine.
|
||||
|
||||
The lifespan adopts every pre-set service (state-injection override);
|
||||
the engine binds OUR provider, so the fixture — not Settings — scripts
|
||||
the LLM. Cloud-free guard: the provider must be a MockProvider family
|
||||
member (conftest rule, enforced here because this module builds its
|
||||
own client rather than consuming the conftest one).
|
||||
"""
|
||||
assert isinstance(provider, MockProvider)
|
||||
settings = Settings(
|
||||
provider="mock",
|
||||
db_path=tmp_path / "grading-test.db",
|
||||
sandbox_dir=tmp_path / "sandboxes",
|
||||
)
|
||||
app = create_app(settings)
|
||||
app.state.trace_store = trace_store
|
||||
app.state.grade_store = grade_store
|
||||
app.state.trace_integrity = integrity
|
||||
app.state.grading_engine = GradingEngine(
|
||||
trace_store, grade_store, integrity, provider, model="gemma4:31b"
|
||||
)
|
||||
return TestClient(app)
|
||||
|
||||
|
||||
def _seed(store: SQLiteTraceStore, events: list[TelemetryEvent]) -> None:
|
||||
for e in events:
|
||||
store.append(e)
|
||||
|
||||
|
||||
def _post(client: TestClient, learner: str = LEARNER, task: str = TASK):
|
||||
return client.post("/v1/assessment/grade", json={"learner_id": learner, "task_id": task})
|
||||
|
||||
|
||||
def _get(client: TestClient, learner: str = LEARNER, task: str = TASK):
|
||||
return client.get(f"/v1/assessment/grade/{learner}/{task}")
|
||||
|
||||
|
||||
# -- happy path: complete trace → GRADED -----------------------------------------
|
||||
|
||||
|
||||
class TestGraded:
|
||||
def test_post_complete_trace_returns_rubric_scores_and_digest(
|
||||
self, tmp_path, trace_store, grade_store, integrity
|
||||
):
|
||||
_seed(trace_store, _complete_trace())
|
||||
provider = CountingScriptedProvider(RUBRIC_PAYLOAD)
|
||||
|
||||
with _make_client(tmp_path, trace_store, grade_store, integrity, provider) as client:
|
||||
response = _post(client)
|
||||
|
||||
assert response.status_code == 200
|
||||
body = response.json()
|
||||
assert body["verdict"] == "GRADED"
|
||||
assert body["learner_id"] == LEARNER
|
||||
assert body["task_id"] == TASK
|
||||
assert body["scores"] == RUBRIC_PAYLOAD # rubric verdict rides in scores
|
||||
assert body["scores"]["verdict"] == "mastered"
|
||||
assert body["variant_seed"] is None # null until P4 (D-029)
|
||||
assert body["model"] == "gemma4:31b" # provenance travels
|
||||
assert provider.calls == 1 # happy path: exactly one LLM call
|
||||
# digest summary rides along (D-028 reproducible input)
|
||||
assert body["digest"]["event_count"] == 7
|
||||
assert body["digest"]["final_test_status"] == "pass"
|
||||
|
||||
def test_corpus_fixture_trace_grades(
|
||||
self, tmp_path, trace_store, grade_store, integrity
|
||||
):
|
||||
"""The strong-builder corpus fixture (real calibration trace) is
|
||||
gradeable over HTTP end to end."""
|
||||
_seed(trace_store, STRONG_BUILDER.events)
|
||||
provider = CountingScriptedProvider(RUBRIC_PAYLOAD)
|
||||
|
||||
with _make_client(tmp_path, trace_store, grade_store, integrity, provider) as client:
|
||||
response = _post(
|
||||
client, learner=STRONG_BUILDER.learner_id, task=STRONG_BUILDER.task_id
|
||||
)
|
||||
|
||||
assert response.status_code == 200
|
||||
body = response.json()
|
||||
assert body["verdict"] == "GRADED"
|
||||
assert body["digest"]["event_count"] == len(STRONG_BUILDER.event_specs)
|
||||
|
||||
def test_get_after_post_returns_same_stored_scores(
|
||||
self, tmp_path, trace_store, grade_store, integrity
|
||||
):
|
||||
_seed(trace_store, _complete_trace())
|
||||
provider = CountingScriptedProvider(RUBRIC_PAYLOAD)
|
||||
|
||||
with _make_client(tmp_path, trace_store, grade_store, integrity, provider) as client:
|
||||
posted = _post(client)
|
||||
assert posted.status_code == 200
|
||||
|
||||
fetched = _get(client)
|
||||
assert fetched.status_code == 200
|
||||
stored = fetched.json()
|
||||
|
||||
assert stored["scores"] == posted.json()["scores"]
|
||||
assert stored["verdict"] == "GRADED"
|
||||
assert stored["digest"] == posted.json()["digest"]
|
||||
assert stored["created_at"] == posted.json()["created_at"]
|
||||
|
||||
def test_post_regrade_upserts_get_returns_latest(
|
||||
self, tmp_path, trace_store, grade_store, integrity
|
||||
):
|
||||
"""POST twice → one row holding the LATEST grade (GradeStore upsert)."""
|
||||
_seed(trace_store, _complete_trace())
|
||||
provider = CountingScriptedProvider(RUBRIC_PAYLOAD)
|
||||
|
||||
with _make_client(tmp_path, trace_store, grade_store, integrity, provider) as client:
|
||||
first = _post(client)
|
||||
assert first.status_code == 200
|
||||
assert first.json()["scores"]["verdict"] == "mastered"
|
||||
|
||||
# the same provider now scripts a different (weaker) rubric
|
||||
updated = {
|
||||
**RUBRIC_PAYLOAD,
|
||||
"criteria": {**RUBRIC_PAYLOAD["criteria"], "process_quality": 1},
|
||||
"verdict": "not_yet",
|
||||
}
|
||||
provider.payload = updated
|
||||
second = _post(client)
|
||||
assert second.status_code == 200
|
||||
assert second.json()["scores"] == updated
|
||||
|
||||
# GET returns the LATEST grade, not the first
|
||||
fetched = _get(client)
|
||||
assert fetched.json()["scores"] == updated
|
||||
assert fetched.json()["scores"]["verdict"] == "not_yet"
|
||||
|
||||
# one row, latest wins (upsert, not append)
|
||||
grades = grade_store.list_for_learner(LEARNER)
|
||||
assert len(grades) == 1
|
||||
assert grades[0].scores == updated
|
||||
|
||||
|
||||
# -- G-4 gate outcomes over HTTP: 200 with the ungradable record -------------------
|
||||
|
||||
|
||||
class TestGateOutcomes:
|
||||
def test_gapped_trace_200_ungradable_incomplete_with_missing_seqs(
|
||||
self, tmp_path, trace_store, grade_store, integrity
|
||||
):
|
||||
_seed(trace_store, [e for e in _complete_trace() if e.seq != 2]) # seq 2 missing
|
||||
provider = CountingScriptedProvider(RUBRIC_PAYLOAD)
|
||||
|
||||
with _make_client(tmp_path, trace_store, grade_store, integrity, provider) as client:
|
||||
response = _post(client)
|
||||
|
||||
assert response.status_code == 200 # ungradable IS a valid result — not 5xx
|
||||
body = response.json()
|
||||
assert body["verdict"] == "UNGRADABLE_TRACE_INCOMPLETE"
|
||||
assert body["scores"]["missing_seqs"] == [2]
|
||||
assert body["scores"]["integrity_flag"] is None
|
||||
assert body["digest"] == {} # nothing was graded
|
||||
assert body["model"] == "none" # no LLM involved (honest provenance)
|
||||
assert provider.calls == 0, "LLM was called despite a gapped trace (G-4)"
|
||||
|
||||
def test_flooded_trace_200_ungradable_incomplete_with_integrity_reason(
|
||||
self, tmp_path, trace_store, grade_store, integrity
|
||||
):
|
||||
"""Integrity-flagged trace (G-3 INCOMPLETE_FLOODED) surfaces the flag
|
||||
reason in scores — the same verdict as gaps, different detail."""
|
||||
_seed(trace_store, _complete_trace())
|
||||
integrity.mark(LEARNER, TASK, "INCOMPLETE_FLOODED")
|
||||
provider = CountingScriptedProvider(RUBRIC_PAYLOAD)
|
||||
|
||||
with _make_client(tmp_path, trace_store, grade_store, integrity, provider) as client:
|
||||
response = _post(client)
|
||||
|
||||
assert response.status_code == 200
|
||||
body = response.json()
|
||||
assert body["verdict"] == "UNGRADABLE_TRACE_INCOMPLETE"
|
||||
assert body["scores"]["integrity_flag"] == "INCOMPLETE_FLOODED"
|
||||
assert body["scores"]["missing_seqs"] == [] # rows complete but untrusted
|
||||
assert provider.calls == 0, "LLM was called despite an integrity flag (G-4)"
|
||||
|
||||
def test_empty_trace_200_ungradable_empty(
|
||||
self, tmp_path, trace_store, grade_store, integrity
|
||||
):
|
||||
"""Zero events for the pair (including a task that never had a
|
||||
trace) → UNGRADABLE_EMPTY_TRACE, persisted like any grade."""
|
||||
provider = CountingScriptedProvider(RUBRIC_PAYLOAD)
|
||||
|
||||
with _make_client(tmp_path, trace_store, grade_store, integrity, provider) as client:
|
||||
response = _post(client, learner="ghost-learner", task="never-started")
|
||||
|
||||
assert response.status_code == 200
|
||||
body = response.json()
|
||||
assert body["verdict"] == "UNGRADABLE_EMPTY_TRACE"
|
||||
assert body["scores"] == {"integrity_flag": None, "missing_seqs": []}
|
||||
assert body["digest"] == {}
|
||||
assert body["model"] == "none"
|
||||
assert provider.calls == 0
|
||||
|
||||
# the gate record is durable: GET now returns it (no longer 404)
|
||||
with _make_client(tmp_path, trace_store, grade_store, integrity, provider) as client:
|
||||
fetched = _get(client, learner="ghost-learner", task="never-started")
|
||||
assert fetched.status_code == 200
|
||||
assert fetched.json()["verdict"] == "UNGRADABLE_EMPTY_TRACE"
|
||||
|
||||
|
||||
# -- provider failure → 502 --------------------------------------------------------
|
||||
|
||||
|
||||
class TestProviderFailure:
|
||||
def test_persistent_malformed_llm_output_502(
|
||||
self, tmp_path, trace_store, grade_store, integrity
|
||||
):
|
||||
"""Stock MockProvider: its json_object reply is wrong-shaped, so the
|
||||
D-020 defense exhausts both attempts → endpoint maps to 502 and
|
||||
nothing is persisted."""
|
||||
_seed(trace_store, _complete_trace())
|
||||
provider = MockProvider() # always malformed for the rubric schema
|
||||
|
||||
with _make_client(tmp_path, trace_store, grade_store, integrity, provider) as client:
|
||||
response = _post(client)
|
||||
|
||||
assert response.status_code == 502
|
||||
assert "grading failed" in response.json()["detail"]
|
||||
# no fabricated/partial record was persisted
|
||||
assert grade_store.get(LEARNER, TASK) is None
|
||||
|
||||
def test_502_leaves_earlier_grade_intact(
|
||||
self, tmp_path, trace_store, grade_store, integrity
|
||||
):
|
||||
"""A failed regrade must not clobber the previously stored grade."""
|
||||
_seed(trace_store, _complete_trace())
|
||||
scripted = CountingScriptedProvider(RUBRIC_PAYLOAD)
|
||||
with _make_client(tmp_path, trace_store, grade_store, integrity, scripted) as client:
|
||||
assert _post(client).status_code == 200
|
||||
|
||||
# regrade attempt hits a provider that now always fails
|
||||
broken = MockProvider()
|
||||
with _make_client(tmp_path, trace_store, grade_store, integrity, broken) as client:
|
||||
assert _post(client).status_code == 502
|
||||
fetched = _get(client)
|
||||
|
||||
assert fetched.status_code == 200 # earlier grade still readable
|
||||
assert fetched.json()["scores"] == RUBRIC_PAYLOAD
|
||||
|
||||
|
||||
# -- stored-grade reads ------------------------------------------------------------
|
||||
|
||||
|
||||
class TestStoredGradeReads:
|
||||
def test_get_unknown_pair_404(self, tmp_path, trace_store, grade_store, integrity):
|
||||
provider = CountingScriptedProvider(RUBRIC_PAYLOAD)
|
||||
with _make_client(tmp_path, trace_store, grade_store, integrity, provider) as client:
|
||||
response = _get(client, learner="nobody", task="never-graded")
|
||||
|
||||
assert response.status_code == 404
|
||||
assert "no stored grade" in response.json()["detail"]
|
||||
|
||||
def test_missing_body_fields_422(self, tmp_path, trace_store, grade_store, integrity):
|
||||
provider = CountingScriptedProvider(RUBRIC_PAYLOAD)
|
||||
with _make_client(tmp_path, trace_store, grade_store, integrity, provider) as client:
|
||||
assert client.post("/v1/assessment/grade", json={}).status_code == 422
|
||||
assert (
|
||||
client.post(
|
||||
"/v1/assessment/grade", json={"learner_id": LEARNER}
|
||||
).status_code
|
||||
== 422
|
||||
)
|
||||
Reference in New Issue
Block a user