Compare commits
33
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
36685a3f16 | ||
|
|
c43f330e2e | ||
|
|
fedb49d322 | ||
|
|
e46b4c5892 | ||
|
|
90503cadd6 | ||
|
|
dceb10129f | ||
|
|
a06dd3d4b0 | ||
|
|
15ef7c65e9 | ||
|
|
c864fe0479 | ||
|
|
c412f1ea75 | ||
|
|
182aa7c47f | ||
|
|
1efbf672bd | ||
|
|
b4872ba6ce | ||
|
|
265d70ed91 | ||
|
|
909bf75edc | ||
|
|
2896bb691c | ||
|
|
a71355e005 | ||
|
|
b1897c01bc | ||
|
|
5d23b747ff | ||
|
|
447e702520 | ||
|
|
2ace54bce1 | ||
|
|
5590ef20b5 | ||
|
|
134f52d3d3 | ||
|
|
8c1e56366b | ||
|
|
101143e67b | ||
|
|
9389b8425b | ||
|
|
47f072ee05 | ||
|
|
8c60f7cc85 | ||
|
|
e60aacc3dc | ||
|
|
02e2ae0afb | ||
|
|
8f06a4c99f | ||
|
|
191b6a9d85 | ||
|
|
2bb0acac19 |
+5
-1
@@ -8,7 +8,11 @@ dist/
|
||||
build/
|
||||
.venv/
|
||||
node_modules/
|
||||
data/
|
||||
data/*
|
||||
# IMP-46 u6 — track only the frame_cache directory marker; cached payloads stay ignored.
|
||||
!data/frame_cache/
|
||||
data/frame_cache/*
|
||||
!data/frame_cache/.gitkeep
|
||||
|
||||
# session workspace (push X — 작업 흐름 trace, 사용자 결정 2026-05-08)
|
||||
forex/
|
||||
|
||||
@@ -19,6 +19,20 @@ interface FramePanelProps {
|
||||
onNoDesignToggle: () => void;
|
||||
}
|
||||
|
||||
// ─── IMP-41 u3 — application_mode consequence tooltip map (issue #70) ────────
|
||||
// Keyed by application_mode VALUE (backend authoritative), NOT V4 label.
|
||||
// Source = src/phase_z2_pipeline.py APPLICATION_MODE_BY_V4_LABEL (:107-112)
|
||||
// emitted via Step 9 unit.application_candidates[] and forwarded by
|
||||
// designAgentApi.ts (IMP-41 u2). When applicationMode is absent (legacy
|
||||
// fixtures pre-IMP-32, or candidate filtered out at Step 9) the tooltip
|
||||
// falls back to the raw V4 label string per Stage 2 contract.
|
||||
const APPLICATION_MODE_TOOLTIP_KR: Record<string, string> = {
|
||||
direct_insert: "코드 직접 적용",
|
||||
same_frame_with_adjustment: "AI 보강 필요",
|
||||
layout_or_region_change: "AI restructure 필요",
|
||||
exclude: "render path 제외",
|
||||
};
|
||||
|
||||
export default function FramePanel({
|
||||
slidePlan,
|
||||
selectedZone,
|
||||
@@ -46,6 +60,26 @@ export default function FramePanel({
|
||||
return userSelection.overrides.zone_frames[targetRegion.id] || targetRegion.frame_match_strategy.frame_id;
|
||||
}, [selectedZone, selectedRegion, userSelection.overrides.zone_frames]);
|
||||
|
||||
// IMP-47B u11 — reject-click confirm guard. Per #76 policy: 사용자가 reject
|
||||
// 카드 명시 클릭 → backend `--override-frame` 전달 + reject frame 유지 + AI 재구성.
|
||||
// The window.confirm makes the AI-rebuild intent explicit (deselecting an
|
||||
// already-applied reject frame does not prompt). Pure UX gate — no state
|
||||
// mutation here; the parent `onFrameSelect` still owns the override apply.
|
||||
const handleFrameSelect = React.useCallback(
|
||||
(candidate: FrameCandidate) => {
|
||||
const isReject = candidate.label === "reject";
|
||||
const alreadyApplied = currentFrameId === candidate.id;
|
||||
if (isReject && !alreadyApplied) {
|
||||
const ok = window.confirm(
|
||||
`"${candidate.name}" 은 V4 reject 라벨입니다.\n선택 시 frame 은 유지되고 AI 가 콘텐츠를 frame 구조에 맞게 재구성합니다.\n계속하시겠습니까?`,
|
||||
);
|
||||
if (!ok) return;
|
||||
}
|
||||
onFrameSelect(candidate.id);
|
||||
},
|
||||
[currentFrameId, onFrameSelect],
|
||||
);
|
||||
|
||||
if (!selectedZone) {
|
||||
return (
|
||||
<div className="h-full flex flex-col items-center justify-center bg-slate-50 p-8 text-center text-slate-400">
|
||||
@@ -82,11 +116,67 @@ export default function FramePanel({
|
||||
) : (
|
||||
candidates.map((candidate, index) => {
|
||||
const isSelected = currentFrameId === candidate.id;
|
||||
|
||||
|
||||
const isReject = candidate.label === "reject";
|
||||
// catalog 미등록 = backend Step 7-A 가 override 시도해도 skip.
|
||||
// catalogRegistered === false 만 체크 (undefined = 정보 없음, 일반 처리).
|
||||
const isCatalogMissing = candidate.catalogRegistered === false;
|
||||
|
||||
// ─── IMP-29 u3 — IMP-05 L2 candidate_evidence surface ───────────
|
||||
// All evidence fields optional; silent degradation when undefined
|
||||
// (pre-IMP-05 fixtures fall back to label/catalogRegistered only).
|
||||
const isFilteredDirect = candidate.filteredForDirectExecution === true;
|
||||
const hasDecision = candidate.decision === "selected" || candidate.decision === "skipped";
|
||||
const isSkipped = candidate.decision === "skipped";
|
||||
const isSelectedDecision = candidate.decision === "selected";
|
||||
const showRouteChip =
|
||||
candidate.routeHint && candidate.routeHint !== "direct_render";
|
||||
const showStatusChip =
|
||||
candidate.phaseZStatus && candidate.phaseZStatus !== "auto_renderable";
|
||||
const hasCapacityFit =
|
||||
candidate.capacityFit && candidate.capacityFit.fit_status;
|
||||
const capacityMismatch =
|
||||
hasCapacityFit && candidate.capacityFit!.fit_status !== "ok";
|
||||
|
||||
// Compose evidence tooltip lines (only when at least one signal present).
|
||||
const evidenceLines: string[] = [];
|
||||
if (candidate.decision) evidenceLines.push(`decision: ${candidate.decision}`);
|
||||
if (candidate.reason) evidenceLines.push(`reason: ${candidate.reason}`);
|
||||
if (candidate.routeHint) evidenceLines.push(`route: ${candidate.routeHint}`);
|
||||
if (candidate.phaseZStatus)
|
||||
evidenceLines.push(`phase_z_status: ${candidate.phaseZStatus}`);
|
||||
if (hasCapacityFit) {
|
||||
const cf = candidate.capacityFit!;
|
||||
const capacityLine =
|
||||
cf.fit_status === "ok"
|
||||
? `capacity: ok${
|
||||
typeof cf.item_count === "number"
|
||||
? ` (items=${cf.item_count})`
|
||||
: ""
|
||||
}`
|
||||
: `capacity: ${cf.fit_status}${
|
||||
cf.mismatch_reason ? ` — ${cf.mismatch_reason}` : ""
|
||||
}`;
|
||||
evidenceLines.push(capacityLine);
|
||||
}
|
||||
const evidenceTooltip =
|
||||
evidenceLines.length > 0 ? evidenceLines.join("\n") : undefined;
|
||||
|
||||
// Compose final tooltip: existing catalog/reject reasons first, then
|
||||
// evidence detail (preserves Phase Q tooltip semantics).
|
||||
const tooltipParts = [
|
||||
isCatalogMissing
|
||||
? "⚠ catalog 미등록 — render path 에서 적용 안 됨 (선택해도 backend 가 skip)"
|
||||
: null,
|
||||
isFilteredDirect
|
||||
? "⚠ filtered_for_direct_execution — MVP1 직접 렌더 경로 제외"
|
||||
: null,
|
||||
isReject ? "V4 reject — render path 비추천" : null,
|
||||
evidenceTooltip,
|
||||
].filter((s): s is string => Boolean(s));
|
||||
const composedTitle =
|
||||
tooltipParts.length > 0 ? tooltipParts.join("\n\n") : undefined;
|
||||
|
||||
return (
|
||||
<motion.div
|
||||
key={candidate.id}
|
||||
@@ -95,7 +185,7 @@ export default function FramePanel({
|
||||
className="w-full"
|
||||
>
|
||||
<button
|
||||
onClick={() => onFrameSelect(candidate.id)}
|
||||
onClick={() => handleFrameSelect(candidate)}
|
||||
draggable
|
||||
onDragStart={(e) => {
|
||||
e.dataTransfer.setData("frameId", candidate.id);
|
||||
@@ -105,17 +195,13 @@ export default function FramePanel({
|
||||
? 'border-blue-500 bg-white shadow-xl shadow-blue-500/10'
|
||||
: isCatalogMissing
|
||||
? 'border-slate-100 bg-slate-50/40 opacity-60 hover:opacity-90 hover:border-amber-200'
|
||||
: isFilteredDirect
|
||||
? 'border-slate-100 bg-slate-50/30 opacity-50 hover:opacity-90 hover:border-amber-200'
|
||||
: isReject
|
||||
? 'border-slate-100 bg-slate-50/30 opacity-50 hover:opacity-90 hover:border-slate-200'
|
||||
: 'border-slate-100 bg-slate-50/50 hover:border-slate-200 hover:bg-white'
|
||||
}`}
|
||||
title={
|
||||
isCatalogMissing
|
||||
? "⚠ catalog 미등록 — render path 에서 적용 안 됨 (선택해도 backend 가 skip)"
|
||||
: isReject
|
||||
? "V4 reject — render path 비추천"
|
||||
: undefined
|
||||
}
|
||||
title={composedTitle}
|
||||
>
|
||||
{/* Rank Badge */}
|
||||
<div className="absolute top-3 left-3 z-10">
|
||||
@@ -183,20 +269,91 @@ export default function FramePanel({
|
||||
</span>
|
||||
)}
|
||||
{/* V4 label badge */}
|
||||
{candidate.label && (
|
||||
{candidate.label && (() => {
|
||||
// IMP-41 u3 — applicationMode-keyed Korean consequence
|
||||
// tooltip with legacy fallback. applicationMode is
|
||||
// forwarded by designAgentApi.ts (u2) from Step 9
|
||||
// unit.application_candidates[]; undefined when the
|
||||
// backend did not emit a mapping for this candidate.
|
||||
const consequence = candidate.applicationMode
|
||||
? APPLICATION_MODE_TOOLTIP_KR[candidate.applicationMode]
|
||||
: undefined;
|
||||
const badgeTitle = consequence
|
||||
? `${consequence} (${candidate.applicationMode})`
|
||||
: `V4 label: ${candidate.label}`;
|
||||
return (
|
||||
<span
|
||||
className={`text-[8px] font-black uppercase tracking-tight px-1.5 py-0.5 rounded ${
|
||||
candidate.label === "use_as_is"
|
||||
? "bg-emerald-100 text-emerald-700"
|
||||
: candidate.label === "light_edit"
|
||||
? "bg-blue-100 text-blue-700"
|
||||
: candidate.label === "restructure"
|
||||
? "bg-amber-100 text-amber-700"
|
||||
: "bg-red-100 text-red-700"
|
||||
}`}
|
||||
title={badgeTitle}
|
||||
>
|
||||
{candidate.label}
|
||||
</span>
|
||||
);
|
||||
})()}
|
||||
{/* IMP-29 u3 — route hint chip (skip when direct_render = default). */}
|
||||
{showRouteChip && (
|
||||
<span
|
||||
className="text-[8px] font-black uppercase tracking-tight px-1.5 py-0.5 rounded bg-slate-100 text-slate-600"
|
||||
title={`route_hint: ${candidate.routeHint}`}
|
||||
>
|
||||
{candidate.routeHint === "deterministic_minor_adjustment"
|
||||
? "adapt"
|
||||
: candidate.routeHint === "ai_adaptation_required"
|
||||
? "ai req"
|
||||
: candidate.routeHint === "design_reference_only"
|
||||
? "ref"
|
||||
: candidate.routeHint}
|
||||
</span>
|
||||
)}
|
||||
{/* IMP-29 u3 — phase_z status warning chip (skip when auto_renderable). */}
|
||||
{showStatusChip && (
|
||||
<span
|
||||
className="text-[8px] font-black uppercase tracking-tight px-1.5 py-0.5 rounded bg-amber-50 text-amber-700"
|
||||
title={`phase_z_status: ${candidate.phaseZStatus}`}
|
||||
>
|
||||
{candidate.phaseZStatus!.replace(/_/g, " ")}
|
||||
</span>
|
||||
)}
|
||||
{/* IMP-29 u3 — capacity_fit indicator (ok = subtle, mismatch = warning). */}
|
||||
{hasCapacityFit && (
|
||||
<span
|
||||
className={`text-[8px] font-black uppercase tracking-tight px-1.5 py-0.5 rounded ${
|
||||
candidate.label === "use_as_is"
|
||||
? "bg-emerald-100 text-emerald-700"
|
||||
: candidate.label === "light_edit"
|
||||
? "bg-blue-100 text-blue-700"
|
||||
: candidate.label === "restructure"
|
||||
capacityMismatch
|
||||
? "bg-amber-100 text-amber-700"
|
||||
: "bg-red-100 text-red-700"
|
||||
: "bg-slate-100 text-slate-500"
|
||||
}`}
|
||||
title={`capacity_fit: ${candidate.capacityFit!.fit_status}${
|
||||
candidate.capacityFit!.mismatch_reason
|
||||
? ` — ${candidate.capacityFit!.mismatch_reason}`
|
||||
: ""
|
||||
}`}
|
||||
title={`V4 label: ${candidate.label}`}
|
||||
>
|
||||
{candidate.label}
|
||||
{capacityMismatch
|
||||
? `fit: ${candidate.capacityFit!.fit_status}`
|
||||
: "fit ok"}
|
||||
</span>
|
||||
)}
|
||||
{/* IMP-29 u3 — decision badge (Stage 2 contract: surface both selected & skipped). */}
|
||||
{hasDecision && (
|
||||
<span
|
||||
className={`text-[8px] font-black uppercase tracking-tight px-1.5 py-0.5 rounded ${
|
||||
isSelectedDecision
|
||||
? "bg-emerald-50 text-emerald-700"
|
||||
: "bg-red-50 text-red-600"
|
||||
}`}
|
||||
title={`decision: ${candidate.decision}${
|
||||
candidate.reason ? ` — ${candidate.reason}` : ""
|
||||
}`}
|
||||
>
|
||||
{isSkipped ? "skip" : "sel"}
|
||||
</span>
|
||||
)}
|
||||
{isSelected && (
|
||||
|
||||
@@ -293,7 +293,7 @@ export default function SlideCanvas({
|
||||
title="Phase Z 렌더 결과"
|
||||
className="w-full h-full border-0 block"
|
||||
scrolling="no"
|
||||
sandbox="allow-same-origin"
|
||||
sandbox="allow-same-origin allow-scripts"
|
||||
style={{ pointerEvents: isEditMode ? "auto" : "none" }}
|
||||
onLoad={(e) => {
|
||||
// IMP-14 (Step 13 A-4) — embedded vs standalone CSS reset 은 backend
|
||||
|
||||
@@ -21,6 +21,7 @@ import {
|
||||
runPipeline,
|
||||
loadRun,
|
||||
computeZonePositions,
|
||||
formatAiRepairHumanReviewMessage,
|
||||
type RunMeta,
|
||||
type PipelineOverrides,
|
||||
} from "../services/designAgentApi";
|
||||
@@ -370,6 +371,13 @@ export default function Home() {
|
||||
}));
|
||||
setRunMeta(runMeta);
|
||||
toast.success(`run "${result.run_id}" 완료 — ${runMeta.status}`);
|
||||
// IMP-47B u11 — surface Step 12 AI repair failure axes (error /
|
||||
// coverage_violated / unsupported_kind) as a human_review notification.
|
||||
// Auto-pipeline first ([[feedback_auto_pipeline_first]]): no review_queue
|
||||
// insertion — just an explicit error toast directing the user to pick
|
||||
// another frame or edit manually. Helper returns null on success path.
|
||||
const aiReviewMsg = formatAiRepairHumanReviewMessage(runMeta.ai_repair_status);
|
||||
if (aiReviewMsg) toast.error(aiReviewMsg);
|
||||
} catch (err) {
|
||||
console.error(err);
|
||||
toast.error(
|
||||
@@ -377,7 +385,7 @@ export default function Home() {
|
||||
);
|
||||
setState((p) => ({ ...p, isLoading: false }));
|
||||
}
|
||||
}, [state.uploadedFile]);
|
||||
}, [state.uploadedFile, state.slidePlan, state.userSelection, pendingZones, pendingLayout]);
|
||||
|
||||
// ── 섹션 드래그 앤 드롭 (Zone으로 재배치) ──
|
||||
const handleSectionDrop = useCallback((sectionId: string, zoneId: string) => {
|
||||
|
||||
@@ -223,6 +223,35 @@ export interface FilteredSectionReason {
|
||||
position?: string | null;
|
||||
}
|
||||
|
||||
// IMP-47B u11 — verbatim mirror of step20_slide_status.ai_repair_status (u8 schema).
|
||||
// Surfaces Step 12 AI repair outcomes so the frontend can render a
|
||||
// human_review notification when AI proposal validation, coverage, or call
|
||||
// itself failed. Enum / field names kept verbatim — no frontend redefinition.
|
||||
export interface AiRepairStatus {
|
||||
status: "ok" | "applied" | "unsupported_kind" | "coverage_violated" | "error" | string;
|
||||
counts: {
|
||||
total: number;
|
||||
applied: number;
|
||||
no_proposal: number;
|
||||
no_zone_match: number;
|
||||
unsupported_kind: number;
|
||||
error: number;
|
||||
};
|
||||
unsupported_kind_records: Array<{
|
||||
unit_index?: number | null;
|
||||
source_section_ids: string[];
|
||||
apply_status: string;
|
||||
}>;
|
||||
error_records: Array<{
|
||||
unit_index?: number | null;
|
||||
source_section_ids: string[];
|
||||
error: string;
|
||||
}>;
|
||||
coverage_status: string;
|
||||
dropped_section_ids: string[];
|
||||
human_review_required: boolean;
|
||||
}
|
||||
|
||||
export interface RunMeta {
|
||||
run_id: string;
|
||||
mdx_path: string;
|
||||
@@ -237,6 +266,38 @@ export interface RunMeta {
|
||||
layout_candidates: string[]; // step07 layout_candidates list
|
||||
region_layout_candidates_by_zone: Record<string, string[]>; // step08 placeholder
|
||||
display_strategy_candidates_by_zone: Record<string, string[]>; // step08 placeholder
|
||||
/** IMP-47B u11 — Step 12 AI repair outcome (u8 surfacing). null when
|
||||
* step20 omits the field (legacy runs / pipeline aborted before Step 12). */
|
||||
ai_repair_status: AiRepairStatus | null;
|
||||
}
|
||||
|
||||
/**
|
||||
* IMP-47B u11 — Build the human_review notification text when Step 12 AI repair
|
||||
* reports a failure axis. Returns null when no notification is needed (success,
|
||||
* no AI invocation, or human_review_required=false). Pure function — no DOM, no
|
||||
* toast side-effect — so it can be unit-tested without React Testing Library.
|
||||
*
|
||||
* Failure axes mapped to user-facing text (verbatim policy from
|
||||
* IMP-47B #76 guardrail: "AI 호출 실패 / proposal validation 실패 / coverage 미달
|
||||
* → frontend 에 명확한 notification").
|
||||
*/
|
||||
export function formatAiRepairHumanReviewMessage(
|
||||
ai: AiRepairStatus | null | undefined,
|
||||
): string | null {
|
||||
if (!ai || !ai.human_review_required) return null;
|
||||
if (ai.status === "error") {
|
||||
const n = ai.counts?.error ?? ai.error_records?.length ?? 0;
|
||||
return `AI 재구성 호출 실패 (${n}건) — 다른 frame 선택 또는 수동 편집 필요`;
|
||||
}
|
||||
if (ai.status === "coverage_violated") {
|
||||
const dropped = (ai.dropped_section_ids || []).join(", ");
|
||||
return `AI 재구성 후 콘텐츠 누락 (dropped: ${dropped || "?"}) — 다른 frame 선택 또는 수동 편집 필요`;
|
||||
}
|
||||
if (ai.status === "unsupported_kind") {
|
||||
const n = ai.counts?.unsupported_kind ?? ai.unsupported_kind_records?.length ?? 0;
|
||||
return `AI 제안 형식 미지원 (${n}건) — 다른 frame 선택 또는 수동 편집 필요`;
|
||||
}
|
||||
return `AI 재구성 human_review 필요 (status: ${ai.status})`;
|
||||
}
|
||||
|
||||
export interface LoadRunResult {
|
||||
@@ -428,6 +489,7 @@ export async function loadRun(runId: string): Promise<LoadRunResult> {
|
||||
z.display_strategy_candidates ?? [],
|
||||
])
|
||||
),
|
||||
ai_repair_status: (slideStatus.data?.ai_repair_status ?? null) as AiRepairStatus | null,
|
||||
};
|
||||
|
||||
// ── NormalizedContent ──
|
||||
@@ -503,17 +565,51 @@ export async function loadRun(runId: string): Promise<LoadRunResult> {
|
||||
restructure: 2,
|
||||
reject: 3,
|
||||
};
|
||||
const rawSource = (unit.v4_all_judgments?.length > 0)
|
||||
? unit.v4_all_judgments
|
||||
: (unit.v4_candidates ?? []);
|
||||
// IMP-29 u2 — source priority (deterministic, no LLM):
|
||||
// 1) unit.candidate_evidence (IMP-05 L2 canonical, 14 fields per entry)
|
||||
// 2) unit.v4_all_judgments (pre-IMP-05 audit array)
|
||||
// 3) unit.v4_candidates (legacy minimal)
|
||||
// fallback_chain alias is intentionally NOT read (Stage 2 guardrail).
|
||||
const candidateEvidence = Array.isArray(unit.candidate_evidence)
|
||||
? unit.candidate_evidence
|
||||
: [];
|
||||
const candidateMap = new Map<string, any>();
|
||||
const pushCandidate = (c: any) => {
|
||||
if (!c) return;
|
||||
const key = c.template_id ?? c.id ?? c.frame_id;
|
||||
if (!key) return;
|
||||
if (!candidateMap.has(key)) candidateMap.set(key, c);
|
||||
};
|
||||
candidateEvidence.forEach(pushCandidate);
|
||||
(unit.v4_all_judgments ?? []).forEach(pushCandidate);
|
||||
(unit.v4_candidates ?? []).forEach(pushCandidate);
|
||||
const rawSource = Array.from(candidateMap.values());
|
||||
const v4Source = [...rawSource].sort((a: any, b: any) => {
|
||||
const lp = (LABEL_PRIORITY[a.label] ?? 99) - (LABEL_PRIORITY[b.label] ?? 99);
|
||||
if (lp !== 0) return lp;
|
||||
return (b.confidence ?? 0) - (a.confidence ?? 0);
|
||||
});
|
||||
// ─── IMP-41 u2 — application_candidates enrichment (issue #70) ───────────
|
||||
// Backend Step 9 emits `unit.application_candidates[]` (src/phase_z2_pipeline.py
|
||||
// _application_candidates_for_unit, :3071-3092) one entry per v4 candidate with
|
||||
// application_mode / auto_applicable / delegated_to derived from
|
||||
// APPLICATION_MODE_BY_V4_LABEL (:107-112). Enrichment ONLY — does NOT alter
|
||||
// candidate source priority, sorting, or TOP_N_FRAMES slicing.
|
||||
const applicationCandidates: any[] = Array.isArray(unit.application_candidates)
|
||||
? unit.application_candidates
|
||||
: [];
|
||||
const applicationModeMap = new Map<string, any>();
|
||||
applicationCandidates.forEach((ac: any) => {
|
||||
const key = ac?.template_id;
|
||||
if (typeof key === "string" && key.length > 0) {
|
||||
applicationModeMap.set(key, ac);
|
||||
}
|
||||
});
|
||||
const frameCandidates: FrameCandidate[] = v4Source
|
||||
.slice(0, TOP_N_FRAMES)
|
||||
.map((c: any) => ({
|
||||
.map((c: any) => {
|
||||
const appMatch = applicationModeMap.get(c.template_id);
|
||||
return ({
|
||||
id: c.template_id,
|
||||
name: c.template_id,
|
||||
score: c.confidence ?? 0,
|
||||
@@ -525,13 +621,33 @@ export async function loadRun(runId: string): Promise<LoadRunResult> {
|
||||
? `/frame-preview/${String(c.frame_number).padStart(2, "0")}`
|
||||
: undefined,
|
||||
// backend step09 의 catalog_registered (frame_contracts.yaml 등록 여부).
|
||||
// v4_all_judgments 에만 있음. v4_candidates fallback 시 undefined.
|
||||
// candidate_evidence 및 v4_all_judgments 에 있음. v4_candidates fallback 시 undefined.
|
||||
catalogRegistered: c.catalog_registered,
|
||||
// backend step09 의 min_height_px (frame_contracts.yaml visual_hints.min_height_px).
|
||||
// logical 1280x720 px 좌표계. contract 미등록 또는 visual_hints 부재 시 undefined.
|
||||
// v4_all_judgments 에만 있음. v4_candidates fallback 시 undefined (graceful).
|
||||
// v4_all_judgments 에만 있음. candidate_evidence / v4_candidates fallback 시 undefined (graceful).
|
||||
minHeightPx: c.min_height_px ?? undefined,
|
||||
}));
|
||||
// ─── IMP-05 L2 candidate_evidence fields (IMP-29 u2) ─────────────────
|
||||
// Populated when source = unit.candidate_evidence; otherwise silently
|
||||
// undefined for legacy fixtures (pre-IMP-05 fallback path).
|
||||
rank: c.rank,
|
||||
frameId: c.frame_id,
|
||||
v4Label: c.v4_label,
|
||||
phaseZStatus: c.phase_z_status,
|
||||
filteredForDirectExecution: c.filtered_for_direct_execution,
|
||||
routeHint: c.route_hint,
|
||||
decision: c.decision,
|
||||
reason: c.reason,
|
||||
capacityFit: c.capacity_fit,
|
||||
// ─── IMP-41 u2 — application_mode forwarding (issue #70) ───────────
|
||||
// Source = unit.application_candidates[] indexed by template_id above.
|
||||
// Optional fields — undefined when no matching application_candidate
|
||||
// (legacy fixtures pre-IMP-32 or candidates filtered out at Step 9).
|
||||
applicationMode: appMatch?.application_mode,
|
||||
autoApplicable: appMatch?.auto_applicable,
|
||||
delegatedTo: appMatch?.delegated_to ?? null,
|
||||
});
|
||||
});
|
||||
|
||||
const displayStrategy = (
|
||||
runMeta.display_strategy_candidates_by_zone[posEntry.name]?.[0] ??
|
||||
|
||||
@@ -116,6 +116,23 @@ export interface InternalRegion {
|
||||
frame_candidates: FrameCandidate[];
|
||||
}
|
||||
|
||||
/** IMP-05 L2 candidate_evidence.capacity_fit — backend capacity vs. content shape audit.
|
||||
* Source = src/phase_z2_pipeline.py compute_capacity_fit(). All fields optional —
|
||||
* frontend tolerates absence for pre-IMP-05 fixtures and contract-less templates. */
|
||||
export interface CapacityFitEvidence {
|
||||
item_count?: number | null;
|
||||
source_shape?: string | null;
|
||||
capacity?: {
|
||||
strict?: number | null;
|
||||
min?: number | null;
|
||||
max?: number | null;
|
||||
truncate_at?: number | null;
|
||||
pad_to?: number | null;
|
||||
} | null;
|
||||
fit_status?: string | null;
|
||||
mismatch_reason?: string | null;
|
||||
}
|
||||
|
||||
/** 프레임 후보 (V4 매칭 결과) */
|
||||
export interface FrameCandidate {
|
||||
id: string;
|
||||
@@ -131,6 +148,46 @@ export interface FrameCandidate {
|
||||
* Source = templates/phase_z2/catalog/frame_contracts.yaml visual_hints.min_height_px.
|
||||
* Undefined when contract unregistered or visual_hints absent (frontend tolerates undefined). */
|
||||
minHeightPx?: number;
|
||||
|
||||
// ─── IMP-05 L2 candidate_evidence fields (IMP-29 u1) ───────────────────────
|
||||
// Source = src/phase_z2_pipeline.py lookup_v4_match_with_fallback() candidate_trace.
|
||||
// All fields optional — pre-IMP-05 fixtures fall back to v4_all_judgments/v4_candidates
|
||||
// (deterministic, no LLM) and silently leave these undefined.
|
||||
|
||||
/** Candidate rank in V4 chain (1-based; 1 = primary). */
|
||||
rank?: number;
|
||||
/** Figma frame node id (backend `frame_id`). Distinct from `id` (= template_id). */
|
||||
frameId?: string;
|
||||
/** Alias of `label`. Kept separate for Codex IMP-05 L2 schema parity. */
|
||||
v4Label?: 'use_as_is' | 'light_edit' | 'restructure' | 'reject';
|
||||
/** Phase Z status enum (e.g. "auto_renderable", "fallback_candidate"). Open vocabulary. */
|
||||
phaseZStatus?: string;
|
||||
/** True when status is outside MVP1_ALLOWED_STATUSES (= excluded from direct render path). */
|
||||
filteredForDirectExecution?: boolean;
|
||||
/** Execution route mapped from `label` (direct_render / deterministic_minor_adjustment /
|
||||
* ai_adaptation_required / design_reference_only). Null on unknown labels. */
|
||||
routeHint?: 'direct_render' | 'deterministic_minor_adjustment' | 'ai_adaptation_required' | 'design_reference_only' | null;
|
||||
/** Selection outcome ("selected" or "skipped"). */
|
||||
decision?: 'selected' | 'skipped';
|
||||
/** Human-readable rationale (e.g. "primary_selected", "fallback_selected",
|
||||
* "duplicate_template_id", "skipped_no_contract", "capacity_mismatch:...",
|
||||
* "phase_z_status_not_allowed:..."). */
|
||||
reason?: string | null;
|
||||
/** Capacity vs. content shape audit (compute_capacity_fit output). */
|
||||
capacityFit?: CapacityFitEvidence | null;
|
||||
|
||||
// ─── IMP-41 application_mode forwarding (issue #70 u1) ─────────────────────
|
||||
// Source = src/phase_z2_pipeline.py APPLICATION_MODE_BY_V4_LABEL (:107-112),
|
||||
// emitted by _application_candidates_for_unit() into Step 9
|
||||
// unit.application_candidates[]. Optional — legacy fixtures pre-IMP-32 omit
|
||||
// these and the FramePanel tooltip falls back to the raw V4 label.
|
||||
|
||||
/** Application mode mapped from V4 label by backend (authoritative). */
|
||||
applicationMode?: 'direct_insert' | 'same_frame_with_adjustment' | 'layout_or_region_change' | 'exclude';
|
||||
/** True when backend marks the candidate as automatically applicable. */
|
||||
autoApplicable?: boolean;
|
||||
/** Delegation target step / actor (e.g. "step10_contract_check", "human_review"). */
|
||||
delegatedTo?: string | null;
|
||||
}
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
@@ -346,8 +346,10 @@ function vitePluginPhaseZApi(): Plugin {
|
||||
const pythonExe = process.platform === "win32" ? "python.exe" : "python";
|
||||
// 2026-05-14 — env toggle forward (보고용 일회성).
|
||||
// PHASE_Z_ALLOW_RESTRUCTURE / PHASE_Z_ALLOW_REJECT : status 통과
|
||||
// PHASE_Z_MAX_RANK=32 : V4 fallback chain 의 max_rank 확대 (등록 frame 까지 검색)
|
||||
// 04-1 (all reject) / 05-2 (rank 1~3 미등록) 등 자동 매칭 가능.
|
||||
// 2026-05-21 — IMP-38 retire PHASE_Z_MAX_RANK env (never read by backend).
|
||||
// v4 fallback chain max_rank 는 templates/phase_z2/catalog/v4_fallback_policy.yaml 의
|
||||
// 정식 정책 (dynamic_usable_count_based) 으로 결정 — backend src/phase_z2_pipeline.py
|
||||
// 의 lookup_v4_match_with_fallback() 가 load_v4_fallback_policy() 로 적용.
|
||||
const proc = spawn(pythonExe, cliArgs, {
|
||||
cwd: DESIGN_AGENT_ROOT,
|
||||
shell: false,
|
||||
@@ -355,7 +357,6 @@ function vitePluginPhaseZApi(): Plugin {
|
||||
...process.env,
|
||||
PHASE_Z_ALLOW_RESTRUCTURE: "1",
|
||||
PHASE_Z_ALLOW_REJECT: "1",
|
||||
PHASE_Z_MAX_RANK: "32",
|
||||
},
|
||||
});
|
||||
|
||||
|
||||
@@ -0,0 +1,135 @@
|
||||
# Dormant trigger registry (L3 layer — machine-readable).
|
||||
#
|
||||
# Purpose :
|
||||
# Closed-but-binding dormant backlog rows ("documented:dormant" /
|
||||
# "documented (deferred)") carry implicit "trigger-on-X" contracts.
|
||||
# L1 (human memory) + L2 (periodic INTEGRATION-AUDIT) are fragile / late.
|
||||
# This file is the single source of truth that scripts/check_dormant_triggers.py
|
||||
# reads to flag activation candidates on every orchestrator run.
|
||||
#
|
||||
# Schema (per entry) :
|
||||
# - issue : int # closed Gitea issue id (the dormant axis)
|
||||
# - title : string
|
||||
# - doc : string # repo-relative path to the dormant reference doc
|
||||
# - doc_evidence_lines : string # "start-end" line range citing the activation-gate text
|
||||
# - status : enum # documented:dormant | documented:deferred | documented:no-runtime | followup-linked
|
||||
# - followup_issue : int|null # set when an open issue already tracks the watch (then no checker watch needed)
|
||||
# - trigger
|
||||
# description : string
|
||||
# file_patterns : [glob] # working-tree paths checked against changed files
|
||||
# content_patterns : [regex] # python re patterns matched against changed-file contents
|
||||
# manual_evidence_required : bool # true → checker skips (human-only gate; e.g. User GO, sign-off, runtime regression analysis)
|
||||
# - on_trigger
|
||||
# action : enum # create_runtime_issue | reactivate_dormant | manual_review | note_only
|
||||
# template : string # suggested follow-up issue title (if action ≠ note_only)
|
||||
#
|
||||
# Guardrails :
|
||||
# - Checker is informational only (exit 0 always; orchestrator never blocks Stage 5 on alerts).
|
||||
# - manual_evidence_required: true entries do NOT auto-fire — they are noted for human review.
|
||||
# - followup_issue is set: the registry entry is note-only; no checker watch (the open issue tracks the axis).
|
||||
# - Out of scope for this registry : IMP-07 (documented:no-runtime — policy decline, reactivation = policy reopen, not a code trigger).
|
||||
|
||||
- issue: 16
|
||||
title: "IMP-16 U2 wiring (Phase Q U1 → Phase Z runtime)"
|
||||
doc: docs/architecture/IMP-16-U2-WIRING-DESIGN.md
|
||||
doc_evidence_lines: "21-25"
|
||||
status: documented:dormant
|
||||
followup_issue: null
|
||||
trigger:
|
||||
description: >-
|
||||
IMP-07 reverse-path actually lands runtime — a non-test module under src/
|
||||
introduces the reverse-path adapter (html_to_slide_mdx / edited_html_to_mdx /
|
||||
reverse_path). At that point IMP-16 U2 wiring (Step 1/2/14 surface use)
|
||||
becomes a live integration axis, not a paper design.
|
||||
file_patterns:
|
||||
- "src/**/*.py"
|
||||
content_patterns:
|
||||
- "html_to_slide_mdx"
|
||||
- "edited_html_to_mdx"
|
||||
- "reverse_path"
|
||||
manual_evidence_required: false
|
||||
on_trigger:
|
||||
action: create_runtime_issue
|
||||
template: "[IMP-16][P5][WIRING] Activate U2 reverse-path wiring against new IMP-07 adapter"
|
||||
|
||||
- issue: 17
|
||||
title: "IMP-17 AI repair fallback carve-out"
|
||||
doc: docs/architecture/IMP-17-CARVE-OUT.md
|
||||
doc_evidence_lines: "25-31"
|
||||
status: documented:dormant
|
||||
followup_issue: null
|
||||
trigger:
|
||||
description: >-
|
||||
3-condition AND gate: (1) explicit User GO for axis activation,
|
||||
(2) B4 frame_selection evidence integration complete (Step 9 evidence trace
|
||||
stabilised), (3) IMP-04 (catalog expansion to 32 frames) + IMP-05 (V4
|
||||
rank-2/3 fallback) live. All three required before the carve-out exits
|
||||
design-only state.
|
||||
file_patterns: []
|
||||
content_patterns: []
|
||||
manual_evidence_required: true
|
||||
on_trigger:
|
||||
action: manual_review
|
||||
template: "[IMP-17][P5][CARVE-OUT] Activate ai_adaptation_required fallback (3-cond gate cleared)"
|
||||
|
||||
- issue: 18
|
||||
title: "IMP-18 SVG coordinate pipeline gap report"
|
||||
doc: docs/architecture/IMP-18-SVG-GAP-REPORT.md
|
||||
doc_evidence_lines: "38-43"
|
||||
status: documented:dormant
|
||||
followup_issue: null
|
||||
trigger:
|
||||
description: >-
|
||||
An SVG-bearing partial lands under templates/phase_z2/ (families or frames)
|
||||
AND the partial declares slots consuming items[*].cx/cy/r + outer_r +
|
||||
viewbox_* (the prepare_venn_data return contract). IMP-04 frame_partials
|
||||
registration is the natural upstream.
|
||||
file_patterns:
|
||||
- "templates/phase_z2/families/*.html"
|
||||
- "templates/phase_z2/frames/*.html"
|
||||
content_patterns:
|
||||
- "<svg"
|
||||
- "viewBox"
|
||||
manual_evidence_required: false
|
||||
on_trigger:
|
||||
action: create_runtime_issue
|
||||
template: "[IMP-18][P5][SVG] Activate SVG coordinate pipeline for new partial"
|
||||
|
||||
- issue: 19
|
||||
title: "IMP-19 zone ratio reference (Phase O role-container pattern)"
|
||||
doc: docs/architecture/IMP-19-ZONE-RATIO-REFERENCE.md
|
||||
doc_evidence_lines: "83-90"
|
||||
status: documented:dormant
|
||||
followup_issue: null
|
||||
trigger:
|
||||
description: >-
|
||||
Phase Z Step 8 solver (min_height_first + content_weight) produces a
|
||||
verifiable regression that the Phase O role-container pattern would have
|
||||
handled correctly, AND the IMP-09 owner confirms the case is not
|
||||
addressable inside the Phase Z solver (visual_hints.min_height_px /
|
||||
content_weight.score adjustments insufficient). Requires failing-case MDX
|
||||
+ frame_contract trace + observed vs expected geometry.
|
||||
file_patterns: []
|
||||
content_patterns: []
|
||||
manual_evidence_required: true
|
||||
on_trigger:
|
||||
action: manual_review
|
||||
template: "[IMP-19][P5][ZONE-RATIO] Re-activate Phase O role-container pattern (IMP-09 sign-off attached)"
|
||||
|
||||
- issue: 20
|
||||
title: "IMP-20 frame contract validation reference"
|
||||
doc: docs/architecture/IMP-20-FRAME-CONTRACT-VALIDATION-REFERENCE.md
|
||||
doc_evidence_lines: "85-91"
|
||||
status: followup-linked
|
||||
followup_issue: 55
|
||||
trigger:
|
||||
description: >-
|
||||
§A5 3-cond AND gate (Step 10 partial frame-contract emit insufficient +
|
||||
evidence + IMP-04 sign-off). Watch surface already owned by open issue
|
||||
#55 — no checker watch installed here to avoid double-tracking.
|
||||
file_patterns: []
|
||||
content_patterns: []
|
||||
manual_evidence_required: true
|
||||
on_trigger:
|
||||
action: note_only
|
||||
template: "Tracked under open issue #55 — no new watch needed."
|
||||
@@ -1,4 +1,13 @@
|
||||
# IMP-16-U2 — Phase Z verification wiring design (design-only)
|
||||
> **⚠️ STATUS UPDATE (2026-05-20, INTEGRATION-AUDIT-02)** — IMP-16 is reclassified
|
||||
> as `documented:dormant` and IMP-07 as `documented:no-runtime`. The 3 "Open items
|
||||
> deferred until IMP-07 lands" below remain dormant until IMP-07 reverse-path
|
||||
> actually lands runtime (no current plan).
|
||||
>
|
||||
> Resolution evidence: see `INTEGRATION-AUDIT-02-REPORT.md` Sections 3, 4, and 7
|
||||
> (final decision: `NEEDS_DOC_SYNC_FOLLOWUP`).
|
||||
>
|
||||
> Do NOT treat this design contract as actionable in current Phase Z runtime.
|
||||
|
||||
**Status**: design-only contract. **No runtime wiring lands in this issue.** All wiring is gated behind IMP-07 reverse-path activation (B-2 main). When IMP-07 lands, this doc becomes the binding contract for the Step 1 / 2 / 14 / 21 / 22 changes that consume the IMP-16-U1 surface in `src/phase_z2_verification_utils.py`.
|
||||
|
||||
|
||||
@@ -1,13 +1,13 @@
|
||||
# IMP-17 — AI repair fallback infrastructure (carve-out)
|
||||
|
||||
**Status**: carve-out, **design-only**. Normal-path AI calls = 0. No runtime fallback code lands until the activation gate clears.
|
||||
**Status**: carve-out infra **scaffolded under IMP-33** (issue #61, Stage 3 u1~u11). Normal-path AI calls = 0 (PZ-1) — `ai_fallback_enabled` flag default `False` in `src/config.py`. Runtime AI is reachable only via fallback path entry points; Step 12 entry is provisional-gated, Step 17 entry is structurally blocked behind IMP-34 + IMP-35.
|
||||
|
||||
**Source anchors**
|
||||
- IMP-17 backlog row — [`PHASE-Z-IMPLEMENTATION-ISSUE-BACKLOG.md`](PHASE-Z-IMPLEMENTATION-ISSUE-BACKLOG.md):68 (carve-out — normal path 밖, soft link IMP-04 + IMP-05).
|
||||
- INSIGHT-MAP §3 — [`PHASE-Q-INSIGHT-TO-22STEP-MAP.md`](PHASE-Q-INSIGHT-TO-22STEP-MAP.md) (G3 AI repair fallback infra registry row, normal path = no).
|
||||
- 22-step pipeline — [`PHASE-Z-PIPELINE-OVERVIEW.md`](PHASE-Z-PIPELINE-OVERVIEW.md) Step 12 (lines 280-287), Step 16 (lines 318-325), Step 17 (lines 326-333).
|
||||
- Pattern shape reference (Phase Q Archive — link-only, do not port) — `src/content_editor.py:21,318` (httpx + retry shape, imports `sse_utils`) + `src/sse_utils.py:16-50` (SSE token parser).
|
||||
- Route hint emission site — `src/phase_z2_pipeline.py:564` (`restructure` → `ai_adaptation_required` route hint; deterministic emission today, AI consumer deferred to IMP-17).
|
||||
- Route hint surface (current anchors) — `src/phase_z2_pipeline.py:570` (conceptual comment), `:572` (`_IMP05_ROUTE_HINTS` table), `:575` (`restructure` → `ai_adaptation_required`), `:580` (`_imp05_route_hint`), `:664` (candidate_evidence emission). Deterministic emission today; AI consumer deferred to IMP-17 (this carve-out). Anchor pin: `tests/orchestrator_unit/test_imp17_comment_anchor.py`.
|
||||
|
||||
## Carve-out boundary
|
||||
|
||||
@@ -42,3 +42,14 @@ Phase Q `content_editor.py` 는 **Archive Candidate** ([`PHASE-Q-AUDIT.md`](PHAS
|
||||
- AI 호출은 normal path 에 없다 (Phase Z 원칙, [memory `feedback_ai_isolation_contract`](../../README.md)).
|
||||
- 출력 단위는 항상 content_object / Internal Region / Frame Slot 또는 restructuring proposal — HTML 구조 / 레이아웃 / 프리셋 결정 X.
|
||||
- Phase Q 자산 (Kei persona prompts, Kei-API endpoint, persona retry semantics) 과 단절. Phase Z 의 fallback runtime 은 별도 prompt / endpoint 설계로 출발한다 (본 carve-out 활성 시).
|
||||
|
||||
## Runtime module surface (IMP-33 u1~u11 binding)
|
||||
|
||||
| Axis | Binding |
|
||||
|---|---|
|
||||
| Module path | `src/phase_z2_ai_fallback/` (locked by [`IMP-31-GATE-AUDIT.md`](IMP-31-GATE-AUDIT.md):31,50,56). |
|
||||
| Step 12 entry | `src.phase_z2_ai_fallback.step12.gather_step12_ai_repair_proposals` — IMP-30 provisional gate (`not_provisional` skip) AND reject gate (`design_reference_only_no_ai` skip) AND non-AI route catch-all run BEFORE `route_ai_fallback`. |
|
||||
| Step 17 entry | `src.phase_z2_ai_fallback.step17.gather_step17_ai_repair_proposals` — STRUCTURALLY BLOCKED. Every unit returns `skip_reason="step17_ai_blocked_imp_34_35_prerequisites_missing"`. Module does NOT import `route_ai_fallback` / `AiFallbackClient` / `anthropic`. |
|
||||
| Cascade order | `src.phase_z2_ai_fallback.step17.OVERFLOW_CASCADE_ORDER = (DETERMINISTIC, POPUP, AI_REPAIR, USER_OVERRIDE)` — single source of truth for Step 17 consumers. Aligns with line 16 of this doc. |
|
||||
| IMP-46 cache gate | `src.phase_z2_ai_fallback.cache.save_proposal(..., visual_check_passed, user_approved, auto_cache=False)` raises `AiFallbackCacheGateError` unless `visual_check_passed=True` AND (`user_approved=True` OR `auto_cache=True`). Persistent JSON backend at `data/frame_cache/{frame_id}/{signature_hash}.json` (u2); cache key = structural signature over 8 axes (u1+u4); read-side fingerprint invalidation via `read_proposal(..., fingerprints=...)` strict equality (u3); `--auto-cache` CLI flag + `settings.ai_fallback_auto_cache` (default `False`) bypasses ONLY the `user_approved` gate (u5); repo root tracked via `data/frame_cache/.gitkeep` with cached payloads git-ignored (u6). `read_proposal` returns `None` on missing / corrupt / fingerprint-mismatched entries — cache is a hint, never a hard dependency. |
|
||||
| AST isolation | `tests/phase_z2_ai_fallback/test_ast_isolation.py` parses every `*.py` under `src/phase_z2_ai_fallback/` and forbids Phase Q runtime / Kei client / `src.phase_z2_*` (non-fallback) imports. Whitelist = `src.config` + intra-package + stdlib + `anthropic` + `pydantic`. |
|
||||
|
||||
@@ -25,9 +25,9 @@ Phase R' implements SVG coordinate pre-compute as a renderer hook. References (d
|
||||
|
||||
Phase Z active partials surface:
|
||||
|
||||
- `templates/phase_z2/families/*.html` — **13** files.
|
||||
- `templates/phase_z2/families/*.html` — **11 contracted + 2 WIP untracked = 13 on disk** (contracted set = `templates/phase_z2/catalog/frame_contracts.yaml` top-level keys; WIP allowlist = [`templates/phase_z2/families/_WIP_FILES.md`](../../templates/phase_z2/families/_WIP_FILES.md), gated on Gitea #42 / #52 F-2 option (c)).
|
||||
- `templates/phase_z2/frames/*.html` — **2** files.
|
||||
- Total surface = **15 partials**.
|
||||
- Total surface = **13 active partials (11 contracted families + 2 frames) + 2 WIP untracked families** (15 on disk; runtime matcher consumes the contracted set only).
|
||||
|
||||
SVG usage scan (evidence): `rg "<svg|viewBox" templates/phase_z2/` → **0 matches** (exit 1).
|
||||
|
||||
@@ -48,7 +48,7 @@ Per `CLAUDE.md` Phase R' regression prevention rules and the Stage 1/2 exit repo
|
||||
|
||||
- `src/renderer.py` — read-only. No edit to `_preprocess_svg_data` body, `SVG_BLOCKS` set, or `render_multi_page` call site.
|
||||
- `src/svg_calculator.py` — read-only. No edit to the five helpers or their public signatures.
|
||||
- `templates/phase_z2/families/*.html` (13) + `templates/phase_z2/frames/*.html` (2) — no `<svg>` / `viewBox` insertion in IMP-18 scope. SVG-bearing partial onboarding is owned by IMP-04.
|
||||
- `templates/phase_z2/families/*.html` (11 contracted + 2 WIP untracked = 13 on disk; WIP set = [`_WIP_FILES.md`](../../templates/phase_z2/families/_WIP_FILES.md)) + `templates/phase_z2/frames/*.html` (2) — no `<svg>` / `viewBox` insertion in IMP-18 scope. SVG-bearing partial onboarding is owned by IMP-04. The 2 WIP family templates are gated on Gitea #42 (promote-or-remove) and remain outside the runtime matcher set per #52 F-2 option (c).
|
||||
- F12 `construction_goals_three_circle_intersection.html` HTML/CSS → SVG migration is **out of scope** (separate post-IMP-04 issue).
|
||||
- No hardcoded SVG coordinates in Phase Z templates — when IMP-18 re-activates, coordinates must be derived from `svg_calculator` helpers (or equivalent forward-port into `phase_z2_renderer`), not hand-copied.
|
||||
|
||||
|
||||
@@ -0,0 +1,97 @@
|
||||
# IMP-19 — Phase O/Q Zone Ratio Container Pattern Reference
|
||||
|
||||
**Status**: documented (reference-only, dormant)
|
||||
**Scope**: doc-only. No runtime surface modified.
|
||||
**Related issue**: https://gitea.hmac.kr/Kyeongmin/C.E.L_Slide_test2/issues/19
|
||||
**Soft dependency**: IMP-09 (Phase Z Step 8 zone-ratio solver) — IMP-19 stays dormant; activates only via the A5 gate.
|
||||
**Source axis**: INSIGHT-MAP §3 / §2.8 I4 — `renderer._group_blocks_by_area` pattern reference.
|
||||
|
||||
---
|
||||
|
||||
## A1 — Phase O/Q consumer pattern (read-only reference)
|
||||
|
||||
Phase O/Q implements role-based block grouping inside body-side zones at the renderer layer. References (do **not** modify):
|
||||
|
||||
- `src/renderer.py:210-295` — `_group_blocks_by_area(blocks, container_specs=None)` — `OrderedDict` grouping by `block["area"]`; when `container_specs` is supplied and `area ∈ {"body","left","right","hero","detail"}` enters the role-container branch (L230).
|
||||
- `src/renderer.py:234` — hardcoded `role_order = ["배경", "본심"]` — two-role role-loop axis (block-level container, **not** zone geometry).
|
||||
- `src/renderer.py:240-253` — topic_id-first match against `spec.topic_ids`, then fallback positional fill when topic_id match yields empty (L248-253).
|
||||
- `src/renderer.py:261-274` — inline-style injection: `height:{spec.height_px}px; overflow:visible; display:flex; flex-direction:column; gap:8px; font-size:{font_size}px; --spacing-inner:{padding}px; --font-body:{font_size/16}rem;`. `font_size` / `padding` are read at `:262-263` via `spec.block_constraints.get("font_size_px", 15.2)` / `.get("padding_px", 20)` — **renderer-side defaults**, not producer-emitted (see A2).
|
||||
- `src/renderer.py:277-279` — leftover (unassigned) blocks appended after role containers.
|
||||
- `src/renderer.py:283-291` — non-container branch: `len(block_list)==1` → single html, else `flex-direction:column` wrapper with `gap:var(--spacing-block); height:100%`.
|
||||
|
||||
Call sites:
|
||||
|
||||
- `src/renderer.py:352-353` — `render_multi_page()` — passes `layout_concept.get("_container_specs")` as `container_specs` argument (Phase O activation path).
|
||||
- `src/renderer.py:426` — `render_slide()` — invokes `_group_blocks_by_area(blocks_raw)` with **no** `container_specs` (legacy fallback / unit-test path).
|
||||
|
||||
Classification: block/role-level container injection at render time. **Not** Phase Z zone geometry.
|
||||
|
||||
## A2 — Phase O upstream producer (read-only reference)
|
||||
|
||||
`ContainerSpec` payloads consumed by A1 are produced upstream. References (do **not** modify):
|
||||
|
||||
- `src/space_allocator.py:445-586` — `build_containers_type_b(page_structure, slide_width=1280, slide_height=720, image_sizes=None)` — Phase X-B 유형 B container builder.
|
||||
- `src/space_allocator.py:462-468` — token load (`_load_design_tokens`) + `pad`, `header_h`, `gap_block`, `gap_small`, `inner_w` derivation.
|
||||
- `src/space_allocator.py:470-484` — role classification into `top_roles` / `bottom_roles` / `footer_role` by `info["zone"] ∈ {"top","bottom","bottom_left","bottom_right","footer"}`.
|
||||
- `src/space_allocator.py:486-503` — usable height calculation against `slide_body_top=65` + `slide_body_h=590` with optional `footer_role` carve-out.
|
||||
- `src/space_allocator.py:505-510` — `zone_overhead = zone_count * zone_title_h(28) + (zone_count-1) * zone_gap(16)`.
|
||||
- `src/space_allocator.py:512-520` — `top_h` / `bottom_h` split by `weight` ratio over `usable_h`.
|
||||
- `src/space_allocator.py:522-537` — image-aware top-zone width split (`img_w = min(top_h*ratio, inner_w*0.45)`).
|
||||
- `src/space_allocator.py:541-556` — top-role `ContainerSpec` emission: `block_constraints = {"img_width_px": img_w, "img_height_px": top_h if img_w>0 else 0, "has_image": img_w>0}` — image-aware keys only.
|
||||
- `src/space_allocator.py:562-574` — bottom-role `ContainerSpec` emission: `block_constraints = {}` (empty; no producer keys).
|
||||
- `src/space_allocator.py:577-588` — footer-role `ContainerSpec` emission: `block_constraints = {}` (empty; `max_height_cost="low"` literal).
|
||||
|
||||
Producer classification: block-level role container with `height_px` + `width_px` + `block_constraints` containing **only** image-aware keys (`img_width_px`, `img_height_px`, `has_image`) on top role, **empty** on bottom/footer roles. **Not** zone-level ratio geometry. `font_size_px` / `padding_px` are **renderer-side defaults** (consumed via `.get(..., 15.2)` / `.get(..., 20)` at `src/renderer.py:262-263`), **not** producer output.
|
||||
|
||||
## A3 — Phase Z Step 8 solver delta (IMP-09 owned)
|
||||
|
||||
The active Phase Z zone-ratio solver lives in `src/phase_z2_pipeline.py` and is **IMP-09 owned**. IMP-19 does **not** absorb, replace, or amend this surface. References (do **not** modify):
|
||||
|
||||
- `src/phase_z2_pipeline.py:794-853` — `compute_zone_layout(zones_data, total_height=SLIDE_BODY_HEIGHT, gap=GRID_GAP)` — row-axis solver. Algorithm = `min_height_first + content_weight_distribution`: Step 1 reserves per-zone `min_height_px` from frame_contract `visual_hints` (with proportional scale-down on overflow), Step 2 distributes the remaining vertical budget by `content_weight.score`, Step 3 absorbs rounding residual into the last zone. Returns `heights_px` + `ratios` + reasoning trace.
|
||||
- `src/phase_z2_pipeline.py:924-972` — `compute_zone_layout_cols(zones_data, total_width=SLIDE_BODY_WIDTH, gap=GRID_GAP)` — col-axis solver. Algorithm = `content_weight_distribution_cols` (weight-only; no `min_width_px` contract exists in `frame_contracts.yaml` per IMP-09 verification). Zero-weight guard splits evenly across `n` zones. Returns `widths_px` + `width_ratios`.
|
||||
- `src/phase_z2_pipeline.py:1125-1452` — topology dispatch surface:
|
||||
- `:1125-1152` `_build_rows_dynamic` — `topology=="rows"` (horizontal-2): dynamic row heights via `compute_zone_layout`, static fr column widths via `_parse_fr_string`.
|
||||
- `:1155+` `_build_grid_dynamic_2d` — `topology ∈ {T, inverted-T, side-T-left, side-T-right, 2x2}`: per-row + per-col virtual-zone aggregation → row solver + col solver → `2d_dynamic_aggregated` computation.
|
||||
- `:1444-1452` dynamic-branch dispatcher: `rows` / `cols` / 2-D / default fr.
|
||||
- `:1380-1434` user-override geometry branch (`computation == "user_override_geometry"`) — preserves raw override percentages without invoking the weight solver.
|
||||
|
||||
Delta vs Phase O/Q (A1+A2):
|
||||
|
||||
| Axis | Phase O/Q (`renderer._group_blocks_by_area`) | Phase Z Step 8 (`compute_zone_layout` + cols) |
|
||||
|---|---|---|
|
||||
| Geometry level | block/role inside one zone | zone-level row/col tracks across slide_body |
|
||||
| Width source | role x-anchor + `top_h` image carve-out | content_weight share (cols) / fr-string (rows) |
|
||||
| Height source | producer `ContainerSpec.height_px` injection | min_height_first + content_weight remainder |
|
||||
| Role axis | hardcoded `["배경","본심"]` (L234) | no role concept — zone position + frame contract |
|
||||
| Min-height source | none (producer-emitted absolute px) | frame_contract `visual_hints.min_height_px` |
|
||||
| Topology dispatch | none (single role-loop) | rows / cols / T / inverted-T / side-T-* / 2x2 / single |
|
||||
| Inline-style injection | yes (height + font_size + spacing-inner) | no (geometry-only; styling handled downstream) |
|
||||
|
||||
Conclusion: Phase O role-container pattern and Phase Z zone-ratio solver operate at **different abstraction layers** (block-in-zone vs zone-in-slide). They are **not** drop-in interchangeable; IMP-19 surfaces this delta only for design-pattern comparison.
|
||||
|
||||
## A4 — IMP-09 boundary statement (soft-link)
|
||||
|
||||
IMP-19 is `soft link: IMP-09`. Ownership separation:
|
||||
|
||||
- **IMP-09 owns**: every algorithmic change to `compute_zone_layout`, `compute_zone_layout_cols`, the topology dispatch surface (`_build_rows_dynamic` / `_build_cols_dynamic` / `_build_grid_dynamic_2d` / `_build_fr_default`), and the frame_contract `visual_hints.min_height_px` contract.
|
||||
- **IMP-19 owns**: reference-only documentation of the Phase O/Q `_group_blocks_by_area` + `build_containers_type_b` pattern (A1 + A2) and the Phase Z solver delta narrative (A3).
|
||||
- **No bidirectional code flow**: IMP-19 does not move Phase O code into Phase Z, and IMP-09 does not consume Phase O `ContainerSpec` payloads. The two solvers remain isolated.
|
||||
- **Reference direction is one-way**: this document points read-only at `src/renderer.py`, `src/space_allocator.py`, and `src/phase_z2_pipeline.py`. No reverse pointer is required in those source files.
|
||||
|
||||
If IMP-09 alters the Phase Z solver signature, A3 must be re-verified (file:line refs); the boundary statement itself does not change.
|
||||
|
||||
## A5 — Re-activation gate + guardrails
|
||||
|
||||
IMP-19 is `documented` (dormant). Re-activation requires **all** of the following gate conditions:
|
||||
|
||||
1. **Trigger**: Phase Z Step 8 produces a verifiable case where the active solver (`min_height_first + content_weight`) yields geometry that the Phase O role-container pattern would have handled correctly — i.e., a regression that maps cleanly to the block-level role abstraction, not the zone-level abstraction.
|
||||
2. **Evidence requirement**: failing-case MDX + frame_contract trace + observed geometry vs expected geometry, attached to a new issue or this issue's reopened state.
|
||||
3. **IMP-09 sign-off**: the IMP-09 owner confirms the failing case is **not** addressable inside the Phase Z solver (e.g., adding `visual_hints.min_height_px` or adjusting `content_weight.score` does not resolve it).
|
||||
4. **Scope re-lock**: the new axis is scope-locked under a fresh implementation issue (not silently reopened in IMP-19) so the soft-link contract is preserved.
|
||||
|
||||
Guardrails (preserved from Stage 1 + Stage 2):
|
||||
|
||||
- **GR1 — No runtime integration**: this document does not authorize merging Phase O role-container code into the Phase Z runtime. Any such integration requires a new scope-locked issue with its own Stage 1/2 review.
|
||||
- **GR2 — Phase O no-regression**: Phase O containers (`render_multi_page` path with `_container_specs`) must not re-enter the Phase Z render path; the `render_slide` legacy fallback at `src/renderer.py:426` (no `container_specs`) remains the unit-test entry.
|
||||
- **GR3 — Reference extract stays in `docs/architecture/`**: never under `src/`. No code body copying; file:line refs only.
|
||||
- **GR4 — Soft-link integrity**: IMP-19 status remains `documented` until the A5 gate fires. The IMP-09 backlog entry carries a back-reference (see u3); IMP-19 carries the forward reference here.
|
||||
@@ -0,0 +1,109 @@
|
||||
# IMP-20 — Phase Q `content_verifier` Frame Contract Validation Pattern Reference
|
||||
|
||||
**Status**: documented (reference-only, dormant)
|
||||
**Scope**: doc-only. No runtime surface modified.
|
||||
**Related issue**: https://gitea.hmac.kr/Kyeongmin/C.E.L_Slide_test2/issues/20
|
||||
**Soft dependency**: IMP-04 (extended catalog application) — IMP-20 stays dormant; activates only via the A5 gate.
|
||||
**Source axis**: INSIGHT-MAP §3 / §2.7 H2 — `content_verifier.verify_structure` pattern reference.
|
||||
|
||||
---
|
||||
|
||||
## A1 — Phase Q consumer pattern (read-only reference)
|
||||
|
||||
Phase Q implements area-level required-pattern validation at the content-verifier layer. References (do **not** modify):
|
||||
|
||||
- `src/content_verifier.py:382-392` — `REQUIRED_PATTERNS: dict[str, list[str]]` — top-level pattern dictionary keyed by area name (`body_bg`, `body_core`, `sidebar`, `footer`). Values verified: `body_bg=[]`, `body_core=["key-msg"]`, `sidebar=["padding-left", "text-indent"]`, `footer=[]`. Phase T (`L379-381` comment) removed the `overflow:hidden` requirement to reconcile with the Phase T prompt's "overflow:hidden 금지" directive — that no-regression boundary is preserved.
|
||||
- `src/content_verifier.py:395-448` — `verify_structure(generated_html, area_name, has_image=False, font_hierarchy=None) → VerificationResult` — the substring-check + OR + tolerance core logic.
|
||||
- `:405-412` — substring presence loop. Each pattern string is split on `|` (`pattern.split("|")` at L410) and treated as an OR alternation: any alternative present passes the pattern. Missing alternatives are appended to a `missing` list.
|
||||
- `:414-416` — `has_image` branch. When `has_image=True` and `area_name == "body_core"`, an additional implicit requirement is enforced: `"slide-img-"` must appear in `generated_html`. Missing image marker is reported as `"slide-img-* (이미지 태그)"` in `missing`.
|
||||
- `:418-436` — `font_hierarchy` branch. When supplied, area-name → max-font lookup uses a fixed `role_font_map = {"body_bg":bg/11, "body_core":core/12, "sidebar":sidebar/10, "footer":core/12}`. HTML `font-size:\s*(\d+(?:\.\d+)?)\s*px` matches are extracted via regex (L430); each measured size > `max_font + 1` (1px tolerance at L433) emits a `font_warnings` entry. Warnings do **not** flip `passed`.
|
||||
- `:438-447` — result construction. `passed = (len(missing) == 0)`. `score = 1.0` on pass else `1.0 - len(missing) / max(1, len(patterns))` (continuous degradation; `max(1, …)` guards empty-pattern division by zero). Errors prefixed `"필수 패턴 누락: "`. Warnings carry font hierarchy violations only.
|
||||
- `src/content_verifier.py:455-487` — `verify_area(original_text, generated_html, area_name, has_image=False) → VerificationResult` — composes L1 (`verify_text_preservation`) + L2 (`verify_no_forbidden_content`) + L3 (`verify_structure`) at L462-466. `verify_structure` call at L465 passes `has_image` but **not** `font_hierarchy` (font_hierarchy is unused inside `verify_area`).
|
||||
- `src/content_verifier.py:490-529` — `verify_all_areas(generated, area_texts, has_image_areas=None)` — area dispatch fan-out. `body_html` is split into `body_bg` + `body_core` (L510-519); `body_core` is the **only** branch that propagates `has_image=("body_core" in has_image_areas)` to `verify_area` (L518). `sidebar_html` (L521-525) and `footer_html` (L527-531) call `verify_area` with default `has_image=False`.
|
||||
|
||||
Classification: area-level (Phase Q HTML area axis) required-pattern validation at content-verifier time. **Not** Phase Z frame_id × sub_zone contract validation.
|
||||
|
||||
## A2 — Phase Q `REQUIRED_PATTERNS` shape (read-only reference)
|
||||
|
||||
The Phase Q pattern-dict shape — **values are Phase Q-specific and excluded from reuse; only the shape is Phase Z design input.**
|
||||
|
||||
| Axis | Phase Q shape | Where observed |
|
||||
|---|---|---|
|
||||
| Key axis | area name (string) | `src/content_verifier.py:382` keys: `body_bg` / `body_core` / `sidebar` / `footer` |
|
||||
| Value type | `list[str]` of substring patterns | `src/content_verifier.py:383-391` |
|
||||
| Alternation semantics | `"a\|b"` → OR (any alt passes) via `pattern.split("|")` | `src/content_verifier.py:410` |
|
||||
| Image-conditional branch | `has_image=True` ∧ `area_name=="body_core"` → implicit `"slide-img-"` requirement | `src/content_verifier.py:414-416` |
|
||||
| Font hierarchy tolerance | 1px (`fs > max_font + 1`); area-name → max-font fixed lookup | `src/content_verifier.py:433`, `:421-426` |
|
||||
| Pass/score rule | `passed = (missing == [])`; score = continuous degradation `1.0 - len(missing)/max(1, len(patterns))` | `src/content_verifier.py:438`, `:445` |
|
||||
| Empty-pattern handling | `max(1, len(patterns))` guards divide-by-zero; empty pattern list always passes | `src/content_verifier.py:445`, `:382-383` (`body_bg=[]`) |
|
||||
|
||||
Shape-only carry-over candidates for Phase Z design (see A3 in u2):
|
||||
|
||||
- `dict[key]→list[pattern]` indirection.
|
||||
- OR via in-string `|` separator (low-ceremony alternation).
|
||||
- Conditional implicit requirement injected by external context flag (here `has_image`; in Phase Z potentially `accepted_content_types` per sub_zone).
|
||||
- Continuous score degradation rather than binary pass/fail (downstream consumers can threshold).
|
||||
- Separate `errors` (block) vs `warnings` (advisory) lanes — font hierarchy lives in warnings, not errors.
|
||||
|
||||
Values that **must not** carry into Phase Z: the literal strings `"key-msg"`, `"padding-left"`, `"text-indent"`, `"slide-img-"`, and the area names `body_bg` / `body_core` / `sidebar` / `footer` themselves — these are Phase Q area-HTML idioms, not Phase Z frame/slot idioms.
|
||||
|
||||
## A3 — Phase Z target pattern dict (design input, not yet active)
|
||||
|
||||
The Phase Z-native target axis = **frame_id × sub_zone** pattern dict, aligned with `templates/phase_z2/catalog/frame_contracts.yaml`. References (do **not** modify):
|
||||
|
||||
- `templates/phase_z2/catalog/frame_contracts.yaml:21` `three_parallel_requirements` (F13, 3 sub_zones), `:77` `process_product_two_way` (F29, 2 sub_zones × strict 3 cardinality), `:128` `bim_issues_quadrant_four` (F16, 4 sub_zones), `:189` `three_persona_benefits` (F14, 3 sub_zones), `:253` `construction_goals_three_circle_intersection` (F12, 3+1 sub_zones — `intersection` is `min:0,max:1`), `:323` `construction_bim_three_usage` (F11, 3 sub_zones), `:391` `bim_dx_comparison_table` (F18, 2 header + 1 `rows` with `min:1,max:12`), `:456` `dx_sw_necessity_three_perspectives` (F20, 3 sub_zones), `:520` `info_management_what_how_when` (F8, 3 sub_zones), `:580` `sw_reality_three_emphasis` (F28, 3 sub_zones), `:637` `bim_current_problems_paired` (F17, 8 sub_zones — row × side 2-axis).
|
||||
- All 11 contracts carry `accepted_content_types` + `sub_zones`; field `density_envelope` is absent across the catalog (verified `grep -c "density_envelope" templates/phase_z2/catalog/frame_contracts.yaml` = 0).
|
||||
- `src/phase_z2_mapper.py:49-57` `load_frame_contracts` / `get_contract` — direct dict lookup against the 11 entries above.
|
||||
- `src/phase_z2_pipeline.py:3776-3805` Step 10 emit — currently surfaces `frame_id` / `family` / `source_shape` / `cardinality` / `visual_hints` / `accepted_content_types` / `sub_zones` / `payload_builder` / `payload_builder_options` to `step10_frame_contract.json` with `step_status="partial"`. No pattern-dict assertion runs against this payload yet.
|
||||
|
||||
Abstraction-mismatch table (Phase Q area-level vs Phase Z frame/slot-level):
|
||||
|
||||
| Axis | Phase Q (A1+A2) | Phase Z target (A3) |
|
||||
|---|---|---|
|
||||
| Key | area name (`body_bg`/`body_core`/`sidebar`/`footer`) | `(frame_id, sub_zone_id)` tuple — e.g. `(1171281190, "pillar_1")` |
|
||||
| Cardinality of keys | 4 fixed area names | open over 11 contracts × N sub_zones (3+2+4+3+4+3+3+3+3+3+8 = 39 sub_zones in current catalog) |
|
||||
| Value semantics | substring presence (HTML-string match) | candidates: substring presence and/or contract-field assertion (`cardinality.strict` / `accepts` membership / `partial_target_path` resolution) |
|
||||
| Conditional branch input | `has_image` external flag | `accepted_content_types` per sub_zone (catalog-driven, not external flag) |
|
||||
| Tolerance | 1px on font-size (single axis) | candidates: font-size 1px tolerance carried over **or** replaced by `visual_hints.min_height_px` envelope check |
|
||||
| Validation timing | post-render HTML (`generated_html` string) | post Step 18 final.html (mirrors Phase Q timing) — Step 12 light_edit/restructure proposal is excluded (proposal is upstream of render) |
|
||||
| Result lanes | `errors` (block) + `warnings` (advisory) | preserved as-is from Phase Q shape (continuous score; separate font-hierarchy warnings) |
|
||||
|
||||
Classification: Phase Q area axis ⇄ Phase Z frame/slot axis are **not** drop-in compatible. The shape (dict indirection + OR alternation + tolerance + conditional implicit-requirement + continuous score) is the only portable element; every value (key strings, area names, literal patterns) is Phase Q-local.
|
||||
|
||||
## A4 — IMP-04 soft-link boundary (catalog vs validation ownership)
|
||||
|
||||
IMP-20 is `soft link: IMP-04` per the backlog (`docs/architecture/PHASE-Z-IMPLEMENTATION-ISSUE-BACKLOG.md:71`). Ownership separation:
|
||||
|
||||
- **IMP-04 owns**: every `frame_contracts.yaml` entry — addition / removal / `accepted_content_types` change / `sub_zones` schema change / `cardinality` change / `visual_hints` change. `templates/phase_z2/catalog/frame_contracts.yaml` is the IMP-04 source of truth.
|
||||
- **IMP-20 owns**: reference-only documentation of the Phase Q pattern-dict shape (A1 + A2) and the Phase Z target axis design narrative (A3). No catalog edits, no Step 10 promotion.
|
||||
- **Coupling direction**: **one-way** read. A Phase Z pattern dict (if/when activated through the A5 gate) consumes `frame_contracts.yaml` as input. It does **not** publish back into the catalog. IMP-04 is unaware of IMP-20.
|
||||
- **No bidirectional code flow**: IMP-20 does not move Phase Q `content_verifier.py` code into Phase Z, and IMP-04 does not consume `REQUIRED_PATTERNS`. The two surfaces remain isolated.
|
||||
- **Reference direction is one-way**: this document points read-only at `src/content_verifier.py`, `src/phase_z2_mapper.py`, `src/phase_z2_pipeline.py`, and `templates/phase_z2/catalog/frame_contracts.yaml`. No reverse pointer is required in those source files.
|
||||
|
||||
If IMP-04 alters the catalog schema (e.g. adds `density_envelope` or renames `sub_zones`), A3 must be re-verified (key axis and conditional-branch row in particular). The boundary statement itself does not change.
|
||||
|
||||
## A5 — Re-activation gate + guardrails
|
||||
|
||||
IMP-20 is `documented` (dormant). Re-activation requires **all** of the following gate conditions (3-cond AND):
|
||||
|
||||
1. **Trigger**: Phase Z Step 10 produces a verifiable case where the partial frame-contract emit alone is insufficient — i.e., a final.html regression that a frame_id × sub_zone pattern dict would have caught (missing slot marker, contract field violation, font-hierarchy breach against a sub_zone-resolved max). The trigger must be a regression that maps cleanly to the frame/slot axis, **not** to a higher layer (composition planning, content adapter, render-time CSS).
|
||||
2. **Evidence requirement**: failing-case MDX + `step10_frame_contract.json` trace + final.html excerpt with the slot path that should have asserted, attached to a new issue or this issue's reopened state.
|
||||
3. **IMP-04 sign-off**: the IMP-04 owner confirms the failing case is **not** addressable inside the catalog (e.g. tightening `cardinality` or `accepted_content_types` does not resolve it) — only then is a Phase Z-native pattern dict justified.
|
||||
|
||||
Design questions resolved in this document (revisit if the gate fires):
|
||||
|
||||
- **Q1 — Key granularity**: `(frame_id, sub_zone_id)`. Frame-only granularity is insufficient because contracts with `sub_zones` of differing `accepts` (e.g. F29 `process_column` accepts `[text_block, transform_table]` vs `product_column` accepts `[text_block]`) require slot-level differentiation.
|
||||
- **Q2 — Value type**: hybrid — substring patterns (Phase Q parity) **plus** contract-field assertions (`cardinality.strict` / `accepts` membership / `partial_target_path` resolved in DOM) **plus** numeric tolerance (carried from font-hierarchy 1px). Three lanes preserved separately so each can fail/pass independently.
|
||||
- **Q3 — Validation timing**: post Step 18 final.html **only**. Step 12 light_edit/restructure proposal is upstream of render and exposes no HTML for substring assertion; running the dict there would either fire false negatives (no DOM yet) or duplicate Step 18 work.
|
||||
- **Q4 — Font-hierarchy carry-over**: replaced — Phase Q's `role_font_map` fixed dict (area → max-font) is Phase Q-local. The Phase Z equivalent reads from `frame_contracts.yaml` `visual_hints` (`min_height_px` already present; a future `max_font_px` field would live in `visual_hints` and is IMP-04-owned). 1px tolerance shape is portable; the lookup source is replaced.
|
||||
|
||||
Guardrails (preserved from Stage 1 + Stage 2):
|
||||
|
||||
- **GR1 — Shape-only reference**: no Phase Q `REQUIRED_PATTERNS` value (`"key-msg"`, `"padding-left"`, `"text-indent"`, `"slide-img-"`) or area name (`body_bg`/`body_core`/`sidebar`/`footer`) may appear in any Phase Z pattern dict activation.
|
||||
- **GR2 — Phase Q no-regression**: `src/content_verifier.py:382-392` `REQUIRED_PATTERNS` is no-touch. The Phase T `L379-381` comment (overflow:hidden removed) remains the no-regression boundary; any Phase Z dict design must not re-introduce removed patterns into Phase Q's surface.
|
||||
- **GR3 — Phase Z dict is Phase Z-owned**: no `import` of `content_verifier.REQUIRED_PATTERNS` from Phase Z code. The two pattern dicts coexist without symbol sharing.
|
||||
- **GR4 — IMP-04 soft-link one-way**: per § A4. Activating IMP-20 must not block on or modify IMP-04; the catalog is read-only input.
|
||||
- **PZ-1 — AI isolation contract**: pattern dict is code/spec, not AI-generated content. No Kei rewrite, no LLM proposal of pattern values (`feedback_ai_isolation_contract`).
|
||||
- **RULE 13 — Anchor sync**: any future activation must update backlog (`PHASE-Z-IMPLEMENTATION-ISSUE-BACKLOG.md`), status board (`PHASE-Z-PIPELINE-STATUS-BOARD.md`), and INSIGHT-MAP (`PHASE-Q-INSIGHT-TO-22STEP-MAP.md`) in the same commit.
|
||||
|
||||
If IMP-04 alters the catalog schema or `src/content_verifier.py` is rewritten upstream, A1–A3 must be re-verified (file:line refs); the A5 gate itself does not change.
|
||||
@@ -0,0 +1,59 @@
|
||||
# IMP-31 — AI-assisted frame-aware adaptation activation gate audit
|
||||
|
||||
**Status**: design-only audit. IMP-31 (#40) = IMP-17 carve-out activation tracking issue. No new design slot. No runtime AI code lands until the 3-condition AND gate clears.
|
||||
|
||||
**Source**
|
||||
- Gitea issue [#40](https://gitea.hmac.kr/Kyeongmin/C.E.L_Slide_test2/issues/40) IMP-31 — AI-assisted frame-aware adaptation (restructure / reject routes).
|
||||
- Carve-out boundary spec: [`IMP-17-CARVE-OUT.md`](IMP-17-CARVE-OUT.md) (allowed / forbidden / activation gate).
|
||||
- Backlog row: [`PHASE-Z-IMPLEMENTATION-ISSUE-BACKLOG.md`](PHASE-Z-IMPLEMENTATION-ISSUE-BACKLOG.md):68 (IMP-17 — carve-out, normal path 밖, soft link IMP-04 + IMP-05).
|
||||
- Stage 1 / Stage 2 exit reports: `.orchestrator/issues/40_stage_problem-review_exit.md` (Stage 1 binding contract).
|
||||
|
||||
## Issue-body anchor drift (axis C1)
|
||||
|
||||
Issue body cites `src/phase_z2_pipeline.py:452` for IMP-05 L5 `_imp05_route_hint()`. Current anchor surface (commit `1efbf67`):
|
||||
|
||||
- `:570` — conceptual comment ("restructure → AI-assisted frame-aware adaptation (deferred to IMP-17 …)").
|
||||
- `:572` — `_IMP05_ROUTE_HINTS: dict[str, str] = {` declaration.
|
||||
- `:575` — `"restructure": "ai_adaptation_required"` entry.
|
||||
- `:580` — `def _imp05_route_hint(label: Optional[str]) -> Optional[str]:`.
|
||||
- `:664` — `"route_hint": _imp05_route_hint(match.label)` candidate_evidence emission.
|
||||
|
||||
Anchor pin: `tests/orchestrator_unit/test_imp17_comment_anchor.py`. Synced in [`IMP-17-CARVE-OUT.md`](IMP-17-CARVE-OUT.md):10 (Stage 3 u1).
|
||||
|
||||
## 3-condition AND gate state (this cycle)
|
||||
|
||||
| # | Condition | State | Evidence |
|
||||
|---|---|---|---|
|
||||
| 1 | User GO — explicit activation request | **NOT CLEAR** | No axis activation directive in #40. Stage 1 root_cause: runtime consumer = 0. |
|
||||
| 2 | B4 frame_selection evidence integration complete | **NOT CLEAR** (⚠ partial) | [`PHASE-Z-PIPELINE-STATUS-BOARD.md`](PHASE-Z-PIPELINE-STATUS-BOARD.md):48 Step 9 ⚠ partial; :82 "B4 frame_selection 의 V4 evidence 미통합"; :126 (j) ❌ pending. |
|
||||
| 3 | IMP-04 catalog expansion + IMP-05 V4 fallback live | **AMBIGUOUS** | `templates/phase_z2/catalog/frame_contracts.yaml` = 11 `template_id:` entries vs 32 target. IMP-05 V4 rank-2/3 fallback selector logic live, but catalog coverage gates real semantics. |
|
||||
|
||||
**Verdict**: gate **NOT CLEAR**. Runtime AI adaptation remains gated. `src/phase_z2_ai_fallback/` = **scaffolded under IMP-33** (#61, Stage 3 u1~u11); module created, but `settings.ai_fallback_enabled` defaults to `False` (u1) so normal-path AI call count remains 0 (PZ-1). Runtime engagement still requires the 3-condition AND gate above.
|
||||
|
||||
## Issue-body axis verdict
|
||||
|
||||
| Axis | Issue-body line | Verdict | Binding boundary |
|
||||
|---|---|---|---|
|
||||
| A1 | restructure → ai_adaptation_required actual adaptation route | **gate-blocked** | Allowed only inside [`IMP-17-CARVE-OUT.md`](IMP-17-CARVE-OUT.md) Step 12 fallback path; runtime AI consumer not added this cycle. |
|
||||
| A2 | reject → design_reference_only | **gate-blocked + frontend ownership** | Reject route = design reference only. Frontend zone-level override remains IMP-29 scope ([`PHASE-Z-PIPELINE-OVERVIEW.md`](PHASE-Z-PIPELINE-OVERVIEW.md) Step 12). |
|
||||
| A3 | AI call provider | **Anthropic API only** | Kei API / `EDITOR_PROMPT` / Kei-API endpoint forbidden (Phase Q Kei persona 영구 단절 — [`IMP-17-CARVE-OUT.md`](IMP-17-CARVE-OUT.md) §"AI 격리 + Kei persona 단절 contract"). |
|
||||
| A4 | candidate_evidence[].route_hint | **live (deterministic emission)** | Emission anchored at `src/phase_z2_pipeline.py:570/:572/:575/:580/:664`; AI consumer deferred. Anchor pin: `tests/orchestrator_unit/test_imp17_comment_anchor.py`. |
|
||||
| A5 | MDX content preservation = strict | **locked** | No invent / rewrite / compress / summarize ([`IMP-17-CARVE-OUT.md`](IMP-17-CARVE-OUT.md) §Forbidden; memory `feedback_phase_z_spacing_direction`). |
|
||||
| A6 | AI prompt = frame-aware placement only, not "rewrite content" | **locked** | Output = content_object → Internal Region / Frame Slot placement proposal at content-object granularity ([`IMP-17-CARVE-OUT.md`](IMP-17-CARVE-OUT.md) §Allowed). HTML / CSS / layout / zone topology / frame selection X. |
|
||||
| A7 | popup / details / zone-resize routing when content cannot fit | **deferred to Step 17 fallback** | Deterministic actions exhausted (zone_ratio_retry / layout_adjust / frame_reselect / details_popup_escalation / image_fit_candidate / frame_internal_fit_candidate) before AI proposal ([`IMP-17-CARVE-OUT.md`](IMP-17-CARVE-OUT.md) §Allowed Step 16/17). |
|
||||
| A8 | no `calculate_fit` migration | **locked** | IMP-05 selector uses V4 labels + frame-contract presence + Phase Z capacity precheck only (`src/phase_z2_pipeline.py:587` `lookup_v4_match_with_fallback` declaration; :599 docstring "it does not call calculate_fit"; secondary anchors :3093 / :4871). |
|
||||
| C1 | Anchor drift `:452` → current | **synced** | Stage 3 u1 — [`IMP-17-CARVE-OUT.md`](IMP-17-CARVE-OUT.md):10. |
|
||||
| C2 | Backlog + status-board cross-ref | **planned (u3)** | Cross-ref discoverability surfaces only; no verdict duplication. |
|
||||
|
||||
## Out of scope (this cycle)
|
||||
|
||||
Runtime AI consumer enablement (flag default OFF), `candidate_evidence` schema change, Phase Q file mutation, Kei API reuse, frontend zone override (IMP-29 scope), IMP-30 invariant change, `calculate_fit` migration. Note: `src/phase_z2_ai_fallback/` directory scaffold itself was created under IMP-33 (#61, Stage 3 u1~u11) — see [`IMP-17-CARVE-OUT.md`](IMP-17-CARVE-OUT.md) §"Runtime module surface".
|
||||
|
||||
## Future activation path
|
||||
|
||||
When the 3-condition AND gate clears (User GO ∧ B4 V4 evidence integrated ∧ catalog 32/32 + IMP-05 V4 fallback live):
|
||||
|
||||
- Runtime AI module path = `src/phase_z2_ai_fallback/` (scaffolded under IMP-33; flag default OFF until gate clears).
|
||||
- Provider = Anthropic API only. Prompt design starts fresh (no Phase Q `EDITOR_PROMPT` import).
|
||||
- Output granularity = content_object → Internal Region / Frame Slot placement proposal. Frame / layout / zone topology selection remains deterministic.
|
||||
- Activation tracker = this issue (#40, IMP-31). No new IMP ID issued.
|
||||
@@ -140,7 +140,7 @@ Axis 3 (Section 5) will verify each pair has agreeing producer-line / consumer-l
|
||||
|---|---|---|
|
||||
| C1 | `debug.json` schema | phase_z2 debug payload paths; no conflicting key type / semantics |
|
||||
| C2 | `visual_check_passed` | `src/phase_z2_pipeline.py` Step 14 / 17; set-site <-> read-site agree |
|
||||
| C3 | `fit_classification` / router | `src/phase_z2_mapper.py` + consumers; labels consistent producer -> consumer |
|
||||
| C3 | `fit_classification` / router | `src/phase_z2_mapper.py` + consumers; labels consistent producer -> consumer (charter mis-cite; live producer = `src/phase_z2_classifier.py` -- see §10 F-1) |
|
||||
| C4 | Step 14 / 17 / 21 interactions | expected state values stay aligned across the trio |
|
||||
| C5 | Phase R vs Phase Z boundary | no R regression, Z additions don't leak into R |
|
||||
| C6 | template / catalog / frame count | all docs / code use same numbers (family = 13) |
|
||||
@@ -174,8 +174,8 @@ Axis 3 (Section 5) will verify each pair has agreeing producer-line / consumer-l
|
||||
|
||||
- 6 invariant categories evaluated. All AGREE for the closed-issue audit scope.
|
||||
- 2 surface notes recorded as Section 10 follow-up candidates :
|
||||
- **F-1** : issue body cites `src/phase_z2_mapper.py` for invariant C3 (`fit_classification`), but the live producer is `src/phase_z2_classifier.py`. Record-keeping correction needed in any future audit charter, not a code conflict.
|
||||
- **F-2** : 2 untracked family templates exist on disk without `frame_contracts.yaml` entries; IMP-18 doc cites "families/*.html (13)" forward-looking. Tracked baseline (11 / 11) is consistent. Contract drift is *not* present for any closed issue; the WIP delta belongs to open work.
|
||||
- **F-1** : issue body cites `src/phase_z2_mapper.py` for invariant C3 (`fit_classification`), but the live producer is `src/phase_z2_classifier.py`. Record-keeping correction needed in any future audit charter, not a code conflict. RESOLVED via IMP-53 (2026-05-19)
|
||||
- **F-2** : 2 untracked family templates exist on disk without `frame_contracts.yaml` entries; IMP-18 doc cites "families/*.html (13)" forward-looking. Tracked baseline (11 / 11) is consistent. Contract drift is *not* present for any closed issue; the WIP delta belongs to open work. RESOLVED via #52 option (c) (2026-05-19) -- WIP allowlist captured in `templates/phase_z2/families/_WIP_FILES.md`; tracked + contracted baseline unchanged at 11/11; promote / remove gated on #42.
|
||||
- 1 documented partial recorded :
|
||||
- Step 21 `_write_step_artifact` at `pipeline.py:4772` carries `step_status="partial"` with note `region marker partial 미주입 -- Step 21 ⚠ partial`. This is *self-honest acknowledged* per `feedback_artifact_status_naming`; no cross-issue conflict.
|
||||
- Phase R' <-> Phase Z boundary clean both directions for the 22 closed issues.
|
||||
@@ -440,7 +440,7 @@ F-4 / F-5 are optional and do not gate #19.
|
||||
|
||||
Five candidates were produced by Axes 1-4. F-3 + F-2 + F-1 are blocking conditions for upgrading §9 CONDITIONAL GO -> unconditional GO for #19; F-4 + F-5 are optional housekeeping. None require source-code changes inside this audit.
|
||||
|
||||
### 10.1 F-1 -- audit charter record-keeping : invariant C3 producer file path
|
||||
### 10.1 F-1 -- audit charter record-keeping : invariant C3 producer file path -- RESOLVED via IMP-53 (2026-05-19)
|
||||
|
||||
- **title** : `[AUDIT-CHARTER-FIX] invariant C3 (fit_classification) producer cited as src/phase_z2_mapper.py; live producer is src/phase_z2_classifier.py`
|
||||
- **source_axis** : Axis 3 (cross-issue conflict, invariant category C3) -- recorded in §5.2 C3 row + §5.4 follow-up bullet F-1.
|
||||
@@ -455,7 +455,7 @@ Five candidates were produced by Axes 1-4. F-3 + F-2 + F-1 are blocking conditio
|
||||
- Live producer site : `src/phase_z2_classifier.py:495-497` (return dict with `visual_check_passed`, `classifications`, `summary`, `categories_seen`, `unclassified_signals`, `placement_diagnostics`).
|
||||
- **priority / gating** : low priority on its own; required for charter cleanliness; **not** a blocker for #19 Stage 2.
|
||||
|
||||
### 10.2 F-2 -- family template count reconciliation : 11 tracked / 11 contracted / 13 on disk
|
||||
### 10.2 F-2 -- family template count reconciliation : 11 tracked / 11 contracted / 13 on disk -- RESOLVED via #52 (option c, 2026-05-19)
|
||||
|
||||
- **title** : `[FAMILY-TEMPLATE-RECONCILE] templates/phase_z2/families/ has 13 .html files on disk but 11 tracked + 11 frame_contracts entries; 2 WIP files (app_sw_package_vs_solution.html, pre_construction_model_info_stacked.html) untracked`
|
||||
- **source_axis** : Axis 3 (invariant category C6 template / catalog / frame count) -- recorded in §5.2 C6 row + §5.4 follow-up bullet F-2 + §6.3 Axis 4 cross-axis consistency bullet.
|
||||
@@ -468,6 +468,12 @@ Five candidates were produced by Axes 1-4. F-3 + F-2 + F-1 are blocking conditio
|
||||
- REPORT §5.5 row "C6 family templates -- on disk" (`ls templates/phase_z2/families/*.html` = 13).
|
||||
- REPORT §6.3 Axis 4 cross-axis consistency bullet (matches IMP-04 evidence).
|
||||
- **priority / gating** : **must land before #19 introduces any new family template** (per §9.3 condition 2). Until #19's catalog touch surface is known, this can be filed independently.
|
||||
- **resolution** : option (c) -- 2 WIP family templates explicitly noted as in-progress and tracked outside `frame_contracts.yaml` (RESOLVED via Gitea #52, 2026-05-19) :
|
||||
- WIP allowlist : `templates/phase_z2/families/_WIP_FILES.md` (added by #52 u1) -- names both files with Figma frame IDs (`app_sw_package_vs_solution.html` -> frame 23 / `1171281203`; `pre_construction_model_info_stacked.html` -> frame 9 / `1171281180`) and explicit "not in `frame_contracts.yaml`, not in runtime matcher set" status; promote / remove gated on Gitea #42.
|
||||
- IMP-18 doc reconciled : `docs/architecture/IMP-18-SVG-GAP-REPORT.md` L28 + L30 + L51 corrected from disk-only "13 files" / "15 partials" wording to "11 contracted + 2 WIP untracked = 13 on disk" (#52 u2) -- runtime matcher consumes the contracted set only; doc / tracked / contracted surfaces agree at 11 active.
|
||||
- baseline guard (planned by #52 u4) : `tests/test_family_contract_baseline.py` will enforce tracked families <-> `frame_contracts.yaml` 1:1 set-equality modulo WIP allowlist parsed from `_WIP_FILES.md`; future drift (#42 or otherwise) fails CI.
|
||||
- tracked baseline (11 contracted families <-> 11 `frame_contracts.yaml` entries) unchanged; no contract entries added or removed; no runtime matcher mutation; **C6 invariant remains AGREE** for the closed-issue audit scope.
|
||||
- **F-2 closed-by-#52** under [[feedback_workflow_atomicity_rules]] (one commit = one decision unit), without re-opening any §5 C-invariant or §6.3 Axis 4 conclusion. #19 catalog-touch gate (per §9.3 condition 2) is now satisfied for the current 11/11 baseline; any #19 / #42 catalog growth must reconcile the WIP allowlist before merge.
|
||||
|
||||
### 10.3 F-3 -- backlog status sweep : 15 rows pending->implemented + 1 row pending->documented(deferred) + IMP-15 children footnote
|
||||
|
||||
@@ -515,6 +521,17 @@ Five candidates were produced by Axes 1-4. F-3 + F-2 + F-1 are blocking conditio
|
||||
- REPORT §8.2 fourth bullet (`tests/fixtures/` not yet established).
|
||||
- **priority / gating** : **optional, very low priority**. Filing is only justified when sample inventory grows; the current state is already aligned with the spirit of the rule.
|
||||
|
||||
#### 10.5.1 F-5 docs-only resolution addendum (#54 Stage 3 u5, 2026-05-19)
|
||||
|
||||
Per issue #54 Stage 2 plan, F-5 is closed as **docs-only**; no root `tests/fixtures/` directory is created in this work. The current fixture inventory does not justify migration, and the existing convention is sufficient. The convention is recorded here so future anti-hardcoding audits can distinguish fixture / test-only paths from production paths without re-discovering the §8 G6 PASS-WITH-NOTE baseline.
|
||||
|
||||
- **Existing convention (DO NOT CHANGE)** : `tests/phase_z2/fixtures/` exists as a YAML regression fixture root (loaded by `tests/phase_z2/test_fixtures_loader.py`). Subdirectories present at audit time : `tests/phase_z2/fixtures/build_layout_css/`, `tests/phase_z2/fixtures/retry_gate/`. This is the canonical home for Phase Z regression fixtures.
|
||||
- **Root `tests/fixtures/` (ABSENT)** : not created in #54. If a future change requires a non-Phase-Z, non-YAML fixture corpus (for example, multi-file MDX golden inputs that grow beyond what `tests/phase_z2/test_*.py` can hold inline), the migration must be filed as its own Gitea issue with its own scope-lock per §10.5.
|
||||
- **Allowed sample references** : `samples/mdx_batch/**` and `samples/mdx/**` may be referenced from `tests/**` (test-only paths) for integration smoke -- e.g. the existing `samples/mdx_batch/02.mdx` references in `tests/phase_z2/test_pz2_vu_integration.py`. These do not violate the §8 anti-hardcoding rule because the spirit of the rule targets production pipeline code, not test runners.
|
||||
- **Forbidden sample references** : production pipeline code (`src/**` runtime path) must NOT hardcode sample-specific MDX filenames or content (e.g. `02.mdx`, `03.mdx`, frame-specific labels keyed to a sample). The 20 legacy Phase R'/Q hits annotated under F-4 (#54 Stage 3 u1-u4) are intentional documented examples in docstrings / comments / glossary regex / sample-data dicts, not runtime input pins; they are out of scope for this rule by §10.4 verdict.
|
||||
- **AI-isolation contract** : this addendum is text-only. No production behavior change, no runtime sample-path mutation, no new fixture file. Compatible with PZ-1 (AI = 0 on normal path) and [[feedback_ai_isolation_contract]].
|
||||
- **Cross-reference** : `tests/CLAUDE.md` fixture convention note (#54 Stage 3 u5) mirrors the test-only / production rule split documented here.
|
||||
|
||||
### 10.6 Follow-up summary
|
||||
|
||||
| candidate | source axis | doc-only? | gates #19? | priority |
|
||||
|
||||
@@ -0,0 +1,197 @@
|
||||
# INTEGRATION-AUDIT-02 — IMP-07 reverse-path ↔ backlog ↔ IMP-16-U2 deferred items
|
||||
|
||||
**Issue**: Gitea #56 ([`Kyeongmin/C.E.L_Slide_test2/issues/56`](https://gitea.hmac.kr/Kyeongmin/C.E.L_Slide_test2/issues/56))
|
||||
**Mode**: audit-only (orchestrator P4/P4a) — no runtime code; reverse-path NOT implemented in this audit.
|
||||
**HEAD at audit**: `47f072e` (`docs: PROJECT-INTENT-AND-GOVERNANCE master doc`)
|
||||
**Binding evidence artifact**: `.orchestrator/tmp/issue7_comments_r3.json` (102144 B, mtime_utc `2026-05-19T17:11:58Z`, 13 comments)
|
||||
**Live Gitea API calls during audit**: 0 (artifact is binding per Stage 1)
|
||||
**Fallback exit-report check**: `ls .orchestrator/issues/ | grep '^7_stage' | wc -l = 0` (no local stage-exit fallback)
|
||||
|
||||
**Scope-lock (u1 binding)**
|
||||
- Forbidden writes (4 surfaces): `src/**`, `templates/**`, `tests/**`, `docs/architecture/IMP-16-U2-WIRING-DESIGN.md`.
|
||||
- Allowed writes (2 surfaces): CREATE this report; line-scoped EDIT to `docs/architecture/PHASE-Z-IMPLEMENTATION-ISSUE-BACKLOG.md` L51 + L67 status cells only.
|
||||
|
||||
**Cross-links**
|
||||
- Project governance: [`PROJECT-INTENT-AND-GOVERNANCE.md`](PROJECT-INTENT-AND-GOVERNANCE.md)
|
||||
- Pipeline anchors: [`PHASE-Z-PIPELINE-OVERVIEW.md`](PHASE-Z-PIPELINE-OVERVIEW.md), [`PHASE-Z-PIPELINE-STATUS-BOARD.md`](PHASE-Z-PIPELINE-STATUS-BOARD.md)
|
||||
- Backlog: [`PHASE-Z-IMPLEMENTATION-ISSUE-BACKLOG.md`](PHASE-Z-IMPLEMENTATION-ISSUE-BACKLOG.md)
|
||||
- Wiring-design (read-only, not edited here): [`IMP-16-U2-WIRING-DESIGN.md`](IMP-16-U2-WIRING-DESIGN.md)
|
||||
- Prior audit: [`INTEGRATION-AUDIT-01-REPORT.md`](INTEGRATION-AUDIT-01-REPORT.md)
|
||||
|
||||
---
|
||||
|
||||
## 1. Executive decision
|
||||
|
||||
| Q | question | verdict | evidence anchor |
|
||||
|---|---|---|---|
|
||||
| Q1 | IMP-07 actual implementation status | **closed-as-no-runtime** (policy close; no backend adapter; no FE trigger) | u2 close-trio c.17970 / c.19226 / c.19240; u3 BE grep 1 hit (docstring) + FE grep 0 hits |
|
||||
| Q2 | Backlog accuracy for IMP-07 (and dependent IMP-16) | **divergent — correct to `documented:no-runtime` (IMP-07) + `documented:dormant` (IMP-16)** | Backlog L51 / L67 currently both `implemented`; status-vocabulary precedent at L68–L71 (`documented`, `documented (deferred)`) |
|
||||
| Q3 | IMP-16-U2 3 deferred items resolution | **all three DORMANT pending reverse-path reactivation** (no runtime substrate to resolve any of the three) | u4 §3 (a/b/c each cite u2 + u3 + IMP-16-U2-WIRING-DESIGN.md L14–16 gate clauses, all NOT CLEARED) |
|
||||
| Q4 | Follow-up needs | **1 backlog correction (applied in u7) + 1 doc-sync follow-up (drafted in §6, NOT posted)**; no runtime follow-up needed under current policy | §5 (backlog patch) + §6 (doc-sync banner draft) |
|
||||
|
||||
**Final decision**: see §7 below.
|
||||
|
||||
---
|
||||
|
||||
## 2. Evidence table (4-axis convergence)
|
||||
|
||||
| axis | claim | observed state | source / anchor |
|
||||
|---|---|---|---|
|
||||
| Gitea #7 close text | reverse-path closed-as-no-runtime (policy) | c.17970 `<< 해당 기능 필요 없음 >>`; c.19226 §5 `"코드 변경 없이 close … '구현 완료'가 아니라 '기능 불필요 / 현 정책상 reverse path 미진행'"`; c.19240 `"이 이슈는 코드 변경 없이 정책 판단으로 close했다."` | `.orchestrator/tmp/issue7_comments_r3.json` (binding artifact); cited verbatim in `.orchestrator/drafts/56_close_evidence.md` §3 / §4 / §5 |
|
||||
| Live BE code grep (`src/`) | no reverse-path adapter exists | pattern P `html_to_slide_mdx\|edited_html_to_mdx\|reverse_path\|reverse-path\|reversePath\|html-to-mdx` → 1 hit at `src/phase_z2_verification_utils.py:68`, classified **docstring-only** inside `extract_text_from_html()` (docstring says `Deterministic, pure: no I/O, no LLM, no network.`) | `src/phase_z2_verification_utils.py:64-73`; `.orchestrator/drafts/56_code_grep.md` §3 |
|
||||
| Live FE code grep (`Front/client/src/`) | no reverse-path payload trigger exists | same pattern P → **0 hits** across populated tree (`App.tsx`, `components/`, `contexts/`, `data/`, `hooks/`, `lib/`, `pages/`, `services/`, `types/`, `utils/`); 0-hit is true absence, not missing-dir false negative | `.orchestrator/drafts/56_code_grep.md` §4 |
|
||||
| Backlog status (`PHASE-Z-IMPLEMENTATION-ISSUE-BACKLOG.md` L51) | currently labels IMP-07 `implemented` — divergent from #7 close text + grep | L51 final cell = `implemented`; row preserves hard link to IMP-02 (normalize schema). Correct token under audit verdict = `documented:no-runtime`. | `docs/architecture/PHASE-Z-IMPLEMENTATION-ISSUE-BACKLOG.md:51`; proposed diff in `.orchestrator/drafts/56_backlog_diff.md` §2 |
|
||||
| Backlog status (L67) | currently labels IMP-16 `implemented` — gated to closed IMP-07, so dormant | L67 final cell = `implemented`; row carries `hard link: IMP-07 (B-2 main 활성 시점 의미)`. Correct token under audit verdict = `documented:dormant`. | `docs/architecture/PHASE-Z-IMPLEMENTATION-ISSUE-BACKLOG.md:67`; proposed diff in `.orchestrator/drafts/56_backlog_diff.md` §3 |
|
||||
| IMP-16-U2 deferred items (`IMP-16-U2-WIRING-DESIGN.md` L71–L73) | three items deferred "until IMP-07 lands" | (a) adapter module path TBD, (b) Step 2 per-section vs whole-MDX undecided, (c) Step 14 telemetry granularity undecided. None can be resolved while IMP-07 remains closed-as-no-runtime. | `docs/architecture/IMP-16-U2-WIRING-DESIGN.md:69-75`; gate at L14–L16 (all 3 clauses NOT CLEARED — `.orchestrator/drafts/56_imp16_deferred.md` §2) |
|
||||
| Fallback orchestrator exit report | absent — binding artifact is sole source | `ls .orchestrator/issues/ \| grep '^7_stage' \| wc -l = 0` | `.orchestrator/drafts/56_close_evidence.md` §1 |
|
||||
| Convergence | zero contradicting evidence across 37 independent passes (Stage 1 → Stage 3) | all 4 evidence axes (close-text / BE grep / FE grep / dependent doc gate) point to **policy-closed, no runtime, dependent doc dormant** | Stage 1 + Stage 2 exit reports; u2–u5 drafts; u4 §4 cross-axis check |
|
||||
|
||||
**Commit SHA at audit time**: `47f072e` (HEAD before u7's backlog patch).
|
||||
|
||||
---
|
||||
|
||||
## 3. IMP-07 verdict (with evidence)
|
||||
|
||||
**Verdict**: `documented:no-runtime` — reverse-path (B-2 Edited HTML → MDX) was closed by user policy decision on 2026-05-15 (c.17970) and re-affirmed by structured close-audit on 2026-05-18 (c.19226 + c.19240). **No backend adapter, no frontend trigger, no `html_to_slide_mdx` port exists in this repository.**
|
||||
|
||||
### Evidence chain (compact form — full verbatim in drafts)
|
||||
|
||||
1. **Initial close decision** — c.17970 (2026-05-15T18:28:22+09:00):
|
||||
> `<< 해당 기능 필요 없음 >>`
|
||||
> `(*) mdx → html 변환 이후 html 수기 수정된 것은 html에서만 적용.`
|
||||
2. **Structured close-audit (v1)** — c.19226 (2026-05-18T08:31:05+09:00). Section 3 enumerates the *absence* of every required runtime surface: SlideCanvas outerHTML capture absent; backend POST absent; `/api/edit | /api/html_to_mdx | /api/save` endpoints absent; glubeot `html_to_slide_mdx` not ported. Section 5 verdict: `"코드 변경 없이 close … '구현 완료'가 아니라 '기능 불필요 / 현 정책상 reverse path 미진행'"`.
|
||||
3. **Structured close-audit (v2 restatement)** — c.19240 (2026-05-18T08:41:19+09:00):
|
||||
> `"이 이슈는 코드 변경 없이 정책 판단으로 close했다."`
|
||||
4. **Live BE grep** (`src/`, pattern P): 1 hit at `src/phase_z2_verification_utils.py:68` inside the docstring of `extract_text_from_html()`. Function body is a deterministic, pure text extractor (`no I/O, no LLM, no network`) — **not** a reverse-path adapter, **not** an HTML→MDX converter, **not** a pipeline re-entry call site.
|
||||
5. **Live FE grep** (`Front/client/src/`, pattern P): 0 hits across populated React/TS tree — true absence, not missing-dir false negative.
|
||||
|
||||
### Why not `implemented:partial`
|
||||
|
||||
c.19226 §3 enumerates the absence of **every** required runtime surface (frontend, backend, converter, endpoint). `implemented:partial` would imply at least one runtime substrate is present; none is.
|
||||
|
||||
### Why not plain `documented` / `documented (deferred)`
|
||||
|
||||
IMP-17/18/19/20 use `documented` / `documented (deferred)` to mean "design captured, runtime deferred pending an explicit activation gate (IMP-17 carve-out, IMP-18 gap report, etc.)". IMP-07 is a stronger statement — **closed by explicit policy decision, no runtime, reactivation requires reopening the policy in a separate issue**. The `:no-runtime` suffix encodes that distinction so future readers can tell IMP-07 apart from the IMP-17/18/19/20 `documented` family.
|
||||
|
||||
### Reactivation contract (informational, NOT a doc edit)
|
||||
|
||||
Per c.19226 §5 and c.19240 closing line, reverse-path reactivation requires reopening IMP-07 policy in a **separate** issue covering: endpoint design, marker coverage, re-entry validation. This audit does NOT reopen that policy.
|
||||
|
||||
---
|
||||
|
||||
## 4. IMP-16-U2 deferred items resolution (with evidence)
|
||||
|
||||
**Source**: `docs/architecture/IMP-16-U2-WIRING-DESIGN.md` lines 69–75 (read-only; this doc is FORBIDDEN to edit in this audit per u1).
|
||||
|
||||
**Governing gate** (doc L12–L16): three clauses MUST be cleared before any IMP-16-U2 wiring lands.
|
||||
|
||||
| gate clause | required state | observed | gate status |
|
||||
|---|---|---|---|
|
||||
| `IMP-07 implemented + verified` | runtime adapter in `src/`, verified | Gitea #7 closed as policy / no-runtime (c.17970 / c.19226 §5 / c.19240) | **NOT CLEARED** |
|
||||
| Repo grep returns runtime hit in non-test `src/` module | ≥1 non-docstring runtime hit for pattern P | u3 hits=1, **docstring only** at `src/phase_z2_verification_utils.py:68` (pure text extractor) | **NOT CLEARED** |
|
||||
| Reverse-path entry emits (a) re-entry MDX + (b) upstream HTML | both as deterministic outputs | c.19226 §3 enumerates absence of every required surface; u3 FE grep hits=0 | **NOT CLEARED** |
|
||||
|
||||
All three gate clauses NOT CLEARED → resolution policy from issue body Q3 branches: "If Q1 confirms no-runtime / dormant → reclassify item as dormant pending reverse-path reactivation."
|
||||
|
||||
### Per-item resolution
|
||||
|
||||
| item | text (verbatim, doc L71–L73) | classification | reason | evidence anchor |
|
||||
|---|---|---|---|---|
|
||||
| (a) | Exact module path of the IMP-07 reverse-path adapter (TBD by IMP-07). | **DORMANT** | No reverse-path adapter exists in `src/`. The TBD slot stays TBD — not answered with a placeholder path. | u3 §3 (single docstring hit at `src/phase_z2_verification_utils.py:68`); c.19226 §3 absent-surface enumeration |
|
||||
| (b) | Step 2 preservation cross-check: per-section variant vs whole-MDX variant. | **DORMANT (gate closed)** | Step 2 surface = `verify_text_preservation(reentry_mdx, upstream_generated_html, area_name=...)` (doc L29). With no emitter producing `reentry_mdx`, the per-section vs whole-MDX choice is unanswerable from runtime evidence. | doc L29; u3 §3; c.19226 §3 (`html_to_slide_mdx` not in repo); c.19226 §5 |
|
||||
| (c) | Step 14 invented-text telemetry: per `area_name` vs global. | **DORMANT (gate closed)** | Step 14 surface = `detect_invented_text(reentry_mdx, final_html)` (doc L35). With no FE producer of area-tagged HTML (u3 FE grep hits=0), the granularity question has no runtime substrate. The current Step 14 `run_overflow_check` path is unchanged because no reverse-path re-entry sets `debug.json["pipeline"]["reverse_path_reentry"] = True` (doc L42 schema gate). | doc L35; doc L42; u3 §4 (FE 0-hits); c.19240 closing line |
|
||||
|
||||
### Axis disambiguation (why DORMANT, not no-runtime)
|
||||
|
||||
IMP-07 is **policy-closed** (active decline). IMP-16's verification helpers are **code-present** in `src/phase_z2_verification_utils.py` (u6 `split_into_sentences`, u8 `verify_text_preservation`, u9 `detect_invented_text` ports). The wiring they would land is **gated by IMP-07** (doc L12–L16). Because the gate is closed, the helpers are runtime-inert — they have no upstream caller. `:dormant` captures "code-shape present, runtime entry-point absent"; `:no-runtime` would imply the helpers themselves are absent (they are not).
|
||||
|
||||
---
|
||||
|
||||
## 5. Backlog status correction proposal
|
||||
|
||||
Target file: `docs/architecture/PHASE-Z-IMPLEMENTATION-ISSUE-BACKLOG.md` — line-scoped edit to L51 and L67 status cells only. **Exactly 2 line changes**; surrounding cells (id / title / step / source / priority / scope / guardrail / dependency) byte-for-byte unchanged on both rows. Adjacent rows (L50 IMP-06, L52 IMP-08, L66 IMP-15, L68 IMP-17) untouched.
|
||||
|
||||
### L51 — IMP-07: `implemented` → `documented:no-runtime`
|
||||
|
||||
```diff
|
||||
-| ... | hard link: IMP-02 (A-1 normalize schema 와 reverse path schema 정합 필요) | implemented |
|
||||
+| ... | hard link: IMP-02 (A-1 normalize schema 와 reverse path schema 정합 필요) | documented:no-runtime |
|
||||
```
|
||||
|
||||
Justification: §3 verdict + close-trio (c.17970 / c.19226 / c.19240) + BE grep (docstring only) + FE grep (0 hits).
|
||||
|
||||
### L67 — IMP-16: `implemented` → `documented:dormant`
|
||||
|
||||
```diff
|
||||
-| ... | hard link: IMP-07 (B-2 main 활성 시점 의미) | implemented |
|
||||
+| ... | hard link: IMP-07 (B-2 main 활성 시점 의미) | documented:dormant |
|
||||
```
|
||||
|
||||
Justification: §4 — all three deferred items DORMANT under the IMP-07 no-runtime gate. The row's own `hard link: IMP-07` declares its meaning is conditioned on IMP-07 activation.
|
||||
|
||||
### Status-vocabulary precedent
|
||||
|
||||
Existing tokens in the file: `pending` (L45), `implemented` (L46–L66 majority), `documented (deferred)` (L68 IMP-17), `documented` (L69 IMP-18 / L70 IMP-19 / L71 IMP-20). The proposed `documented:<qualifier>` form is a minimal suffix extension of an already-present family — and is **explicitly enumerated by the issue body's Q2**: "propose corrected status (`implemented` / `implemented:partial` / `documented:dormant` / `documented:no-runtime` / etc.)".
|
||||
|
||||
---
|
||||
|
||||
## 6. Follow-up issue recommendations (drafts, NOT posted)
|
||||
|
||||
Auto-posting follow-ups is out-of-scope per u1. The drafts below are recommended text only; this audit does **not** post them.
|
||||
|
||||
### Recommended follow-up #1 — doc-sync banner for `IMP-16-U2-WIRING-DESIGN.md`
|
||||
|
||||
- **Draft title**: `[DOC-SYNC] IMP-16-U2-WIRING-DESIGN.md — add cross-reference banner to INTEGRATION-AUDIT-02-REPORT.md (IMP-07 closed-as-no-runtime context)`
|
||||
- **Scope sketch**:
|
||||
- Add a one-paragraph banner near the top of `IMP-16-U2-WIRING-DESIGN.md` (post-§1 "Status" paragraph) cross-referencing this audit report.
|
||||
- Banner content: IMP-07 was closed-as-no-runtime per Gitea #7 (c.17970 / c.19226 / c.19240). The L12–L16 gate clauses remain unchanged but are currently NOT CLEARED; the 3 deferred items (L71–L73) are DORMANT pending a future reverse-path reactivation issue.
|
||||
- **Do NOT** modify the gate clauses, the per-step wiring contract, or the deferred items themselves — preserve them verbatim as the binding contract for any future IMP-07 reactivation.
|
||||
- **Allowed file changes**: `docs/architecture/IMP-16-U2-WIRING-DESIGN.md` (banner add only); optionally a one-line back-link in `INTEGRATION-AUDIT-02-REPORT.md`.
|
||||
- **Forbidden**: any change to the doc's gate clauses, per-step contract, or deferred items list; any change to `src/**`, `templates/**`, `tests/**`.
|
||||
- **Acceptance**: banner contains explicit cross-link to `INTEGRATION-AUDIT-02-REPORT.md`, cites c.17970 / c.19226 / c.19240, and states the 3 deferred items are DORMANT (not resolved, not closed).
|
||||
- **Rationale for separating from this audit**: per u1 scope-lock, `IMP-16-U2-WIRING-DESIGN.md` is a forbidden write surface in INTEGRATION-AUDIT-02 (issue #56). The banner addition is a separate doc-sync axis.
|
||||
|
||||
### No runtime follow-up needed under current policy
|
||||
|
||||
Reverse-path runtime activation is **out-of-scope under current user policy** (c.17970 / c.19226 §5 / c.19240). A runtime follow-up would require reopening IMP-07 policy in a separate issue — that decision lies with the user, not with this audit. This audit does NOT recommend a runtime follow-up at this time.
|
||||
|
||||
### Pre-existing follow-up linkage (informational)
|
||||
|
||||
Per the issue body's "Sequence note", the next planned issue #57 ([P5][DORMANT-TRIGGER-GUARD]) will register IMP-17 / IMP-18 / IMP-19 + (per #56 outcome) IMP-16 / IMP-07 + IMP-20 as followup-linked to #55. This audit's verdict feeds #57's dormant-trigger registry input: IMP-07 enters as `documented:no-runtime`; IMP-16 enters as `documented:dormant`.
|
||||
|
||||
---
|
||||
|
||||
## 7. Final decision
|
||||
|
||||
**`NEEDS_DOC_SYNC_FOLLOWUP`**
|
||||
|
||||
Rationale: the in-scope reconciliation (backlog L51 + L67 status corrections) is performed in u7. However, `IMP-16-U2-WIRING-DESIGN.md` opens with `**Status**: design-only contract. **No runtime wiring lands in this issue.** All wiring is gated behind IMP-07 reverse-path activation (B-2 main). When IMP-07 lands, this doc becomes the binding contract …` (L3) — written under the original assumption that IMP-07 would eventually land as runtime. With IMP-07 now classified `documented:no-runtime` (policy decline, not deferred-pending-future), this framing is stale without a cross-reference banner pointing readers to the present audit. Because u1 forbids direct edits to that doc, the banner addition must be a separate follow-up issue (drafted in §6, NOT posted by this audit).
|
||||
|
||||
Why not `BACKLOG_PATCH_ONLY`: the backlog patch alone leaves `IMP-16-U2-WIRING-DESIGN.md` reading as a future-binding contract without acknowledging the IMP-07 close. A reader landing on that doc would not know to consult this audit.
|
||||
|
||||
Why not `NEEDS_RUNTIME_FOLLOWUP`: reverse-path runtime is out-of-scope under current user policy (c.17970 / c.19226 §5 / c.19240); recommending a runtime follow-up would contradict the binding close-decision.
|
||||
|
||||
---
|
||||
|
||||
## Acceptance Criteria checklist (issue body)
|
||||
|
||||
| AC | requirement | status |
|
||||
|---|---|---|
|
||||
| 1 | No production source code (`src/**`, `templates/**`, `tests/**`) changes | ✅ — u1 forbids; u2–u6 verified empty tracked diff on these surfaces; u7 scoped to BACKLOG.md only |
|
||||
| 2 | No direct modification of `IMP-16-U2-WIRING-DESIGN.md` | ✅ — u1 forbids; banner addition deferred to follow-up #1 in §6 |
|
||||
| 3 | Each of Q1~Q4 has evidence-backed answer | ✅ — §1 table cites u2/u3/u4 drafts; §3, §4, §5, §6 expand each answer |
|
||||
| 4 | Evidence table includes concrete `file:line`, comment IDs, commit SHAs | ✅ — §2 cites `src/phase_z2_verification_utils.py:68`, c.17970 / c.19226 / c.19240, SHA `47f072e`, `BACKLOG.md:51` / `:67`, `IMP-16-U2-WIRING-DESIGN.md:69-75` / `:12-16` |
|
||||
| 5 | Final decision ∈ {BACKLOG_PATCH_ONLY, NEEDS_DOC_SYNC_FOLLOWUP, NEEDS_RUNTIME_FOLLOWUP} | ✅ — §7 = `NEEDS_DOC_SYNC_FOLLOWUP` |
|
||||
| 6 | Body size budget: each Gitea comment ≤ 8000 chars | ✅ — Stage 3 comments split large evidence into `.orchestrator/drafts/56_*.md` + this report; report body itself is not a comment |
|
||||
|
||||
---
|
||||
|
||||
## Evidence drafts (RULE-6 evidence-only; NOT staged for commit)
|
||||
|
||||
- u1: `.orchestrator/drafts/56_scope_lock.md` — scope binding + forbidden / allowed writes.
|
||||
- u2: `.orchestrator/drafts/56_close_evidence.md` — c.17970 / c.19226 / c.19240 verbatim.
|
||||
- u3: `.orchestrator/drafts/56_code_grep.md` — `src/` 1 hit (docstring) + `Front/client/src/` 0 hits.
|
||||
- u4: `.orchestrator/drafts/56_imp16_deferred.md` — 3 deferred items DORMANT (per-item table).
|
||||
- u5: `.orchestrator/drafts/56_backlog_diff.md` — L51 + L67 status-cell diff proposal.
|
||||
|
||||
These drafts are evidence-only per RULE 6 and remain untracked. The committed deliverables of INTEGRATION-AUDIT-02 are: (i) this report (`INTEGRATION-AUDIT-02-REPORT.md`), and (ii) the 2 line-scoped status-cell edits applied in u7 (`PHASE-Z-IMPLEMENTATION-ISSUE-BACKLOG.md` L51 + L67).
|
||||
@@ -96,13 +96,13 @@ Phase Z 는 본체이고, Phase Q 는 부품 창고 / 참고 자산이다. Phase
|
||||
| id | 보완 항목 | 목적 | input | output | Phase Q 후보 파일 | 우선순위 |
|
||||
|---|---|---|---|---|---|---|
|
||||
| **A-1** | Stage 0 normalize 통합 | HTML-heavy / 비정형 raw MDX 를 Phase Z canonical input 으로 변환 | raw MDX text | `{clean_text, title, images, popups, tables, sections}` (frontmatter / 코드블록 보호 / list/table HTML 변환 / AST 구조 추출) | `mdx_normalizer.py`, `section_parser.py` | 높음 |
|
||||
| **A-2** | Catalog 확장 (frame_contracts + frame_partials) | V4 32 후보 중 backend 적용 가능한 frame 수 증가 (현재 3 → 32 목표) | `figma_to_html_agent/blocks/{frame_id}/` 의 index.html / assets / analysis.md | `templates/phase_z2/catalog/frame_contracts.yaml` entry + `templates/phase_z2/frames/{template_id}.html` partial | `block_reference.py`, `block_selector.py` | 높음 |
|
||||
| **A-3** | Frame preview png 일관성 | 모든 catalog frame 의 일관된 preview.png 자동 생성 (현재 figma_previews 우회) | frame partial HTML + assets | `figma_to_html_agent/blocks/{frame_id}/preview.png` | `renderer.py`, `html_generator.py` (selenium 캡처 흔적 추정) | 중 |
|
||||
| **A-4** | slide-base.html iframe-friendly mode | iframe embed 시 body padding / centering / min-height 미적용 (frontend CSS injection 제거) | slide-base.html template + query string `?embedded=1` 같은 시그널 | conditional CSS (standalone vs embedded) | `html_generator.py` | 중 |
|
||||
| **A-2** | Catalog 확장 (frame_contracts + frame_partials) | V4 32 후보 중 backend 적용 가능한 frame 수 증가 (현재 3 → 32 목표) | `figma_to_html_agent/blocks/{frame_id}/` 의 index.html / assets / analysis.md | `templates/phase_z2/catalog/frame_contracts.yaml` entry + `templates/phase_z2/frames/{template_id}.html` partial | `block_reference.py`, `block_selector.py` (간접 — catalog 로딩 / block 검색 패턴 reference; A-2 main = frame_contracts.yaml + frame_partials 신규 구축, Phase Q catalog schema ≠ Phase Z) | 높음 |
|
||||
| **A-3** | Frame preview png 일관성 | 모든 catalog frame 의 일관된 preview.png 자동 생성 (현재 figma_previews 우회) | frame partial HTML + assets | `figma_to_html_agent/blocks/{frame_id}/preview.png` | `slide_measurer.capture_slide_screenshot` (main), `renderer.py` (간접 — render-path 자료) | 중 |
|
||||
| **A-4** | slide-base.html iframe-friendly mode | iframe embed 시 body padding / centering / min-height 미적용 (frontend CSS injection 제거) | slide-base.html template + query string `?embedded=1` 같은 시그널 | conditional CSS (standalone vs embedded) | `renderer.py` (legacy `slide-base.html` 호출 지점 보유, embedded/standalone CSS 분기 미구현) | 중 |
|
||||
| **A-5** | V4 후보 자동 fallback | rank-1 capacity / cardinality / structure mismatch 시 자동 rank-2/3 시도 | V4 후보 list + 각 frame contract 의 cardinality + 추출된 content items | 통과한 frame template_id (모두 fail 시 filtered_capacity) | `fit_verifier.py` | 높음 |
|
||||
| **A-6** | Zone DOM 좌표 export | backend 가 zone 절대 px 좌표를 step08 / 별도 step 에 export (frontend 측정 우회) | layout_css + slide-base 좌표 | `zone_geometries_px: [{position, x, y, w, h}]` | `slide_measurer.py` | 중 |
|
||||
| **B-1** | Zone-section assignment override | 사용자 drag drop 결과를 backend 가 받아 composition planner 의 자동 결정 강제 변경 | `--override-section-assignment ZONE_ID=section_id,section_id` (CLI multi) | units 배치가 사용자 매핑 따름 | `pipeline.py`, `content_editor.py` | 중 |
|
||||
| **B-2** | Edited HTML → MDX 역변환 | frontend 편집 모드의 텍스트 변경이 새 final.html 에 반영 | edited HTML (iframe contentDocument outerHTML) | 새 MDX text 또는 patched mapper input | 글벗 `fmt_slide.py html_to_slide_mdx`, `content_editor.py` | 중 |
|
||||
| **B-1** | Zone-section assignment override | 사용자 drag drop 결과를 backend 가 받아 composition planner 의 자동 결정 강제 변경 | `--override-section-assignment ZONE_ID=section_id,section_id` (CLI multi) | units 배치가 사용자 매핑 따름 | `pipeline.py` (간접 — orchestration entry, Stage Y page_structure 생성 흐름 보유) | 중 |
|
||||
| **B-2** | Edited HTML → MDX 역변환 | frontend 편집 모드의 텍스트 변경이 새 final.html 에 반영 | edited HTML (iframe contentDocument outerHTML) | 새 MDX text 또는 patched mapper input | 글벗 `fmt_slide.py html_to_slide_mdx` | 중 |
|
||||
| **B-3** | Sub-section (### 단위) drag drop backend 처리 | backend 가 sub-section id 를 인식해서 zone 에 sub-section 단위로 매핑 | sub-section id (e.g., "03-1-sub-2") + zone_id | 그 sub-content 단위로 unit 분할 | `section_parser.py` | 낮 |
|
||||
| **B-4** | 다른 layout 의 zone-geometry override 확장 | top-1-bottom-2 / top-2-bottom-1 / left-1-right-2 / left-2-right-1 / grid-2x2 도 사용자 ratio override 적용 (현재 horizontal-2 / vertical-2 만) | `--override-zone-geometry` 인자 + 새 layout_preset 분기 | build_layout_css 의 grid 표현 (areas / cols / rows) | `space_allocator.py` | 낮 |
|
||||
| **D-1** | filtered_section_reasons 노출 UI | 사용자가 어떤 섹션이 왜 빠졌는지 즉시 인지 (Step 8 coverage UI) | `step20_slide_status.json.data.filtered_section_reasons` | frontend header / 패널 UI | N/A (frontend 만) — Phase Q audit 외 | 중 |
|
||||
@@ -122,7 +122,7 @@ Phase Z 는 본체이고, Phase Q 는 부품 창고 / 참고 자산이다. Phase
|
||||
3. `slide_measurer.py` (A-6)
|
||||
4. `fit_verifier.py` (A-5, D-2 간접)
|
||||
5. `space_allocator.py` (B-4)
|
||||
6. `content_editor.py` (B-1, B-2)
|
||||
6. `content_editor.py`
|
||||
7. `content_verifier.py` (검증 — B-2 후속)
|
||||
8. `renderer.py` (A-3, A-4)
|
||||
9. `html_generator.py` (A-3, A-4)
|
||||
|
||||
@@ -122,8 +122,8 @@
|
||||
| B-2 verification 보조 | Step 1, 2, 14, 21, 22 | §2.7 H3 (text 추출 / 정규화 / 비교 utility) | pending | yes (UI/backend) |
|
||||
| IMP-17 AI repair fallback infra (carve-out — see [`IMP-17-CARVE-OUT.md`](IMP-17-CARVE-OUT.md)) | Step 12, 16, 17 | §2.6 G3 (`httpx` + SSE streaming + retry + JSON parse pattern) | pending | no (AI fallback only) |
|
||||
| I3 SVG 좌표 보강 | Step 0, 9 | §2.8 I3 (`renderer._preprocess_svg_data`) | pending | yes (deterministic) |
|
||||
| I4 zone 비중 분배 | Step 8 | §2.8 I4 (`renderer._group_blocks_by_area`) | pending | yes (deterministic) |
|
||||
| H2 frame contract validation | Step 10 | §2.7 H2 (`content_verifier.verify_structure` pattern) | pending | yes (deterministic) |
|
||||
| IMP-19 I4 zone 비중 분배 (reference — see [`IMP-19-ZONE-RATIO-REFERENCE.md`](IMP-19-ZONE-RATIO-REFERENCE.md)) | Step 8 | §2.8 I4 (`renderer._group_blocks_by_area`) | pending | yes (deterministic) |
|
||||
| IMP-20 H2 frame contract validation (reference — see [`IMP-20-FRAME-CONTRACT-VALIDATION-REFERENCE.md`](IMP-20-FRAME-CONTRACT-VALIDATION-REFERENCE.md)) | Step 10 | §2.7 H2 (`content_verifier.verify_structure` pattern) | pending | yes (deterministic) |
|
||||
|
||||
---
|
||||
|
||||
@@ -147,7 +147,7 @@
|
||||
|
||||
| candidate ID | 출처 | cleanup 대상 | trigger axis |
|
||||
|---|---|---|---|
|
||||
| J3 | §2.9 `html_generator` | utility 중복 — `normalize_mdx` / `_slice_mdx_sections` / `_get_definitions` / `_get_conclusion` (vs §2.1 / §2.2 SoT) | Phase R' cleanup axis 활성 시 |
|
||||
| J3 | §2.9 `html_generator` | utility 중복 — `normalize_mdx` / `_slice_mdx_sections` / `_get_definitions` / `_get_conclusion` (vs §2.1 / §2.2 SoT) | Phase R' archive trigger AND §2.1/§2.2 SoT signature unification (both preconditions required to keep guardrail = code-removal-only) |
|
||||
| K5 | §2.10 `block_reference` + `block_selector` + §2.8 `renderer` | catalog 로드 + `_get_block_by_id` 중복 (3 module) | Phase R' cleanup 또는 Phase Z catalog 확장 axis 활성 시 |
|
||||
| L4 | §2.11 `pipeline` + §2.6 `content_editor` + §2.9 `html_generator` | `_parse_json` 중복 (3 module) | Phase R' cleanup 또는 Phase Z utility 통합 axis 활성 시 |
|
||||
|
||||
|
||||
@@ -43,16 +43,16 @@
|
||||
| ID | title | related step | source | priority | scope | guardrail / validation | dependency | status |
|
||||
|---|---|---|---|---|---|---|---|---|
|
||||
| IMP-01 | A-6 Zone DOM 좌표 export | Step 14, 21 | §2 A-6 Salvage | ↑ high (small) | `_MEASURE_SCRIPT` JS extension `getBoundingClientRect()` + artifact field 추가 | AI/Kei/V4/frame 선택 변경 X / DOM bbox trace / 기존 debug.json schema 보존 (additive) | none | pending |
|
||||
| IMP-02 | A-1 Stage 0 normalize chained adapter | Step 2 | §2 A-1 Salvage chained | ↑ high (medium) | `normalize_mdx_content` + `extract_major_sections` + `extract_conclusion_text` chained adapter + dual-write | AI/Kei normalize 회귀 X / step02 sections / sub_sections trace 설명 가능 | none | pending |
|
||||
| IMP-03 | A-1 popup/image/table trace | Step 3 | §2 A-1 chained 보강 | medium | normalized popups / images / tables → ContentObject 변환 (B1 v0 보강) | AI/Kei content extraction 회귀 X / popup/image/table 추출 trace 설명 가능 | hard link: IMP-02 (Stage 0 normalize output 의 popup/image/table list 의존) | pending |
|
||||
| IMP-04 | A-2 Catalog 확장 | Step 0, 9 | §2 A-2 새로 만들기 (핵심 unblocker) | medium (large) | `frame_contracts.yaml` + frame_partials 32 frame 등록/확장 | Phase R' frame catalog 회귀 X / V4 logic 변경 X / catalog 확장 후 PASS/FAIL 변화와 frame 선택 trace 설명 가능 | none | pending |
|
||||
| IMP-05 | A-5 V4 fallback | Step 9, 16, 17, 20 | §2 A-5 새로 만들기 | medium | Step 9 / Step 16 router 확장 (rank-1 fail 시 rank-2/3 fallback) + step20 status semantics | `calculate_fit` 통째 Migrate X (dual path 위험) / 신설 status (`PASS_WITH_FALLBACK` 등) 일관성 / frame 변경 허용 trace 설명 | hard link: IMP-04 (catalog 확장 후 fallback path 의미 있음) | pending |
|
||||
| IMP-06 | B-1 Zone-section override | Step 6 + input Step 1, 22 | §2 B-1 새로 만들기 (backend path) | medium | CLI 인자 + composition planner override path 신설 | Kei composition / Phase R' frame 보조 회귀 X / override 적용 시 composition_unit schema 정합 + trace | soft link: IMP-04 (frame 후보 ↑ 시 override 의미 ↑) | pending |
|
||||
| IMP-07 | B-2 Edited HTML → MDX reverse path | Step 22 + Step 1, 2 | §2 B-2 새로 만들기 (backend path) | medium | frontend edited HTML → backend → MDX 변환 → pipeline 재진입 (글벗 `html_to_slide_mdx` 참조) | AI/Kei reverse 회귀 X / 재진입 후 step02 정합 + visual_check 통과 | hard link: IMP-02 (A-1 normalize schema 와 reverse path schema 정합 필요) | pending |
|
||||
| IMP-08 | B-3 Sub-section drag drop | Step 3 | §2 B-3 새로 만들기 (backend schema) | ↓ low | Phase Z `section_id` schema 확장 (sub_sections 단위 매핑) | AI/Kei schema 회귀 X / backward compatible / step03 trace | hard link: IMP-02 (A-1 normalize sub_sections schema 의존) | pending |
|
||||
| IMP-09 | B-4 다른 layout zone-geometry | Step 8 | §2 B-4 새로 만들기 (backend layout) | ↓ low | `build_layout_css` 분기 확장 (top-1-bottom-2 / left-1-right-2 / grid-2x2 등) | Kei `build_containers_type_b` 회귀 X / step08 trace | none | pending |
|
||||
| IMP-10 | D-1 filtered_section_reasons UI | Step 20, 22 | §2 D-1 frontend 신규 | ↓ low | frontend UI — backend artifact read-only 표시 | AI/Kei UI 회귀 X / backend artifact read-only | none | pending |
|
||||
| IMP-11 | D-2 Frame min_height 표시 | Step 22 | §2 D-2 새로 만들기 (frontend hint + catalog 참조) | ↓ low | frontend UI — frame contract `min_height_px` read-only + resize hint | AI/Kei UI 회귀 X / catalog 참조 + resize limit | none | pending |
|
||||
| IMP-02 | A-1 Stage 0 normalize chained adapter | Step 2 | §2 A-1 Salvage chained | ↑ high (medium) | `normalize_mdx_content` + `extract_major_sections` + `extract_conclusion_text` chained adapter + dual-write | AI/Kei normalize 회귀 X / step02 sections / sub_sections trace 설명 가능 | none | implemented |
|
||||
| IMP-03 | A-1 popup/image/table trace | Step 3 | §2 A-1 chained 보강 | medium | normalized popups / images / tables → ContentObject 변환 (B1 v0 보강) | AI/Kei content extraction 회귀 X / popup/image/table 추출 trace 설명 가능 | hard link: IMP-02 (Stage 0 normalize output 의 popup/image/table list 의존) | implemented |
|
||||
| IMP-04 | A-2 Catalog 확장 | Step 0, 9 | §2 A-2 새로 만들기 (핵심 unblocker) | medium (large) | `frame_contracts.yaml` + frame_partials 32 frame 등록/확장 | Phase R' frame catalog 회귀 X / V4 logic 변경 X / catalog 확장 후 PASS/FAIL 변화와 frame 선택 trace 설명 가능 | none | implemented |
|
||||
| IMP-05 | A-5 V4 fallback | Step 9, 16, 17, 20 | §2 A-5 새로 만들기 | medium | Step 9 / Step 16 router 확장 (rank-1 fail 시 rank-2/3 fallback) + step20 status semantics | `calculate_fit` 통째 Migrate X (dual path 위험) / 신설 status (`PASS_WITH_FALLBACK` 등) 일관성 / frame 변경 허용 trace 설명 | hard link: IMP-04 (catalog 확장 후 fallback path 의미 있음) | implemented |
|
||||
| IMP-06 | B-1 Zone-section override | Step 6 + input Step 1, 22 | §2 B-1 새로 만들기 (backend path) | medium | CLI 인자 + composition planner override path 신설 | Kei composition / Phase R' frame 보조 회귀 X / override 적용 시 composition_unit schema 정합 + trace | soft link: IMP-04 (frame 후보 ↑ 시 override 의미 ↑) | implemented |
|
||||
| IMP-07 | B-2 Edited HTML → MDX reverse path | Step 22 + Step 1, 2 | §2 B-2 새로 만들기 (backend path) | medium | frontend edited HTML → backend → MDX 변환 → pipeline 재진입 (글벗 `html_to_slide_mdx` 참조) | AI/Kei reverse 회귀 X / 재진입 후 step02 정합 + visual_check 통과 | hard link: IMP-02 (A-1 normalize schema 와 reverse path schema 정합 필요) | documented:no-runtime |
|
||||
| IMP-08 | B-3 Sub-section drag drop | Step 3 | §2 B-3 새로 만들기 (backend schema) | ↓ low | Phase Z `section_id` schema 확장 (sub_sections 단위 매핑) | AI/Kei schema 회귀 X / backward compatible / step03 trace | hard link: IMP-02 (A-1 normalize sub_sections schema 의존) | implemented |
|
||||
| IMP-09 | B-4 다른 layout zone-geometry | Step 8 | §2 B-4 새로 만들기 (backend layout) | ↓ low | `build_layout_css` 분기 확장 (top-1-bottom-2 / left-1-right-2 / grid-2x2 등) | Kei `build_containers_type_b` 회귀 X / step08 trace | soft back-link: IMP-19 ([reference doc](IMP-19-ZONE-RATIO-REFERENCE.md) — Phase O block-level pattern reference, no runtime integration) | implemented |
|
||||
| IMP-10 | D-1 filtered_section_reasons UI | Step 20, 22 | §2 D-1 frontend 신규 | ↓ low | frontend UI — backend artifact read-only 표시 | AI/Kei UI 회귀 X / backend artifact read-only | none | implemented |
|
||||
| IMP-11 | D-2 Frame min_height 표시 | Step 22 | §2 D-2 새로 만들기 (frontend hint + catalog 참조) | ↓ low | frontend UI — frame contract `min_height_px` read-only + resize hint | AI/Kei UI 회귀 X / catalog 참조 + resize limit | none | implemented |
|
||||
|
||||
---
|
||||
|
||||
@@ -60,15 +60,17 @@
|
||||
|
||||
| ID | title | related step | source | priority | scope | guardrail / validation | dependency | status |
|
||||
|---|---|---|---|---|---|---|---|---|
|
||||
| IMP-12 | Step 16/17 retry 정밀화 | Step 16, 17 | §3 group B (Salvage deterministic) | medium | `redistribute` + glue + font compression — Step 16 router action 신설 + Step 17 action 실행 | AI fallback X / Kei retry loop (H5) 회귀 X / status semantics 일관 | soft link: IMP-05 (Step 16 router 영역 공유, 병렬 가능) | pending |
|
||||
| IMP-13 | A-3 frame preview 일관성 | Step 0, 14, 21 | §3 Salvage 후보 | ↓ low | `capture_slide_screenshot` Salvage — preview.png 자동 생성 path | Phase R' reference path 회귀 X / preview artifact trace | soft link: IMP-04 (catalog frame_partial 확장 시 의미 ↑) | pending |
|
||||
| IMP-14 | A-4 slide-base iframe mode | Step 13 | §3 새로 만들기 | ↓ low | `slide-base.html` conditional CSS (embedded vs standalone) | Claude / Phase R' HTML generation 회귀 X / Jinja2 deterministic | none | pending |
|
||||
| IMP-15 | Step 14 visual_check 보강 | Step 14, 21 | §3 H1 Reference Only | medium | image_aspect_mismatch / tabular_overflow 검사 추가 | AI/Kei classification 회귀 X / deterministic 검사 + trace | soft link: IMP-01 (Step 14 측정/trace layer 공유) | pending |
|
||||
| IMP-16 | B-2 verification 보조 axis | Step 1, 2, 14, 21, 22 | §3 H3 Reference Only | ↓ low | B-2 reverse path 의 verification 보조. main reverse path 는 IMP-07, 본 issue 는 text/visual/trace 검증 layer | AI/Kei verification 회귀 X / utility deterministic | hard link: IMP-07 (B-2 main 활성 시점 의미) | pending |
|
||||
| **IMP-17** | **AI repair fallback infra** (**carve-out — normal path 밖**) | Step 12, 16, 17 | §3 G3 | (별 axis priority — pending) | [carve-out boundary + activation gate](IMP-17-CARVE-OUT.md) (3-cond AND: User GO ∧ B4 frame_selection evidence ∧ IMP-04/05 live — full def in u2 doc) — `httpx` + SSE streaming + retry + JSON parse pattern reference — light_edit / restructure proposal | **normal path AI 호출 0 — 본 axis = fallback only, normal path 와 분리 설계** / Kei persona 단절 (Phase Q 자산과 단절) | soft link: IMP-04 + IMP-05 (catalog 확장 + V4 fallback 활성 시 의미) | pending |
|
||||
| IMP-12 | Step 16/17 retry 정밀화 | Step 16, 17 | §3 group B (Salvage deterministic) | medium | `redistribute` + glue + font compression — Step 16 router action 신설 + Step 17 action 실행 | AI fallback X / Kei retry loop (H5) 회귀 X / status semantics 일관 | soft link: IMP-05 (Step 16 router 영역 공유, 병렬 가능) | implemented |
|
||||
| IMP-13 | A-3 frame preview 일관성 | Step 0, 14, 21 | §3 Salvage 후보 | ↓ low | `capture_slide_screenshot` Salvage — preview.png 자동 생성 path | Phase R' reference path 회귀 X / preview artifact trace | soft link: IMP-04 (catalog frame_partial 확장 시 의미 ↑) | implemented |
|
||||
| IMP-14 | A-4 slide-base iframe mode | Step 13 | §3 새로 만들기 | ↓ low | `slide-base.html` conditional CSS (embedded vs standalone) | Claude / Phase R' HTML generation 회귀 X / Jinja2 deterministic | none | implemented |
|
||||
| IMP-15 | Step 14 visual_check 보강 | Step 14, 21 | §3 H1 Reference Only | medium | image_aspect_mismatch / tabular_overflow 검사 추가 | AI/Kei classification 회귀 X / deterministic 검사 + trace | soft link: IMP-01 (Step 14 측정/trace layer 공유) | implemented |
|
||||
| IMP-16 | B-2 verification 보조 axis | Step 1, 2, 14, 21, 22 | §3 H3 Reference Only | ↓ low | B-2 reverse path 의 verification 보조. main reverse path 는 IMP-07, 본 issue 는 text/visual/trace 검증 layer | AI/Kei verification 회귀 X / utility deterministic | hard link: IMP-07 (B-2 main 활성 시점 의미) | documented:dormant |
|
||||
| **IMP-17** | **AI repair fallback infra** (**carve-out — normal path 밖**) | Step 12, 16, 17 | §3 G3 | (별 axis priority — pending) | [carve-out boundary + activation gate](IMP-17-CARVE-OUT.md) (3-cond AND: User GO ∧ B4 frame_selection evidence ∧ IMP-04/05 live — full def in u2 doc) — `httpx` + SSE streaming + retry + JSON parse pattern reference — light_edit / restructure proposal. Activation tracker = IMP-31 (#40); current gate state in [`IMP-31-GATE-AUDIT.md`](IMP-31-GATE-AUDIT.md) | **normal path AI 호출 0 — 본 axis = fallback only, normal path 와 분리 설계** / Kei persona 단절 (Phase Q 자산과 단절) | soft link: IMP-04 + IMP-05 (catalog 확장 + V4 fallback 활성 시 의미) | documented (deferred) |
|
||||
| IMP-18 | I3 SVG 좌표 보강 | Step 0, 9 | §3 Reference Only | ↓ low | `renderer._preprocess_svg_data` 패턴 reference — frame_partials SVG 좌표 사전 박힘 — [gap report](IMP-18-SVG-GAP-REPORT.md) | Phase R' (renderer.py) 회귀 X | soft link: IMP-04 (frame_partials 등록 후 의미 ↑) | documented |
|
||||
| IMP-19 | I4 zone 비중 분배 | Step 8 | §3 Reference Only | ↓ low | `renderer._group_blocks_by_area` 패턴 reference — zone-level ratio 분배 | Phase O 컨테이너 회귀 X / 직접 통합 X | soft link: IMP-09 (zone 비중 분배 영역 공유) | pending |
|
||||
| IMP-20 | H2 frame contract validation | Step 10 | §3 Reference Only | ↓ low | `content_verifier.verify_structure` pattern reference — Phase Z frame contract 검증 pattern | Phase Q `REQUIRED_PATTERNS` 값 회귀 X / Phase Z 자체 pattern dict 설계 | soft link: IMP-04 (확장 catalog 적용 시 검증 범위 확대) | pending |
|
||||
| IMP-19 | I4 zone 비중 분배 | Step 8 | §3 Reference Only | ↓ low | `renderer._group_blocks_by_area` 패턴 reference — zone-level ratio 분배 — [reference doc](IMP-19-ZONE-RATIO-REFERENCE.md) | Phase O 컨테이너 회귀 X / 직접 통합 X | soft link: IMP-09 (zone 비중 분배 영역 공유) | documented |
|
||||
| IMP-20 | H2 frame contract validation | Step 10 | §3 Reference Only | ↓ low | `content_verifier.verify_structure` pattern reference — Phase Z frame contract 검증 pattern — [reference doc](IMP-20-FRAME-CONTRACT-VALIDATION-REFERENCE.md) | Phase Q `REQUIRED_PATTERNS` 값 회귀 X / Phase Z 자체 pattern dict 설계 | soft link: IMP-04 (확장 catalog 적용 시 검증 범위 확대) | documented |
|
||||
|
||||
> **IMP-15 child issues note (#45–#49)** — IMP-15 (Step 14 visual_check 보강) is the parent row; child sub-axes were tracked as separate Gitea issues and are not given standalone backlog rows. Children: #45 (e9b3d2e), #46 (2827622), #47 (535c484), #48 (614c533), #49 (verification-only). Per INTEGRATION-AUDIT-01 §10.3 footnote option to avoid double-counting under IMP-15.
|
||||
|
||||
---
|
||||
|
||||
@@ -88,7 +90,7 @@
|
||||
|
||||
| ID | title | related module | source | priority | scope | guardrail / validation | trigger axis | status |
|
||||
|---|---|---|---|---|---|---|---|---|
|
||||
| IMP-26 | J3 — html_generator utility 중복 cleanup | §2.9 html_generator | §5 J3 | ↓ low (future) | `normalize_mdx` / `_slice_mdx_sections` / `_get_definitions` / `_get_conclusion` 중복 제거 (vs §2.1/§2.2 SoT) | Phase R' 영역 — 코드 제거만 | Phase R' cleanup axis 활성 시 | pending |
|
||||
| IMP-26 | J3 — html_generator utility 중복 cleanup | §2.9 html_generator | §5 J3 | ↓ low (future) | `normalize_mdx` / `_slice_mdx_sections` / `_get_definitions` / `_get_conclusion` 중복 제거 (vs §2.1/§2.2 SoT) | Phase R' 영역 — 코드 제거만 | Phase R' archive trigger AND §2.1/§2.2 SoT signature unification (both preconditions required to keep guardrail = code-removal-only) | deferred |
|
||||
| IMP-27 | K5 — catalog 로드 + `_get_block_by_id` 중복 cleanup | §2.10 + §2.8 (3 module) | §5 K5 | ↓ low (future) | block_reference / block_selector / renderer 의 catalog 로드 중복 제거 | Phase R' 영역 또는 Phase Z catalog 확장 axis | Phase Z catalog 확장 axis 활성 시 (soft link: IMP-04) | pending |
|
||||
| IMP-28 | L4 — `_parse_json` 중복 cleanup | §2.11 + §2.6 + §2.9 (3 module) | §5 L4 | ↓ low (future) | pipeline / content_editor / html_generator 의 `_parse_json` 중복 제거 | Phase R' 영역 또는 Phase Z utility 통합 axis | Phase R' cleanup 또는 Phase Z utility 통합 axis 활성 시 | pending |
|
||||
|
||||
|
||||
@@ -46,7 +46,7 @@ Step 0 은 본체가 아닌 *준비 조건*. Step 1 (MDX 업로드) 부터가 ru
|
||||
| A | 7 | Slide-Level Layout Planning | ⚠ partial (count-based / 7-A catalog + 7-B candidate fn 추가, runtime 호출처 X) |
|
||||
| A | 8 | Zone + Internal Region Ratio Planning | ⚠ partial (zone-level horizontal-2 만 dynamic / 8-A region+display catalog + 8-B-1/2 candidate fn 추가, runtime 호출처 X / region-level 은 B2 안 partial) |
|
||||
| A | 9 | Region-Level Frame / Display Selection | ⚠ partial (B4 가 catalog cover + declaration order 로 frame 선택 분담 / V4 evidence 미통합 / Step 5 와 conflate 잔존) |
|
||||
| A | 10 | Frame Contract 확인 | ⚠ partial (B3 의 accepted_content_types + sub_zones 선언 추가 — B4 만 읽음, mapper 미읽음 / density envelope 별 axis) |
|
||||
| A | 10 | Frame Contract 확인 | ⚠ partial (B3 의 accepted_content_types + sub_zones 선언 추가 — B4 만 읽음, mapper 미읽음 / density envelope 별 axis) — IMP-20 ref: [reference doc](IMP-20-FRAME-CONTRACT-VALIDATION-REFERENCE.md) |
|
||||
| A | 11 | Content Unit / Child Group → Internal Region → Frame Slot Mapping | ⚠ partial (B4 v0 dormant 2-stage + region 1:1 sub_zone + narrowest first + trace-only runtime 호출, render path 미연결) |
|
||||
| A | 12 | Slot Payload 생성 | ✅ (deterministic) |
|
||||
| B | 13 | Render | ✅ |
|
||||
@@ -157,6 +157,8 @@ Step 0 (사전 준비) 의 Figma → HTML 변환은 *precondition phase 의 작
|
||||
|
||||
다른 step 에서의 AI 호출은 본 도면 안에 *없음*.
|
||||
|
||||
> **Activation status reference** : runtime AI fallback (Step 12 light_edit / restructure) 는 IMP-17 carve-out infra + IMP-31 activation tracker (#40) 로 관리. carve-out boundary = [`IMP-17-CARVE-OUT.md`](IMP-17-CARVE-OUT.md). current 3-condition AND gate state + issue-body axis verdict = [`IMP-31-GATE-AUDIT.md`](IMP-31-GATE-AUDIT.md). 본 board 는 verdict 중복 X — gate / axis 판정은 audit doc 따름.
|
||||
|
||||
---
|
||||
|
||||
## 6. 현재 병목 (한 줄)
|
||||
|
||||
@@ -0,0 +1,182 @@
|
||||
# 프로젝트의 목적과 거버넌스
|
||||
|
||||
> 이 문서는 **왜** 이 프로젝트를 하는지, **무엇을 위해** 이슈와 audit 을 도는지, 그리고 **그 구조가 어떻게 짜여있는지** 기록한다. 매번 처음부터 설명하지 않기 위함.
|
||||
>
|
||||
> 작성: 2026-05-20.
|
||||
|
||||
---
|
||||
|
||||
## 1. Destination (도착점)
|
||||
|
||||
**Phase Z 가 다음 두 가지까지 작동하면 프로젝트 목표 달성**:
|
||||
|
||||
1. **22-step pipeline** end-to-end 작동
|
||||
- 참조: [`PHASE-Z-PIPELINE-OVERVIEW.md`](PHASE-Z-PIPELINE-OVERVIEW.md)
|
||||
- 현재 status: [`PHASE-Z-PIPELINE-STATUS-BOARD.md`](PHASE-Z-PIPELINE-STATUS-BOARD.md)
|
||||
2. **AI 가 zone fit 평가 → 안 맞는 frame reject → zone 에 맞는 frame 생성**
|
||||
- frame 이 zone 안에 들어가지 않으면 AI 가 reject
|
||||
- reject 후 zone 에 맞춰 frame 을 생성하는 것까지가 destination
|
||||
|
||||
이 두 가지가 작동하면 끝. 그 이상은 별도 결정.
|
||||
|
||||
---
|
||||
|
||||
## 2. Q~Y 검토 = 이미 끝났음 (과거형)
|
||||
|
||||
Phase Z 구현 갭을 메우기 위해 Phase Q~Y 의 코드/기능을 **이미 다 검토했고**, 참고할 만한 것들을 22-step 에 매칭해서 **이슈로 다 정리해놓은 상태**.
|
||||
|
||||
- Q~Y 새로 다시 보지 않음 — 작업은 끝남
|
||||
- 결과물 = INSIGHT-MAP 문서 + 28 개 초기 IMP 이슈 (#1~#28)
|
||||
- 회귀 금지선 4 항목 (Q/R'/T 의 폐기된 path 로 돌아가지 않음) 도 [`PHASE-Q-INSIGHT-TO-22STEP-MAP.md §0`](PHASE-Q-INSIGHT-TO-22STEP-MAP.md) 에 같이 박혀있음
|
||||
|
||||
이제 남은 일 = **정리된 이슈를 orchestrator 로 처리해서 Phase Z 에 반영하는 것**.
|
||||
|
||||
---
|
||||
|
||||
## 3. 그 검토 결과 = INSIGHT-MAP 문서
|
||||
|
||||
**문서**: [`PHASE-Q-INSIGHT-TO-22STEP-MAP.md`](PHASE-Q-INSIGHT-TO-22STEP-MAP.md)
|
||||
|
||||
Q~Y 검토 결과를 22-step 의 어느 step 에 어떤 부품을 가져올지 매핑해서 정리한 catalog. 섹션 구성:
|
||||
|
||||
- §0: 목적 + 회귀 금지 4 항목 + Archive marker inventory (9 개)
|
||||
- §1: SoT read result + 22 Step status snapshot
|
||||
- §2: Salvage chained + new-make backend axes
|
||||
- §3: Reference / carve-out
|
||||
- §4: audit §1 lens column 정정
|
||||
- §5: Module duplication cleanup
|
||||
|
||||
각 § cell 이 IMP 이슈로 1-to-1 분해됨.
|
||||
|
||||
---
|
||||
|
||||
## 4. IMP 이슈 = INSIGHT-MAP § cell 의 execution unit
|
||||
|
||||
**초기 28 개 (2026-05-12 한 번에 생성, #1~#28)**:
|
||||
|
||||
| INSIGHT-MAP § | 이슈 |
|
||||
|---|---|
|
||||
| §2 (Salvage chained + new-make backend) | #1~#11 (IMP-01~11: A-1~A-6, B-1~B-4, D-1, D-2) |
|
||||
| §3 (Reference / carve-out) | #12~#20 (IMP-12~20: A-3/A-4, B-2, AI fallback, frame contract 등) |
|
||||
| §4 (audit §1 lens column 정정) | #21~#25 (IMP-21~25: G2, I6, J5, K6, L5) |
|
||||
| §5 (Module duplication cleanup) | #26~#28 (IMP-26~28: J3, K5, L4) |
|
||||
|
||||
**모든 IMP 이슈 본문에 표준 anchor**:
|
||||
```
|
||||
**관련 step**: Phase Z 22-step 좌표
|
||||
**source**: INSIGHT-MAP §X (Q~Y 부품 출처)
|
||||
**priority**: ↑ high / medium / ↓ low
|
||||
**scope**: 구체 작업
|
||||
**guardrails**: 깨면 안 되는 contract
|
||||
```
|
||||
|
||||
**이후 추가된 이슈** (모두 source 명시):
|
||||
|
||||
| 이슈 | source | 의미 |
|
||||
|---|---|---|
|
||||
| #38~#41 (IMP-29~32) | IMP-05 §5 defer + Codex 분석 | V4 fallback 후 frontend bridge / AI adaptation 등 |
|
||||
| #42 (IMP-04b) | IMP-04 milestone close 후 잔여 | Catalog 32 frames 확장 |
|
||||
| #43, #44 | MDX 03/04/05 작업 중 발견 | 프론트 작업에서 발견된 새 axis |
|
||||
| #45~#49 | #15 (Step 14 visual_check) decomposition | parent → 5 execution children |
|
||||
| #50 | governance audit | 초반 28 다수 close 후 INTEGRATION-AUDIT-01 |
|
||||
| #51~#54 | #50 audit 의 발견 (F-1~F-5) | follow-up 분리 처리 |
|
||||
| #55 | #20 closed 후 runtime defer | doc-axis closed, runtime 별도 |
|
||||
|
||||
→ 추가 이슈도 모두 (관련 step, source, priority) 좌표로 anchor.
|
||||
|
||||
---
|
||||
|
||||
## 5. orchestrator 의 역할
|
||||
|
||||
이슈 처리의 **disciplined executor**.
|
||||
|
||||
**파일**: [`orchestrator.py`](../../orchestrator.py) (현재 line 수: ~1500)
|
||||
**테스트**: [`tests/orchestrator_unit/`](../../tests/orchestrator_unit/) (현재 94 케이스)
|
||||
|
||||
**6 stage workflow**:
|
||||
1. problem-review — 문제 검토
|
||||
2. simulation-plan — 시뮬 기반 계획 수립 (IMPLEMENTATION_UNITS YAML 강제)
|
||||
3. code-edit — 코드 수정 / 이슈 분기
|
||||
4. test-verify — 테스트 및 검증
|
||||
5. commit-push — 커밋 및 푸쉬
|
||||
6. final-close — 최종 확인 / close
|
||||
|
||||
**원칙**:
|
||||
- Claude (executor) + Codex (verifier) 양쪽 합의 + evidence required
|
||||
- 단일 LLM 의견 X
|
||||
- 매 stage 마다 dual-write (local draft + Gitea comment)
|
||||
- exit report = stage 완료의 binding contract
|
||||
|
||||
**audit-only mode (P4/P4a)**:
|
||||
- 제목에 `[INTEGRATION-AUDIT-*]`/`[AUDIT-ONLY]` 또는 `--audit-only` CLI flag
|
||||
- Stage 3 에서 `src/`, `templates/`, `tests/` 변경 자동 reject (deterministic git diff guard)
|
||||
- Stage 5 commit 범위 = `docs/architecture/INTEGRATION-AUDIT-*.md` + `BACKLOG.md` 만 허용
|
||||
- audit 이슈는 fix 안 함 → follow-up 이슈로 분리
|
||||
|
||||
---
|
||||
|
||||
## 6. Audit cycle (meta-governance)
|
||||
|
||||
이슈 진행으로 인한 누적 drift / 충돌 / 하드코딩 / 매핑 누락을 주기적으로 검증.
|
||||
|
||||
**audit 자체는 코드 안 만짐**. 발견 사항은 별도 이슈로 분리해서 일반 workflow 로 처리.
|
||||
|
||||
**현재까지**:
|
||||
- #50 INTEGRATION-AUDIT-01 (closed 2026-05-19)
|
||||
- 산출: [`INTEGRATION-AUDIT-01-REPORT.md`](INTEGRATION-AUDIT-01-REPORT.md) + [`INTEGRATION-AUDIT-01-MATRIX.md`](INTEGRATION-AUDIT-01-MATRIX.md)
|
||||
- 발견 F-1~F-5 → #51~#54 로 분리 (모두 closed)
|
||||
|
||||
**다음 audit 시점 trigger**:
|
||||
- 닫힌 IMP 이슈가 일정 수 누적될 때 (5+ 연속)
|
||||
- debug.json schema / layout / frame contract / router / visual_check_passed 의미가 바뀔 때
|
||||
- 새 parent axis 진입 직전 (예: #19 → #20 → ...)
|
||||
- 큰 feature 축 (#42 catalog 확장 / #38~#41 frontend bridge) 완료 후
|
||||
|
||||
---
|
||||
|
||||
## 7. 도착점 도달 기준
|
||||
|
||||
다음이 모두 작동해야 destination 도달:
|
||||
|
||||
- [ ] 22-step pipeline end-to-end (Step 0~22 모두 contract 준수, 회귀 0)
|
||||
- [ ] AI 가 frame 을 zone fit 기준으로 평가 → 안 맞으면 reject
|
||||
- [ ] reject 후 AI 가 zone 에 맞춰 frame 생성
|
||||
- [ ] 하드코딩 0 (sample-specific 코드 없음 — anti-hardcoding mechanical check 통과)
|
||||
- [ ] 모든 IMP 이슈 backlog 의 closed / documented (deferred) / pending 분류가 [`PHASE-Z-IMPLEMENTATION-ISSUE-BACKLOG.md`](PHASE-Z-IMPLEMENTATION-ISSUE-BACKLOG.md) 와 code reality 일치
|
||||
|
||||
---
|
||||
|
||||
## 8. 자주 헷갈리는 것들 (anti-patterns — 하지 말 것)
|
||||
|
||||
| 잘못된 framing | 옳은 framing |
|
||||
|---|---|
|
||||
| "Phase Q~Y heritage 를 보존한다" | Q~Y 는 부품 창고. 갭에 필요한 것만 선택적 참조 |
|
||||
| "MDX 03 잘 만들면 끝" | 재사용 가능한 pipeline contract 가 목표. 특정 샘플 최적화 X |
|
||||
| "audit 가 발견하면 그 자리에서 고친다" | follow-up 이슈로 분리. audit 자체는 코드 안 만짐 |
|
||||
| "Claude 가 좋다고 하면 OK" | Claude + Codex 합의 + evidence 필수 |
|
||||
| "이슈 본문은 참고일뿐" | 본문의 (관련 step, source, scope, guardrails) 가 binding anchor |
|
||||
| "Phase R / R' / Q 의 path 로 돌아가도 됨" | 회귀 금지선 4 항목 (INSIGHT-MAP §0) 절대 위반 X |
|
||||
| "destination 외 추가 기능도 욕심내자" | 22-step + AI frame generation 까지가 목표. 그 이상은 별도 결정 |
|
||||
| "문서에 박힌 dormant 항목은 자동 실행 안 됨" | L3 registry [`DORMANT-TRIGGERS.yaml`](DORMANT-TRIGGERS.yaml) + `scripts/check_dormant_triggers.py` 가 orchestrator Stage 4→5 transition 에서 informational alert 로 발화 (closed 이슈 #16/#17/#18/#19/#20 의 trigger-on-X contract) |
|
||||
|
||||
---
|
||||
|
||||
## 9. 핵심 참조 문서 한 곳에
|
||||
|
||||
| 문서 | 역할 |
|
||||
|---|---|
|
||||
| [`PROJECT-INTENT-AND-GOVERNANCE.md`](PROJECT-INTENT-AND-GOVERNANCE.md) | **이 문서** — 왜/무엇을 |
|
||||
| [`PHASE-Q-INSIGHT-TO-22STEP-MAP.md`](PHASE-Q-INSIGHT-TO-22STEP-MAP.md) | INSIGHT-MAP — Q~Y → Z 매핑 catalog |
|
||||
| [`PHASE-Z-PIPELINE-OVERVIEW.md`](PHASE-Z-PIPELINE-OVERVIEW.md) | 22-step pipeline 정의 |
|
||||
| [`PHASE-Z-PIPELINE-STATUS-BOARD.md`](PHASE-Z-PIPELINE-STATUS-BOARD.md) | 22-step 현재 status |
|
||||
| [`PHASE-Z-IMPLEMENTATION-ISSUE-BACKLOG.md`](PHASE-Z-IMPLEMENTATION-ISSUE-BACKLOG.md) | IMP 이슈 backlog (closed/documented/pending) |
|
||||
| [`PHASE-Z-ROADMAP.md`](PHASE-Z-ROADMAP.md) | 진행 로드맵 |
|
||||
| [`INTEGRATION-AUDIT-01-REPORT.md`](INTEGRATION-AUDIT-01-REPORT.md) | 첫 audit 사이클 결과 |
|
||||
| [`../../orchestrator.py`](../../orchestrator.py) | disciplined executor (Claude + Codex 합의 workflow) |
|
||||
| [`../../CLAUDE.md`](../../CLAUDE.md) | AI 가 코드 작업할 때 따를 규칙 |
|
||||
|
||||
---
|
||||
|
||||
## 10. 한 줄 요약
|
||||
|
||||
> **Phase Z 가 "22-step pipeline + AI zone-fit frame generation" 까지 작동하는 것이 destination. Z 구현의 갭은 Phase Q~Y 를 부품 창고로 보고 선택적으로 참조해서 메움. INSIGHT-MAP 이 그 catalog, IMP 이슈가 execution unit. orchestrator 가 Claude + Codex 합의 + evidence 로 disciplined 하게 처리. INTEGRATION-AUDIT 가 주기적으로 누적 정합성 검증, 발견은 follow-up 이슈로 분리. 도착점은 22-step + AI frame generation 까지이고 그 이상은 별도 결정.**
|
||||
+149
-3
@@ -607,6 +607,32 @@ RULE 7: No hardcoding. RULE 8: AI finds 1px first. RULE 9: LLM classifies, code
|
||||
RULE 10: Don't uncritically accept. RULE 11: Checkpoint. RULE 12: Full paths. RULE 13: Anchor sync.
|
||||
PZ-1: AI=0 normal. PZ-2: 1turn=1step. PZ-3: No speculative. PZ-4: No silent shrink.
|
||||
|
||||
=== COMMENT FORMAT (P5b 2026-05-20 — STRICT, OVERRIDES ALL STAGE-SPECIFIC BODY RULES) ===
|
||||
The FIRST non-empty line of EVERY Gitea comment MUST start with one of:
|
||||
[Claude #N] <stage description>
|
||||
[Codex #N] <stage description>
|
||||
|
||||
This rule applies to ALL stages (Stage 1 ~ Stage 6) and ALL issue types
|
||||
(regular, execution-issue, audit-only). No prefix, no decoration, no banner,
|
||||
no audit anchor before the agent header. Examples:
|
||||
|
||||
CORRECT:
|
||||
[Codex #3] Stage 2 simulation-plan review — IMP-24
|
||||
|
||||
📌 Verification table
|
||||
...
|
||||
|
||||
WRONG (orchestrator detect_agent will fail; stage cannot advance):
|
||||
📌 **[Claude #3] Stage 2 ...**
|
||||
## [Codex #3] Stage 2 ...
|
||||
=== IMPLEMENTATION_UNITS === (header missing entirely)
|
||||
Audit anchor: ... (preface before header)
|
||||
|
||||
This first-line-strict rule OVERRIDES any stage-specific "body MUST contain
|
||||
ONLY" rule (e.g., COMPACT_PLAN_RULE). Those body rules apply AFTER the
|
||||
mandatory first-line agent header. Decorations / banners / anchors go on
|
||||
line 2 or later.
|
||||
|
||||
=== CONSENSUS + REWIND (2026-05-16 lock) ===
|
||||
Final line of every Codex review comment MUST be exactly one of:
|
||||
FINAL_CONSENSUS: YES
|
||||
@@ -855,15 +881,60 @@ def _check_audit_commit_scope():
|
||||
bad.append(path)
|
||||
return bad
|
||||
|
||||
# P5-2 (2026-05-20) — Dormant trigger guard (L3 layer, issue #58).
|
||||
# Closed dormant backlog rows (documented:dormant / documented:deferred) carry
|
||||
# implicit "trigger-on-X" contracts. This helper invokes the standalone
|
||||
# checker (scripts/check_dormant_triggers.py) which reads the machine-readable
|
||||
# registry (docs/architecture/DORMANT-TRIGGERS.yaml) and writes activation
|
||||
# candidates to .orchestrator/dormant_alerts.json.
|
||||
#
|
||||
# Guardrails (per Stage 1 scope-lock) :
|
||||
# - Informational only. Returns the alert list; orchestrator never blocks.
|
||||
# - manual_evidence_required / followup-linked entries are skipped INSIDE
|
||||
# the checker (not duplicated here — registry is single source of truth).
|
||||
# - No LLM call. Deterministic subprocess invocation only.
|
||||
# - Fail-open : any subprocess / json error returns [] (no false positives).
|
||||
def _check_dormant_triggers():
|
||||
"""P5-2 — Run scripts/check_dormant_triggers.py and return the alert list.
|
||||
|
||||
Returns: list[dict] of activation-candidate alerts (empty list = no
|
||||
candidates OR script / parse error). Orchestrator never blocks on this."""
|
||||
script_path = Path(PROJECT_DIR) / "scripts" / "check_dormant_triggers.py"
|
||||
if not script_path.exists():
|
||||
return [] # registry / checker not installed yet — fail open
|
||||
try:
|
||||
r = subprocess.run(
|
||||
[sys.executable, str(script_path)],
|
||||
capture_output=True, text=True, encoding="utf-8", errors="replace",
|
||||
cwd=PROJECT_DIR, timeout=30,
|
||||
)
|
||||
if r.returncode != 0:
|
||||
return [] # script error — fail open
|
||||
except Exception:
|
||||
return []
|
||||
alert_path = ORCH_DIR / "dormant_alerts.json"
|
||||
if not alert_path.exists():
|
||||
return []
|
||||
try:
|
||||
payload = json.loads(alert_path.read_text(encoding="utf-8"))
|
||||
alerts = payload.get("alerts", [])
|
||||
return alerts if isinstance(alerts, list) else []
|
||||
except Exception:
|
||||
return []
|
||||
|
||||
# P1-5 (2026-05-18) — Stage 2 compact rule (모든 issue 적용).
|
||||
# Stage 2 의 c-role 에 size budget + code snippet 금지 명시. 29 KB plan 차단.
|
||||
COMPACT_PLAN_RULE = """
|
||||
|
||||
COMPACT PLAN REQUIREMENTS (strict):
|
||||
- The FIRST non-empty line of your comment MUST be the agent header
|
||||
([Claude #N] ... or [Codex #N] ...). This is enforced by RULES (P5b 2026-05-20)
|
||||
and OVERRIDES the "body" constraints below. The Stage 2 compact body begins
|
||||
AFTER the first-line agent header — NOT on line 1.
|
||||
- Total Stage 2 plan body MUST be ≤ 5,000 chars (4,000 chars target).
|
||||
- NO code snippets in this comment. Code goes in Stage 3 (code-edit), not Stage 2 plan.
|
||||
References to file:line locations are fine. Inline code blocks are forbidden.
|
||||
- The Stage 2 plan body MUST contain ONLY:
|
||||
- After the first-line agent header, the Stage 2 plan body MUST contain ONLY:
|
||||
a) === IMPLEMENTATION_UNITS === YAML block (units with id/summary/files/tests/estimate_lines)
|
||||
b) Brief per-unit rationale (≤ 3 lines per unit, no full code)
|
||||
c) Out-of-scope notes
|
||||
@@ -900,7 +971,22 @@ AUDIT-ONLY MODE (this issue is an integration audit / report-only):
|
||||
status_integrity / report_assembly / followup_proposal). Each unit's tests: field MUST list verification
|
||||
commands or report artifacts (NOT pytest tests:[] which the orchestrator rejects).
|
||||
- Stage 5 commit = only audit report files. pipeline run artifacts under data/runs/ or .orchestrator/
|
||||
are evidence-only and must NOT be staged for commit."""
|
||||
are evidence-only and must NOT be staged for commit.
|
||||
- COMMENT FORMAT (CRITICAL — orchestrator detect_agent is first-line strict, P0-1):
|
||||
The FIRST non-empty line of every Gitea comment MUST be exactly one of:
|
||||
[Claude #<N>] <stage description>
|
||||
[Codex #<N>] <stage description>
|
||||
Audit anchor citation, banners, prefaces of any kind MUST appear AFTER the first line
|
||||
(line 2 or later). If you put `Audit anchor:` or any other preface BEFORE the [Claude #N] /
|
||||
[Codex #N] header, the orchestrator will fail to detect the agent and the stage cannot
|
||||
advance — your work will be discarded and re-attempted with token waste.
|
||||
Correct example:
|
||||
[Codex #14] Stage 4 test-verify — INTEGRATION-AUDIT-02
|
||||
|
||||
Audit anchor: This audit verifies pipeline contracts...
|
||||
...
|
||||
FINAL_CONSENSUS: YES
|
||||
"""
|
||||
|
||||
|
||||
def build_context_pack(n, title, body, sid, agent, rnd, start_cnt, compact=None):
|
||||
@@ -1272,7 +1358,40 @@ def run_stage(n, title, body, sid):
|
||||
last = comments[-1]["body"]
|
||||
is_codex = detect_agent(last) == "codex"
|
||||
if not is_codex:
|
||||
log(" Codex 응답 미감지 — continuing")
|
||||
log(f" Codex 응답 미감지 — first line: {last.lstrip().splitlines()[0][:80]!r}" if last and last.strip() else " Codex 응답 미감지 — empty body")
|
||||
# P5b (2026-05-20) — detect_agent None 시 supplement 가드.
|
||||
# 범위 변경: audit-only 제한 해제 — 모든 issue 에서 작동 (#24 같은 일반 이슈 silent loop fix).
|
||||
# Throttle: 현재 stage 안에 이미 N (=2) 회 supplement 가 누적되면 stop + user-action-required.
|
||||
# 직전 N supplement 가 박혀도 LLM 이 또 위반하면 4 번째 round 부터는 hard stop.
|
||||
SUPP_MAX = 2
|
||||
SUPP_MARKER = "⚠️ **[Orchestrator]** Agent header missing"
|
||||
stage_cmts = comments[start_cnt:]
|
||||
supp_count = sum(1 for c in stage_cmts if (c.get("body") or "").lstrip().startswith(SUPP_MARKER))
|
||||
if supp_count >= SUPP_MAX:
|
||||
log(f"⛔ Agent header supplement {supp_count}/{SUPP_MAX} reached — STOP (user action required)")
|
||||
try: gitea(f"issues/{n}/comments", "POST", {"body":
|
||||
f"⛔ **[Orchestrator]** STOP — Stage `{sid}` cannot advance.\n\n"
|
||||
f"`detect_agent` failed {supp_count}+ times in this stage. The LLM is not honoring "
|
||||
f"the first-line agent header contract despite supplements.\n\n"
|
||||
"**Action required (human)**: review last few comments, ensure FIRST non-empty line is "
|
||||
"`[Claude #N]` or `[Codex #N]`, then restart `python -u .\\orchestrator.py --issue {n}`.\n\n"
|
||||
"Orchestrator run is exiting this issue to prevent further token waste."})
|
||||
except: pass
|
||||
return False # exit run_stage → run_issue treats as external close → moves on
|
||||
try: gitea(f"issues/{n}/comments", "POST", {"body":
|
||||
f"{SUPP_MARKER} — orchestrator `detect_agent` could not find "
|
||||
"`[Claude #N]` or `[Codex #N]` on the first non-empty line.\n\n"
|
||||
"**Comment format contract (P5b 2026-05-20, see RULES)**:\n"
|
||||
"The FIRST non-empty line of EVERY Gitea comment (both Claude and Codex, ALL stages) MUST be:\n"
|
||||
" `[Claude #N] <stage description>`\n"
|
||||
" `[Codex #N] <stage description>`\n\n"
|
||||
"No prefix. No decoration. No banner. No audit anchor before the header.\n"
|
||||
"Decorations (`📌`, `##`, `**`, audit anchor, etc.) go on line 2 or later.\n\n"
|
||||
"This rule OVERRIDES any stage-specific 'body MUST contain ONLY' rule (e.g., COMPACT_PLAN_RULE) — "
|
||||
"those body rules apply AFTER the mandatory first-line agent header.\n\n"
|
||||
f"Supplement count for this stage: {supp_count + 1}/{SUPP_MAX}. "
|
||||
f"At {SUPP_MAX}+ violations the orchestrator will hard-stop this issue."})
|
||||
except: pass
|
||||
continue
|
||||
|
||||
status, target = parse_consensus(last)
|
||||
@@ -1402,6 +1521,33 @@ def run_stage(n, title, body, sid):
|
||||
except: pass
|
||||
continue
|
||||
|
||||
# P5-2 (2026-05-20) — Dormant trigger guard (L3 layer, issue #58).
|
||||
# Stage 4 (test-verify) PASS → run dormant trigger checker against the
|
||||
# current change surface. If alerts written, post INFORMATIONAL supplement
|
||||
# comment. NEVER blocks Stage 5 entry (checker is exit 0; helper fail-open).
|
||||
# Audit-only issues skip — their change surface is restricted to audit docs,
|
||||
# which the registry does not watch.
|
||||
if sid == "test-verify" and not _audit_mode(title):
|
||||
alerts = _check_dormant_triggers()
|
||||
if alerts:
|
||||
log(f"ℹ️ Dormant trigger guard: {len(alerts)} activation candidate(s) detected (informational)")
|
||||
try: gitea(f"issues/{n}/comments", "POST", {"body":
|
||||
"ℹ️ **[Orchestrator]** Dormant trigger guard — informational alert (does NOT block Stage 5).\n\n"
|
||||
"The following closed dormant backlog axes have changed-file evidence matching their "
|
||||
"activation triggers. Registry: `docs/architecture/DORMANT-TRIGGERS.yaml`. "
|
||||
"Alert artifact: `.orchestrator/dormant_alerts.json`.\n\n" +
|
||||
"\n".join(
|
||||
f"- **#{a.get('issue')}** {a.get('title')} → "
|
||||
f"`{(a.get('on_trigger') or {}).get('action', '?')}` "
|
||||
f"({len(((a.get('match') or {}).get('files')) or [])} file(s))"
|
||||
for a in alerts[:10]
|
||||
) +
|
||||
("\n - ... (truncated)" if len(alerts) > 10 else "") + "\n\n"
|
||||
"Recommended next step : open a follow-up issue using the `template:` field in the "
|
||||
"registry, OR acknowledge in the next stage comment. Stage 5 proceeds regardless."})
|
||||
except: pass
|
||||
# Never `continue` — checker is informational only (Stage 1 guardrail).
|
||||
|
||||
log(f"✅ {si['label']} — YES (evidence verified)")
|
||||
# stage 완료 = unit counter + remaining tracker 모두 reset
|
||||
update_issue_state(n, continue_same_count=0, last_remaining_units=None)
|
||||
|
||||
@@ -0,0 +1,191 @@
|
||||
"""Dormant trigger guard — L3 machine-readable check (issue #58, P5-2).
|
||||
|
||||
Reads docs/architecture/DORMANT-TRIGGERS.yaml, scans the changed-file surface
|
||||
(working tree via `git status --porcelain` + recent commit via
|
||||
`git diff HEAD~1..HEAD --name-only`), and writes any matching activation
|
||||
candidates to .orchestrator/dormant_alerts.json.
|
||||
|
||||
Guardrails (per Stage 1 scope-lock) :
|
||||
- Informational only. Exit code is ALWAYS 0 — orchestrator never blocks on alerts.
|
||||
- manual_evidence_required entries are skipped (require human gate).
|
||||
- followup_issue entries are skipped (already tracked by the open follow-up).
|
||||
- No LLM call. Deterministic file-pattern + content-pattern matching only.
|
||||
- No hardcoding : the registry yaml is the single source of truth.
|
||||
|
||||
Run :
|
||||
python scripts/check_dormant_triggers.py
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
import yaml
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parent.parent
|
||||
REGISTRY_PATH = REPO_ROOT / "docs" / "architecture" / "DORMANT-TRIGGERS.yaml"
|
||||
ALERT_OUT_PATH = REPO_ROOT / ".orchestrator" / "dormant_alerts.json"
|
||||
|
||||
|
||||
def load_registry(path: Path = REGISTRY_PATH) -> list[dict]:
|
||||
if not path.exists():
|
||||
return []
|
||||
with path.open("r", encoding="utf-8") as f:
|
||||
data = yaml.safe_load(f) or []
|
||||
if not isinstance(data, list):
|
||||
raise ValueError(f"{path} must be a YAML list of entries.")
|
||||
return data
|
||||
|
||||
|
||||
def _git_lines(args: list[str]) -> list[str]:
|
||||
try:
|
||||
out = subprocess.run(
|
||||
["git"] + args,
|
||||
cwd=str(REPO_ROOT),
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=20,
|
||||
check=False,
|
||||
)
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
return []
|
||||
if out.returncode != 0:
|
||||
return []
|
||||
return [ln for ln in out.stdout.splitlines() if ln.strip()]
|
||||
|
||||
|
||||
def collect_changed_files() -> list[str]:
|
||||
files: set[str] = set()
|
||||
for ln in _git_lines(["status", "--porcelain"]):
|
||||
path = ln[3:].strip() if len(ln) >= 4 else ln.strip()
|
||||
if "->" in path:
|
||||
path = path.split("->", 1)[1].strip()
|
||||
path = path.strip('"')
|
||||
if path:
|
||||
files.add(path.replace("\\", "/"))
|
||||
for ln in _git_lines(["diff", "HEAD~1..HEAD", "--name-only"]):
|
||||
if ln.strip():
|
||||
files.add(ln.strip().replace("\\", "/"))
|
||||
return sorted(files)
|
||||
|
||||
|
||||
def _glob_to_regex(pat: str) -> str:
|
||||
"""Translate a posix-style glob with ``**`` to an anchored regex.
|
||||
|
||||
``**/`` matches zero or more directory levels (so ``src/**/*.py`` matches
|
||||
both ``src/adapter.py`` and ``src/foo/adapter.py``). ``*`` and ``?`` do
|
||||
NOT cross directory separators. Mirrors common ``.gitignore``-style
|
||||
semantics; ``fnmatch.fnmatch`` alone cannot express this.
|
||||
"""
|
||||
out: list[str] = []
|
||||
i = 0
|
||||
n = len(pat)
|
||||
while i < n:
|
||||
if pat[i : i + 3] == "**/":
|
||||
out.append("(?:.*/)?")
|
||||
i += 3
|
||||
elif pat[i : i + 2] == "**":
|
||||
out.append(".*")
|
||||
i += 2
|
||||
elif pat[i] == "*":
|
||||
out.append("[^/]*")
|
||||
i += 1
|
||||
elif pat[i] == "?":
|
||||
out.append("[^/]")
|
||||
i += 1
|
||||
else:
|
||||
out.append(re.escape(pat[i]))
|
||||
i += 1
|
||||
return "^" + "".join(out) + "$"
|
||||
|
||||
|
||||
def _glob_match(path: str, patterns: list[str]) -> bool:
|
||||
for pat in patterns:
|
||||
if re.match(_glob_to_regex(pat), path):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _content_match(file_path: Path, patterns: list[str]) -> list[str]:
|
||||
if not patterns or not file_path.exists() or not file_path.is_file():
|
||||
return []
|
||||
try:
|
||||
text = file_path.read_text(encoding="utf-8", errors="replace")
|
||||
except OSError:
|
||||
return []
|
||||
hits = []
|
||||
for pat in patterns:
|
||||
try:
|
||||
if re.search(pat, text):
|
||||
hits.append(pat)
|
||||
except re.error:
|
||||
if pat in text:
|
||||
hits.append(pat)
|
||||
return hits
|
||||
|
||||
|
||||
def check_entry(entry: dict, changed: list[str]) -> dict | None:
|
||||
trig = entry.get("trigger") or {}
|
||||
if trig.get("manual_evidence_required"):
|
||||
return None
|
||||
if entry.get("followup_issue"):
|
||||
return None
|
||||
file_patterns = trig.get("file_patterns") or []
|
||||
content_patterns = trig.get("content_patterns") or []
|
||||
if not file_patterns:
|
||||
return None
|
||||
matched_files = [p for p in changed if _glob_match(p, file_patterns)]
|
||||
if not matched_files:
|
||||
return None
|
||||
if content_patterns:
|
||||
hits: list[dict] = []
|
||||
for mf in matched_files:
|
||||
hit_patterns = _content_match(REPO_ROOT / mf, content_patterns)
|
||||
if hit_patterns:
|
||||
hits.append({"file": mf, "patterns": hit_patterns})
|
||||
if not hits:
|
||||
return None
|
||||
match_info = {"files": [h["file"] for h in hits], "content_hits": hits}
|
||||
else:
|
||||
match_info = {"files": matched_files, "content_hits": []}
|
||||
return {
|
||||
"issue": entry.get("issue"),
|
||||
"title": entry.get("title"),
|
||||
"doc": entry.get("doc"),
|
||||
"status": entry.get("status"),
|
||||
"on_trigger": entry.get("on_trigger"),
|
||||
"match": match_info,
|
||||
}
|
||||
|
||||
|
||||
def write_alerts(alerts: list[dict], path: Path = ALERT_OUT_PATH) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
payload = {
|
||||
"generated_at": datetime.now(timezone.utc).isoformat(),
|
||||
"registry": str(REGISTRY_PATH.relative_to(REPO_ROOT)).replace("\\", "/"),
|
||||
"alerts": alerts,
|
||||
}
|
||||
path.write_text(json.dumps(payload, indent=2, ensure_ascii=False), encoding="utf-8")
|
||||
|
||||
|
||||
def main() -> int:
|
||||
entries = load_registry()
|
||||
changed = collect_changed_files()
|
||||
alerts = [a for a in (check_entry(e, changed) for e in entries) if a]
|
||||
write_alerts(alerts)
|
||||
if alerts:
|
||||
print(f"[dormant-trigger-guard] {len(alerts)} alert(s) written -> "
|
||||
f"{ALERT_OUT_PATH.relative_to(REPO_ROOT)}")
|
||||
for a in alerts:
|
||||
print(f" - #{a['issue']} {a['title']} (files: {len(a['match']['files'])})")
|
||||
else:
|
||||
print("[dormant-trigger-guard] no dormant trigger alerts on current change surface.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -8,6 +8,7 @@
|
||||
- 블록 CSS의 글씨 크기를 font_hierarchy에 맞게 조정 (프로세스 내 조정)
|
||||
- 콘텐츠는 PipelineContext에서 가져옴 (하드코딩 아님)
|
||||
- 블록은 콘텐츠에 맞게 재구성 (items 수 동적)
|
||||
[legacy Phase R'/Q example — INTEGRATION-AUDIT-01 §10.4]
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
@@ -107,6 +107,7 @@ class TfidfBlockMatcher:
|
||||
text = text.replace("S/W", "SW 소프트웨어")
|
||||
text = text.replace("H/W", "HW 하드웨어")
|
||||
text = re.sub(r'\bDX\b', 'DX 디지털전환', text)
|
||||
# [legacy Phase R'/Q example — INTEGRATION-AUDIT-01 §10.4]
|
||||
text = re.sub(r'\bBIM\b', 'BIM 건설정보모델링', text)
|
||||
text = text.replace("(", " ").replace(")", " ")
|
||||
text = text.replace("[", " ").replace("]", " ")
|
||||
|
||||
+10
-20
@@ -20,9 +20,10 @@ import re
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import yaml
|
||||
from jinja2 import Environment, FileSystemLoader
|
||||
|
||||
from src import catalog as _catalog_mod
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# 템플릿 디렉토리
|
||||
@@ -101,32 +102,18 @@ RELATION_CATEGORY_MAP: dict[str, list[str]] = {
|
||||
|
||||
|
||||
# ══════════════════════════════════════
|
||||
# 카탈로그 로딩 (mtime 캐싱)
|
||||
# 카탈로그 로딩 (IMP-27: src.catalog 공유 로더 위임)
|
||||
# ══════════════════════════════════════
|
||||
|
||||
_catalog_cache: dict[str, Any] = {"data": None, "mtime": 0}
|
||||
|
||||
|
||||
def _load_catalog() -> list[dict]:
|
||||
"""catalog.yaml 로드 (mtime 캐싱)."""
|
||||
path = TEMPLATES_DIR / "catalog.yaml"
|
||||
mtime = path.stat().st_mtime
|
||||
if _catalog_cache["data"] is not None and _catalog_cache["mtime"] == mtime:
|
||||
return _catalog_cache["data"]
|
||||
|
||||
data = yaml.safe_load(path.read_text(encoding="utf-8"))
|
||||
blocks = data.get("blocks", [])
|
||||
_catalog_cache["data"] = blocks
|
||||
_catalog_cache["mtime"] = mtime
|
||||
return blocks
|
||||
"""catalog.yaml blocks list (IMP-27: shared loader delegation)."""
|
||||
return _catalog_mod.load_blocks()
|
||||
|
||||
|
||||
def _get_block_by_id(block_id: str) -> dict | None:
|
||||
"""블록 ID로 카탈로그 엔트리 조회."""
|
||||
for b in _load_catalog():
|
||||
if b["id"] == block_id:
|
||||
return b
|
||||
return None
|
||||
"""블록 ID로 카탈로그 엔트리 조회 (IMP-27: shared loader delegation)."""
|
||||
return _catalog_mod.get_block_by_id(block_id)
|
||||
|
||||
|
||||
# ══════════════════════════════════════
|
||||
@@ -399,6 +386,7 @@ _SAMPLE_DATA: dict[str, dict[str, Any]] = {
|
||||
"center_label": "DX",
|
||||
"center_sub": "디지털 전환",
|
||||
"items": [
|
||||
# [legacy Phase R'/Q example — INTEGRATION-AUDIT-01 §10.4]
|
||||
{"label": "BIM", "color": "#ff6b35"},
|
||||
{"label": "GIS", "color": "#00d4aa"},
|
||||
{"label": "DT", "color": "#ffd700"},
|
||||
@@ -406,6 +394,7 @@ _SAMPLE_DATA: dict[str, dict[str, Any]] = {
|
||||
},
|
||||
"keyword-circle-row": {
|
||||
"keywords": [
|
||||
# [legacy Phase R'/Q example — INTEGRATION-AUDIT-01 §10.4]
|
||||
{"letter": "B", "label": "BIM", "description": "건물정보모델링"},
|
||||
{"letter": "G", "label": "GIS", "description": "지리정보시스템"},
|
||||
{"letter": "D", "label": "DX", "description": "디지털 전환"},
|
||||
@@ -432,6 +421,7 @@ _SAMPLE_DATA: dict[str, dict[str, Any]] = {
|
||||
"right_title": "개선",
|
||||
"rows": [
|
||||
{"left": "수작업", "center": "프로세스", "right": "자동화"},
|
||||
# [legacy Phase R'/Q example — INTEGRATION-AUDIT-01 §10.4]
|
||||
{"left": "2D 도면", "center": "설계 도구", "right": "3D BIM"},
|
||||
],
|
||||
},
|
||||
|
||||
+7
-32
@@ -5,24 +5,18 @@ AI에게 불가능한 선택지를 주지 않는다 (Beautiful.ai 원칙).
|
||||
|
||||
주요 함수:
|
||||
- select_block_candidates(): topic + 컨테이너 → 물리적으로 가능한 후보 2-4개
|
||||
- load_catalog(): catalog.yaml 로딩 + 캐싱
|
||||
- load_catalog(): catalog.yaml 로딩 + 캐싱 (IMP-27: src.catalog 공유 로더 위임)
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import yaml
|
||||
|
||||
from src import catalog as _catalog_mod
|
||||
from src.space_allocator import ContainerSpec, HEIGHT_COST_ORDER
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
CATALOG_PATH = Path("templates/catalog.yaml")
|
||||
_catalog_cache: dict | None = None
|
||||
_catalog_mtime: float = 0.0
|
||||
|
||||
|
||||
# ──────────────────────────────────────
|
||||
# relation_type → 블록 카테고리 매핑 (Napkin.ai 방식)
|
||||
@@ -52,35 +46,16 @@ BLOCKS_FORCING_FORMAT_CHANGE = {
|
||||
|
||||
|
||||
# ──────────────────────────────────────
|
||||
# catalog.yaml 로딩 (mtime 캐시)
|
||||
# catalog.yaml 로딩 (IMP-27: src.catalog 공유 로더 위임)
|
||||
# ──────────────────────────────────────
|
||||
def load_catalog() -> dict:
|
||||
"""catalog.yaml을 로딩한다. mtime 기반 캐싱."""
|
||||
global _catalog_cache, _catalog_mtime
|
||||
|
||||
if not CATALOG_PATH.exists():
|
||||
logger.error(f"catalog.yaml 미발견: {CATALOG_PATH}")
|
||||
return {"blocks": []}
|
||||
|
||||
current_mtime = CATALOG_PATH.stat().st_mtime
|
||||
if _catalog_cache is not None and current_mtime == _catalog_mtime:
|
||||
return _catalog_cache
|
||||
|
||||
with open(CATALOG_PATH, encoding="utf-8") as f:
|
||||
_catalog_cache = yaml.safe_load(f)
|
||||
_catalog_mtime = current_mtime
|
||||
|
||||
block_count = len(_catalog_cache.get("blocks", []))
|
||||
logger.info(f"[Q-2] catalog.yaml 로딩: {block_count}개 블록")
|
||||
return _catalog_cache
|
||||
"""catalog.yaml root dict (IMP-27: shared loader delegation)."""
|
||||
return _catalog_mod.load_root_catalog()
|
||||
|
||||
|
||||
def _get_block_by_id(block_id: str, catalog: dict) -> dict | None:
|
||||
"""catalog에서 블록 ID로 검색."""
|
||||
for block in catalog.get("blocks", []):
|
||||
if block.get("id") == block_id:
|
||||
return block
|
||||
return None
|
||||
"""catalog-injected 블록 ID 조회 (IMP-27: shared loader delegation)."""
|
||||
return _catalog_mod.get_block_by_id(block_id, catalog)
|
||||
|
||||
|
||||
# ──────────────────────────────────────
|
||||
|
||||
@@ -0,0 +1,76 @@
|
||||
"""IMP-27: Shared catalog.yaml loader (single file-read + mtime cache).
|
||||
|
||||
Phase Q evolution 중 block_reference, block_selector, renderer 가 각각 templates/
|
||||
catalog.yaml 을 읽고 mtime 캐시하던 중복을 한 곳으로 통합한다. call-site
|
||||
signature 는 그대로 유지되며, 각 wrapper 는 본 모듈의 결과를 자신이 약속하는
|
||||
형태(list[dict] / root dict / id→path projection)로 변환만 수행한다.
|
||||
|
||||
Functions:
|
||||
load_root_catalog() -> dict : raw catalog dict (matches block_selector contract)
|
||||
load_blocks() -> list[dict] : root_catalog.get("blocks", []) projection
|
||||
get_block_by_id(block_id, catalog=None) -> dict | None
|
||||
get_catalog_mtime() -> float : current cached mtime (renderer projection key)
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
import yaml
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
CATALOG_PATH = Path(__file__).parent.parent / "templates" / "catalog.yaml"
|
||||
|
||||
_catalog_cache: dict | None = None
|
||||
_catalog_mtime: float = 0.0
|
||||
|
||||
|
||||
def load_root_catalog() -> dict:
|
||||
"""Load templates/catalog.yaml as root dict, with mtime caching.
|
||||
|
||||
Missing file → logs warning and returns ``{"blocks": []}`` (matches the
|
||||
pre-IMP-27 behavior of block_selector.load_catalog and renderer._load_catalog_map).
|
||||
"""
|
||||
global _catalog_cache, _catalog_mtime
|
||||
|
||||
if not CATALOG_PATH.exists():
|
||||
logger.warning(f"catalog.yaml 미발견: {CATALOG_PATH}")
|
||||
return {"blocks": []}
|
||||
|
||||
current_mtime = CATALOG_PATH.stat().st_mtime
|
||||
if _catalog_cache is not None and current_mtime == _catalog_mtime:
|
||||
return _catalog_cache
|
||||
|
||||
with open(CATALOG_PATH, encoding="utf-8") as f:
|
||||
_catalog_cache = yaml.safe_load(f)
|
||||
_catalog_mtime = current_mtime
|
||||
|
||||
block_count = len((_catalog_cache or {}).get("blocks", []))
|
||||
logger.info(f"[catalog] load: {block_count} blocks")
|
||||
return _catalog_cache
|
||||
|
||||
|
||||
def load_blocks() -> list[dict]:
|
||||
"""Return blocks list (= root_catalog.get('blocks', []))."""
|
||||
return load_root_catalog().get("blocks", [])
|
||||
|
||||
|
||||
def get_block_by_id(block_id: str, catalog: dict | None = None) -> dict | None:
|
||||
"""Locate a block entry by id.
|
||||
|
||||
``catalog=None`` → uses shared loader. caller-supplied catalog dict is
|
||||
accepted as-is so the existing block_selector contract (catalog-injected)
|
||||
keeps working unchanged.
|
||||
"""
|
||||
if catalog is None:
|
||||
catalog = load_root_catalog()
|
||||
for block in catalog.get("blocks", []):
|
||||
if block.get("id") == block_id:
|
||||
return block
|
||||
return None
|
||||
|
||||
|
||||
def get_catalog_mtime() -> float:
|
||||
"""Current cached mtime (renderer projection caches key off this)."""
|
||||
return _catalog_mtime
|
||||
@@ -14,6 +14,26 @@ class Settings(BaseSettings):
|
||||
slide_width: int = 1280
|
||||
slide_height: int = 720
|
||||
|
||||
# IMP-33 u1 — AI fallback policy. Fallback-path only; normal path AI=0.
|
||||
# Defaults locked by Stage 2 plan; do NOT inline literals downstream.
|
||||
ai_fallback_enabled: bool = False
|
||||
ai_fallback_model: str = "claude-opus-4-6-20250415"
|
||||
ai_fallback_timeout_s: float = 60.0
|
||||
ai_fallback_max_retries: int = 3
|
||||
ai_fallback_backoff_base_s: float = 1.0
|
||||
ai_fallback_backoff_cap_s: float = 8.0
|
||||
ai_fallback_backoff_jitter: float = 0.3
|
||||
ai_fallback_budget_per_run: int = 10
|
||||
ai_fallback_circuit_breaker_threshold: int = 5
|
||||
|
||||
# IMP-46 u5 — auto-cache flag. When True, `save_proposal` bypasses the
|
||||
# `user_approved` gate only (`visual_check_passed` is never bypassed).
|
||||
# Default OFF preserves the dual-gate contract; the CLI flag
|
||||
# `--auto-cache` in `src/phase_z2_pipeline.py` mutates this setting at
|
||||
# parse time. Downstream callers MUST source the flag from Settings,
|
||||
# never inline literals.
|
||||
ai_fallback_auto_cache: bool = False
|
||||
|
||||
model_config = {"env_file": ".env", "env_file_encoding": "utf-8"}
|
||||
|
||||
|
||||
|
||||
+4
-37
@@ -8,9 +8,7 @@ Kei API 필수. fallback 없음. 성공할 때까지 무한 재시도.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
from typing import Any
|
||||
|
||||
import anthropic
|
||||
@@ -18,10 +16,14 @@ import httpx
|
||||
|
||||
from src.config import settings
|
||||
from src.design_director import BLOCK_SLOTS
|
||||
from src.json_utils import parse_json as _parse_json
|
||||
from src.sse_utils import stream_sse_tokens
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# [legacy Phase R'/Q examples — INTEGRATION-AUDIT-01 §10.4]
|
||||
# (sample-text literals at L43-L44 / L67 inside the EDITOR_PROMPT string below
|
||||
# — "건설산업 디지털화", "BIM 전면 도입", "DX와 BIM 개념" preserved verbatim)
|
||||
EDITOR_PROMPT = """당신은 도메인 전문가이자 콘텐츠 편집자이다.
|
||||
원본 콘텐츠의 핵심 내용을 유지하면서 각 블록의 슬롯에 맞게 텍스트를 정리한다.
|
||||
|
||||
@@ -438,38 +440,3 @@ async def fill_candidates(
|
||||
logger.warning(f"[Phase P] 꼭지 {tid}: 텍스트 편집 파싱 실패")
|
||||
|
||||
return candidates
|
||||
|
||||
|
||||
def _parse_json(text: str) -> dict[str, Any] | None:
|
||||
"""텍스트에서 JSON을 추출한다.
|
||||
|
||||
Kei API가 마크다운 리스트 접두사(- )를 붙여 응답하는 경우에도 처리.
|
||||
"""
|
||||
# 전처리: 각 줄 앞의 마크다운 리스트 접두사(- ) 제거
|
||||
lines = text.split("\n")
|
||||
cleaned_lines = []
|
||||
for line in lines:
|
||||
stripped = line.lstrip()
|
||||
if stripped.startswith("- "):
|
||||
cleaned_lines.append(stripped[2:])
|
||||
elif stripped.startswith("* "):
|
||||
cleaned_lines.append(stripped[2:])
|
||||
else:
|
||||
cleaned_lines.append(stripped)
|
||||
cleaned = "\n".join(cleaned_lines)
|
||||
|
||||
# 원본 먼저 시도 → 클린 버전 시도
|
||||
for target in [text, cleaned]:
|
||||
patterns = [
|
||||
r"```json\s*(.*?)```",
|
||||
r"```\s*(.*?)```",
|
||||
r"(\{.*\})",
|
||||
]
|
||||
for pattern in patterns:
|
||||
match = re.search(pattern, target, re.DOTALL)
|
||||
if match:
|
||||
try:
|
||||
return json.loads(match.group(1).strip())
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
return None
|
||||
|
||||
+3
-37
@@ -5,9 +5,7 @@ Step B: 프리셋 안에서 블록 매핑 + 글자 수 가이드 (Sonnet)
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
@@ -15,6 +13,7 @@ import httpx
|
||||
import yaml
|
||||
|
||||
from src.config import settings
|
||||
from src.json_utils import parse_json as _parse_json
|
||||
from src.sse_utils import stream_sse_tokens
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -29,6 +28,7 @@ BLOCK_SLOTS = {
|
||||
"slot_desc": {
|
||||
"title_ko": "한글 메인 타이틀",
|
||||
"title_en": "영문 서브 타이틀 (없으면 생략)",
|
||||
# [legacy Phase R'/Q example — INTEGRATION-AUDIT-01 §10.4]
|
||||
"breadcrumb": "상위 카테고리 경로 (예: 디지털전환 > BIM)",
|
||||
"bg_image": "배경 이미지 경로",
|
||||
},
|
||||
@@ -965,6 +965,7 @@ def _validate_height_budget(blocks: list[dict], preset: dict) -> list[dict]:
|
||||
for block in blocks_to_remove:
|
||||
blocks.remove(block)
|
||||
|
||||
# [legacy Phase R'/Q example — INTEGRATION-AUDIT-01 §10.4]
|
||||
# 삭제 후 zone_blocks 재구성 (후속 pill-pair/높이 체크에 반영)
|
||||
zone_blocks.clear()
|
||||
for block in blocks:
|
||||
@@ -1064,38 +1065,3 @@ def _validate_height_budget(blocks: list[dict], preset: dict) -> list[dict]:
|
||||
})
|
||||
|
||||
return overflows
|
||||
|
||||
|
||||
def _parse_json(text: str) -> dict[str, Any] | None:
|
||||
"""텍스트에서 JSON을 추출한다.
|
||||
|
||||
Kei API가 마크다운 리스트 접두사(- )를 붙여 응답하는 경우에도 처리.
|
||||
"""
|
||||
# 전처리: 각 줄 앞의 마크다운 리스트 접두사(- ) 제거
|
||||
lines = text.split("\n")
|
||||
cleaned_lines = []
|
||||
for line in lines:
|
||||
stripped = line.lstrip()
|
||||
if stripped.startswith("- "):
|
||||
cleaned_lines.append(stripped[2:])
|
||||
elif stripped.startswith("* "):
|
||||
cleaned_lines.append(stripped[2:])
|
||||
else:
|
||||
cleaned_lines.append(stripped)
|
||||
cleaned = "\n".join(cleaned_lines)
|
||||
|
||||
# 원본 먼저 시도 → 클린 버전 시도
|
||||
for target in [text, cleaned]:
|
||||
patterns = [
|
||||
r"```json\s*(.*?)```",
|
||||
r"```\s*(.*?)```",
|
||||
r"(\{.*\})",
|
||||
]
|
||||
for pattern in patterns:
|
||||
match = re.search(pattern, target, re.DOTALL)
|
||||
if match:
|
||||
try:
|
||||
return json.loads(match.group(1).strip())
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
return None
|
||||
|
||||
@@ -84,6 +84,9 @@ border-radius: 8px, padding: 14px 30px, text-align: center
|
||||
|
||||
def get_layout_rules() -> str:
|
||||
"""Phase S 검증 결과 기반 레이아웃 규칙."""
|
||||
# [legacy Phase R'/Q example — INTEGRATION-AUDIT-01 §10.4]
|
||||
# (sample-text literal "DX와 BIM의 상세 비교" at ~L109 inside the return
|
||||
# string below is preserved verbatim as a documented intentional example)
|
||||
return """
|
||||
## 레이아웃 규칙 (검증 결과 기반 — 반드시 따를 것)
|
||||
|
||||
|
||||
@@ -609,6 +609,7 @@ class SupplementBlock:
|
||||
role: str
|
||||
block_id: str
|
||||
variant: str
|
||||
# [legacy Phase R'/Q example — INTEGRATION-AUDIT-01 §10.4]
|
||||
content_source: str # "popup:DX와 BIM의 구분" 등
|
||||
estimated_height_px: float
|
||||
available_px: float
|
||||
|
||||
@@ -164,6 +164,7 @@ def _preprocess_text(text: str) -> str:
|
||||
text = text.replace("S/W", "SW 소프트웨어")
|
||||
text = text.replace("H/W", "HW 하드웨어")
|
||||
text = re.sub(r'\bDX\b', 'DX 디지털전환', text)
|
||||
# [legacy Phase R'/Q example — INTEGRATION-AUDIT-01 §10.4]
|
||||
text = re.sub(r'\bBIM\b', 'BIM 건설정보모델링', text)
|
||||
|
||||
# 괄호 내용 유지하되 괄호 제거
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
"""JSON 추출 공용 유틸리티.
|
||||
|
||||
Kei / Claude API 응답 텍스트에서 JSON 객체를 추출한다.
|
||||
content_editor, design_director, kei_client, pipeline 공통 헬퍼.
|
||||
|
||||
응답이 마크다운 리스트 접두사("- " / "* ")로 감싸진 경우에도 처리.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from typing import Any
|
||||
|
||||
_JSON_PATTERNS: tuple[str, ...] = (
|
||||
r"```json\s*(.*?)```",
|
||||
r"```\s*(.*?)```",
|
||||
r"(\{.*\})",
|
||||
)
|
||||
|
||||
|
||||
def parse_json(text: str) -> dict[str, Any] | None:
|
||||
"""텍스트에서 JSON을 추출한다.
|
||||
|
||||
Kei API가 마크다운 리스트 접두사(- )를 붙여 응답하는 경우에도 처리.
|
||||
원본 → 리스트 접두사 제거 버전 순서로 fenced JSON / plain fenced / 베어 brace 패턴을
|
||||
차례로 시도한다. 모두 실패하면 None.
|
||||
"""
|
||||
lines = text.split("\n")
|
||||
cleaned_lines: list[str] = []
|
||||
for line in lines:
|
||||
stripped = line.lstrip()
|
||||
if stripped.startswith("- ") or stripped.startswith("* "):
|
||||
cleaned_lines.append(stripped[2:])
|
||||
else:
|
||||
cleaned_lines.append(stripped)
|
||||
cleaned = "\n".join(cleaned_lines)
|
||||
|
||||
for target in (text, cleaned):
|
||||
for pattern in _JSON_PATTERNS:
|
||||
match = re.search(pattern, target, re.DOTALL)
|
||||
if match:
|
||||
try:
|
||||
return json.loads(match.group(1).strip())
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
return None
|
||||
+7
-36
@@ -13,6 +13,7 @@ from typing import Any
|
||||
import httpx
|
||||
|
||||
from src.config import settings
|
||||
from src.json_utils import parse_json as _parse_json
|
||||
from src.sse_utils import stream_sse_tokens
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -53,6 +54,7 @@ KEI_PROMPT = (
|
||||
" 문장을 재작성하지 마라. 원본 문장을 그대로 가져와라.\n"
|
||||
"- **결론 텍스트도 원본 그대로.** 임의로 만들지 마라.\n"
|
||||
"- 원본에 있는 내용을 임의로 제거하거나 다른 의미로 바꾸지 마라.\n"
|
||||
# [legacy Phase R'/Q example — INTEGRATION-AUDIT-01 §10.4]
|
||||
"- 텍스트 재구성이 허용되는 경우는 **빈 공간에 채울 요약(표, 팝업 요약)만**.\n"
|
||||
"- 각 꼭지의 source_hint에 원본의 어떤 부분이 가는지 명시.\n\n"
|
||||
"## 배치 규칙\n"
|
||||
@@ -162,6 +164,7 @@ KEI_PROMPT_B = (
|
||||
" - 원본에 이미지가 참조되면 반드시 [이미지: 제목] 마커를 포함하라.\n"
|
||||
" - 출처가 있으면 포함하라.\n"
|
||||
" - '활용 필요', '구체화 필요' 같은 지시사항을 쓰지 마라. 실제 콘텐츠 항목만 쓰라.\n"
|
||||
# [legacy Phase R'/Q examples — INTEGRATION-AUDIT-01 §10.4]
|
||||
" - 예시: '건설산업(종합산업, 기술 통합 융합), BIM(정보관리 도구, 출처: 국토교통부 2020)'\n"
|
||||
" - 예시: '[이미지: DX와 핵심기술간 상호관계] 다이어그램, GIS 역할(공간 분석). [팝업: DX와 BIM의 구분] 비교표'\n\n"
|
||||
"## 출력 형식 (JSON만)\n"
|
||||
@@ -789,6 +792,10 @@ async def call_kei_final_review(
|
||||
# I-9: Kei 넘침 판단 호출
|
||||
# ──────────────────────────────────────
|
||||
|
||||
# [legacy Phase R'/Q example — INTEGRATION-AUDIT-01 §10.4]
|
||||
# (sample-text literal "Option 2 (핵심 재구성 + 팝업 분리)" inside the
|
||||
# KEI_OVERFLOW_PROMPT triple-quoted string below is preserved verbatim
|
||||
# as a documented intentional example of overflow-judgment output)
|
||||
KEI_OVERFLOW_PROMPT = """당신은 슬라이드 콘텐츠 전문가이다.
|
||||
디자인 팀장이 배치한 블록들이 컨테이너(zone)의 높이 예산을 초과했다.
|
||||
콘텐츠의 중요도와 전달 메시지를 기준으로 어떻게 처리할지 판단하라.
|
||||
@@ -883,42 +890,6 @@ async def call_kei_overflow_judgment(
|
||||
return None
|
||||
|
||||
|
||||
def _parse_json(text: str) -> dict[str, Any] | None:
|
||||
"""텍스트에서 JSON을 추출한다.
|
||||
|
||||
Kei API가 마크다운 리스트 접두사(- )를 붙여 응답하는 경우에도 처리.
|
||||
"""
|
||||
# 전처리: 각 줄 앞의 마크다운 리스트 접두사(- ) 제거
|
||||
# Kei API가 JSON을 마크다운 리스트로 감싸서 응답하는 경우 대응
|
||||
lines = text.split("\n")
|
||||
cleaned_lines = []
|
||||
for line in lines:
|
||||
stripped = line.lstrip()
|
||||
if stripped.startswith("- "):
|
||||
cleaned_lines.append(stripped[2:])
|
||||
elif stripped.startswith("* "):
|
||||
cleaned_lines.append(stripped[2:])
|
||||
else:
|
||||
cleaned_lines.append(stripped)
|
||||
cleaned = "\n".join(cleaned_lines)
|
||||
|
||||
# 원본 + 클린 버전 둘 다 시도
|
||||
for target in [text, cleaned]:
|
||||
patterns = [
|
||||
r"```json\s*(.*?)```",
|
||||
r"```\s*(.*?)```",
|
||||
r"(\{.*\})",
|
||||
]
|
||||
for pattern in patterns:
|
||||
match = re.search(pattern, target, re.DOTALL)
|
||||
if match:
|
||||
try:
|
||||
return json.loads(match.group(1).strip())
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
return None
|
||||
|
||||
|
||||
async def select_best_candidate(
|
||||
topic_results: list[dict[str, Any]],
|
||||
analysis: dict[str, Any],
|
||||
|
||||
@@ -0,0 +1,15 @@
|
||||
"""IMP-33 AI fallback package (fallback path only).
|
||||
|
||||
Module path locked by IMP-31-GATE-AUDIT.md (Stage 1 binding).
|
||||
Normal path AI call count MUST remain 0; this package only executes under
|
||||
classified fallback routes (reject / restructure / overflow). See
|
||||
`feedback_ai_isolation_contract`.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from src.phase_z2_ai_fallback.schema import (
|
||||
AiFallbackProposal,
|
||||
ProposalKind,
|
||||
)
|
||||
|
||||
__all__ = ["AiFallbackProposal", "ProposalKind"]
|
||||
@@ -0,0 +1,243 @@
|
||||
"""IMP-46 u2 + u3 + u5 — Persistent JSON cache backend for AI fallback proposals.
|
||||
|
||||
Replaces the IMP-33 u6 ``NotImplementedError`` stub with a content-addressed
|
||||
store at ``data/frame_cache/{frame_id}/{signature_hash}.json``.
|
||||
|
||||
Key format:
|
||||
|
||||
* ``read_proposal(key)`` / ``save_proposal(key, ...)`` accept a string ``key``
|
||||
of the form ``"{frame_id}::{signature_hash}"``. The two components are
|
||||
parsed inside this module so that upstream callers (router, step 12)
|
||||
remain unaware of the on-disk layout.
|
||||
* ``read_proposal`` on a malformed (legacy) key silently returns ``None``
|
||||
— the IMP-33 u7 router currently passes a legacy ``cache_key`` string,
|
||||
and u4 will switch to the structural form. Until then, all such reads
|
||||
must miss safely (no exception, no false hit).
|
||||
* ``save_proposal`` on a malformed key raises ``ValueError`` (loud, never
|
||||
silent) — writes are gated and must use the structural form.
|
||||
|
||||
Stored payload (one JSON file per (frame_id, signature_hash) pair):
|
||||
|
||||
{
|
||||
"schema_version": 1,
|
||||
"proposal": <AiFallbackProposal.model_dump(mode="json")>,
|
||||
"slide_css": <str | null>,
|
||||
"fingerprints": {"contract_sha": ..., "partial_sha": ..., "catalog_sha": ...}
|
||||
}
|
||||
|
||||
u3 invalidation contract (this module is a *comparator*, not a *computer*):
|
||||
|
||||
* ``save_proposal`` persists the ``fingerprints`` dict supplied by the
|
||||
caller verbatim. Cache.py never computes any fingerprint — the three
|
||||
declared shas (``contract_sha`` / ``partial_sha`` / ``catalog_sha``) are
|
||||
computed by callers from the live contract YAML / partial templates /
|
||||
catalog payloads and handed in. Keeping the computation out of cache.py
|
||||
preserves AI isolation (no Phase Z runtime knowledge in the cache
|
||||
module) and keeps the cache schema-agnostic — additional fingerprint
|
||||
axes can be added without editing cache.py.
|
||||
* ``read_proposal`` accepts an optional ``fingerprints`` kwarg. When
|
||||
supplied, the stored ``fingerprints`` dict must equal the caller's dict
|
||||
exactly (strict equality, NOT subset). Any mismatch — including a key
|
||||
the caller demands but the stored entry lacks, OR a key the stored
|
||||
entry has but the caller does not pass — returns ``None``. Default
|
||||
``fingerprints=None`` performs no comparison (back-compat for legacy
|
||||
callers that have not yet adopted fingerprint-aware lookup).
|
||||
|
||||
Guardrails (locked by Stage 2 plan):
|
||||
|
||||
* Both write gates preserved — ``visual_check_passed=False`` always
|
||||
raises ``AiFallbackCacheGateError`` BEFORE any filesystem touch.
|
||||
``user_approved=False`` also raises by default; the IMP-46 u5
|
||||
``auto_cache=True`` override bypasses ONLY the ``user_approved`` gate
|
||||
(``visual_check_passed`` is never bypassed). Gate violation never
|
||||
silently no-ops.
|
||||
* Missing or corrupt files cause ``read_proposal`` to return ``None`` —
|
||||
the cache is a hint, never a hard dependency. Errors are not propagated
|
||||
to callers because the AI fallback path can always recompute.
|
||||
* ``mkdir(parents=True, exist_ok=True)`` is performed lazily on save.
|
||||
* No Anthropic / MDX / Phase Z runtime imports (AI isolation contract).
|
||||
* Cache root is held as a module-level :data:`CACHE_ROOT` so tests can
|
||||
redirect writes via ``monkeypatch.setattr`` without subclassing.
|
||||
|
||||
u5 auto-cache contract (CLI ``--auto-cache`` + ``settings.ai_fallback_auto_cache``):
|
||||
|
||||
* ``save_proposal(..., auto_cache=True)`` only bypasses the
|
||||
``user_approved`` gate; ``visual_check_passed`` remains mandatory.
|
||||
* ``auto_cache`` is keyword-only and defaults to ``False`` — existing
|
||||
callers (and the test suite) see the original dual-gate behaviour
|
||||
unless they opt in explicitly.
|
||||
* The truth table over ``(visual_check_passed, user_approved, auto_cache)``
|
||||
has eight cells; exactly three succeed:
|
||||
``(True, True, False)``, ``(True, True, True)``, and
|
||||
``(True, False, True)``. Every other cell raises
|
||||
``AiFallbackCacheGateError``.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import pathlib
|
||||
|
||||
from src.phase_z2_ai_fallback.schema import AiFallbackProposal
|
||||
|
||||
|
||||
SCHEMA_VERSION = 1
|
||||
KEY_DELIMITER = "::"
|
||||
CACHE_ROOT: pathlib.Path = pathlib.Path("data/frame_cache")
|
||||
|
||||
|
||||
class AiFallbackCacheGateError(RuntimeError):
|
||||
"""Raised when ``save_proposal`` is called without both IMP-46 gates True."""
|
||||
|
||||
|
||||
def _parse_key(key: str) -> tuple[str, str] | None:
|
||||
"""Parse a ``frame_id::signature_hash`` key. Returns ``None`` if malformed."""
|
||||
if KEY_DELIMITER not in key:
|
||||
return None
|
||||
frame_id, _, signature_hash = key.partition(KEY_DELIMITER)
|
||||
if not frame_id or not signature_hash:
|
||||
return None
|
||||
if KEY_DELIMITER in signature_hash:
|
||||
return None
|
||||
return frame_id, signature_hash
|
||||
|
||||
|
||||
def _cache_path(frame_id: str, signature_hash: str) -> pathlib.Path:
|
||||
return CACHE_ROOT / frame_id / f"{signature_hash}.json"
|
||||
|
||||
|
||||
def read_proposal(
|
||||
key: str,
|
||||
*,
|
||||
fingerprints: dict | None = None,
|
||||
) -> AiFallbackProposal | None:
|
||||
"""Look up a previously cached proposal by ``key``.
|
||||
|
||||
Returns ``None`` for:
|
||||
|
||||
* empty / non-string key → ``ValueError`` (loud);
|
||||
* non-dict ``fingerprints`` (when supplied) → ``TypeError`` (loud,
|
||||
symmetric with :func:`save_proposal`);
|
||||
* legacy key format (no ``::`` delimiter) → silent ``None`` (router
|
||||
back-compat until u4 switches to the structural form);
|
||||
* missing file under ``data/frame_cache/{frame_id}/{signature_hash}.json``;
|
||||
* corrupt JSON / payload schema mismatch — read errors never propagate;
|
||||
* ``fingerprints`` supplied AND stored ``fingerprints`` field is not a
|
||||
dict OR does not equal the supplied dict (strict equality,
|
||||
u3 invalidation).
|
||||
"""
|
||||
if not isinstance(key, str) or not key:
|
||||
raise ValueError("cache key must be a non-empty string")
|
||||
if fingerprints is not None and not isinstance(fingerprints, dict):
|
||||
raise TypeError("fingerprints must be a dict or None")
|
||||
parsed = _parse_key(key)
|
||||
if parsed is None:
|
||||
return None
|
||||
frame_id, signature_hash = parsed
|
||||
path = _cache_path(frame_id, signature_hash)
|
||||
if not path.is_file():
|
||||
return None
|
||||
try:
|
||||
data = json.loads(path.read_text(encoding="utf-8"))
|
||||
except (OSError, json.JSONDecodeError):
|
||||
return None
|
||||
if not isinstance(data, dict):
|
||||
return None
|
||||
if fingerprints is not None:
|
||||
stored = data.get("fingerprints")
|
||||
if not isinstance(stored, dict) or stored != fingerprints:
|
||||
return None
|
||||
proposal_dict = data.get("proposal")
|
||||
if not isinstance(proposal_dict, dict):
|
||||
return None
|
||||
try:
|
||||
return AiFallbackProposal.model_validate(proposal_dict)
|
||||
except Exception: # noqa: BLE001 — corrupt payload must miss, not raise
|
||||
return None
|
||||
|
||||
|
||||
def save_proposal(
|
||||
key: str,
|
||||
proposal: AiFallbackProposal,
|
||||
*,
|
||||
visual_check_passed: bool,
|
||||
user_approved: bool,
|
||||
slide_css: str | None = None,
|
||||
fingerprints: dict | None = None,
|
||||
auto_cache: bool = False,
|
||||
) -> pathlib.Path:
|
||||
"""Persist ``proposal`` under ``key`` once the IMP-46 gates clear.
|
||||
|
||||
Gate contract (IMP-46 u5 truth table):
|
||||
|
||||
* ``visual_check_passed=False`` -> :class:`AiFallbackCacheGateError`
|
||||
always (never bypassable; ``auto_cache`` cannot override).
|
||||
* ``user_approved=False`` AND ``auto_cache=False`` ->
|
||||
:class:`AiFallbackCacheGateError`.
|
||||
* ``user_approved=False`` AND ``auto_cache=True`` -> bypass the
|
||||
user-approval gate (IMP-46 u5 CLI / settings opt-in).
|
||||
* Otherwise (``visual_check_passed=True`` AND either
|
||||
``user_approved=True`` OR ``auto_cache=True``) -> persist payload.
|
||||
|
||||
Gate violations are raised BEFORE any filesystem touch — no parent
|
||||
directory is created, no file is written. When the gates clear the
|
||||
JSON payload (schema_version + proposal + slide_css + fingerprints)
|
||||
is written to ``data/frame_cache/{frame_id}/{signature_hash}.json``
|
||||
and the resolved :class:`pathlib.Path` is returned.
|
||||
|
||||
``slide_css`` may be ``None`` (no slide-level CSS captured) or a
|
||||
string. ``fingerprints`` may be ``None`` (treated as empty dict) or a
|
||||
dict mapping fingerprint name to SHA hex digest.
|
||||
|
||||
``auto_cache`` is keyword-only and defaults to ``False``. It is wired
|
||||
from :data:`src.config.settings.ai_fallback_auto_cache`, which the
|
||||
``--auto-cache`` CLI flag in ``src/phase_z2_pipeline.py`` toggles at
|
||||
parse time. The cache module never reads the setting itself — the
|
||||
caller passes the resolved boolean — so AI-isolation contracts
|
||||
(no Phase Z runtime / no Anthropic import) remain intact.
|
||||
"""
|
||||
if not isinstance(key, str) or not key:
|
||||
raise ValueError("cache key must be a non-empty string")
|
||||
if not isinstance(proposal, AiFallbackProposal):
|
||||
raise TypeError(
|
||||
"proposal must be an AiFallbackProposal instance "
|
||||
f"(got {type(proposal).__name__})"
|
||||
)
|
||||
if not isinstance(auto_cache, bool):
|
||||
raise TypeError("auto_cache must be a bool")
|
||||
if not visual_check_passed:
|
||||
raise AiFallbackCacheGateError(
|
||||
"IMP-46 gate: visual_check_passed=False; refusing to cache an "
|
||||
"unverified proposal. (auto_cache cannot bypass this gate.)"
|
||||
)
|
||||
if not user_approved and not auto_cache:
|
||||
raise AiFallbackCacheGateError(
|
||||
"IMP-46 gate: user_approved=False and auto_cache=False; "
|
||||
"refusing to cache without explicit user approval. Pass "
|
||||
"auto_cache=True (or --auto-cache on the CLI) to bypass."
|
||||
)
|
||||
if slide_css is not None and not isinstance(slide_css, str):
|
||||
raise TypeError("slide_css must be a string or None")
|
||||
if fingerprints is None:
|
||||
fingerprints = {}
|
||||
elif not isinstance(fingerprints, dict):
|
||||
raise TypeError("fingerprints must be a dict or None")
|
||||
parsed = _parse_key(key)
|
||||
if parsed is None:
|
||||
raise ValueError(
|
||||
"cache key must be in "
|
||||
f"'frame_id{KEY_DELIMITER}signature_hash' format; got {key!r}"
|
||||
)
|
||||
frame_id, signature_hash = parsed
|
||||
path = _cache_path(frame_id, signature_hash)
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
payload = {
|
||||
"schema_version": SCHEMA_VERSION,
|
||||
"proposal": proposal.model_dump(mode="json"),
|
||||
"slide_css": slide_css,
|
||||
"fingerprints": dict(fingerprints),
|
||||
}
|
||||
path.write_text(
|
||||
json.dumps(payload, sort_keys=True, ensure_ascii=False, indent=2),
|
||||
encoding="utf-8",
|
||||
)
|
||||
return path
|
||||
@@ -0,0 +1,92 @@
|
||||
"""IMP-33 u4 — AI fallback Anthropic client (fallback path only).
|
||||
|
||||
Wraps ``anthropic.Anthropic.messages.create`` with the timeout / retry /
|
||||
backoff / budget / circuit-breaker policy locked in u1 ``Settings``. NO
|
||||
inline policy literals: every knob is sourced from ``src.config.settings``.
|
||||
Transient errors (timeout / connection / 429 / 5xx) are retried with
|
||||
capped exponential backoff + jitter; all other errors propagate without
|
||||
retry. PZ-1 invariant: this module is fallback-path only and MUST NOT be
|
||||
imported on the normal pipeline path.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import random
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from typing import Any
|
||||
|
||||
import anthropic
|
||||
|
||||
from src.config import settings
|
||||
from src.phase_z2_ai_fallback.schema import AiFallbackProposal
|
||||
|
||||
_TRANSIENT_ERRORS: tuple[type[BaseException], ...] = (
|
||||
anthropic.APITimeoutError,
|
||||
anthropic.APIConnectionError,
|
||||
anthropic.RateLimitError,
|
||||
anthropic.InternalServerError,
|
||||
)
|
||||
|
||||
# Output cap is an Anthropic API requirement, not a policy knob (u1).
|
||||
_MAX_OUTPUT_TOKENS = 4096
|
||||
|
||||
|
||||
class AiFallbackBudgetExceeded(RuntimeError):
|
||||
"""Per-run AI call budget (u1 ai_fallback_budget_per_run) exhausted."""
|
||||
|
||||
|
||||
class AiFallbackCircuitOpen(RuntimeError):
|
||||
"""Circuit breaker tripped (u1 ai_fallback_circuit_breaker_threshold)."""
|
||||
|
||||
|
||||
@dataclass
|
||||
class AiFallbackClient:
|
||||
"""Stateful per-run fallback client (budget + circuit accounting)."""
|
||||
|
||||
client: Any = None
|
||||
_calls: int = 0
|
||||
_consecutive_failures: int = 0
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if self.client is None:
|
||||
self.client = anthropic.Anthropic(
|
||||
api_key=settings.anthropic_api_key,
|
||||
timeout=settings.ai_fallback_timeout_s,
|
||||
)
|
||||
|
||||
def request_proposal(self, prompt: dict[str, str]) -> AiFallbackProposal:
|
||||
if self._calls >= settings.ai_fallback_budget_per_run:
|
||||
raise AiFallbackBudgetExceeded(
|
||||
f"per-run budget {settings.ai_fallback_budget_per_run} exhausted"
|
||||
)
|
||||
if self._consecutive_failures >= settings.ai_fallback_circuit_breaker_threshold:
|
||||
raise AiFallbackCircuitOpen(
|
||||
f"circuit open after {self._consecutive_failures} consecutive failures"
|
||||
)
|
||||
self._calls += 1
|
||||
last_error: BaseException | None = None
|
||||
for attempt in range(settings.ai_fallback_max_retries + 1):
|
||||
try:
|
||||
response = self.client.messages.create(
|
||||
model=settings.ai_fallback_model,
|
||||
max_tokens=_MAX_OUTPUT_TOKENS,
|
||||
system=prompt["system"],
|
||||
messages=[{"role": "user", "content": prompt["user"]}],
|
||||
)
|
||||
text = "".join(
|
||||
block.text for block in response.content if hasattr(block, "text")
|
||||
)
|
||||
self._consecutive_failures = 0
|
||||
return AiFallbackProposal.model_validate(json.loads(text))
|
||||
except _TRANSIENT_ERRORS as err:
|
||||
last_error = err
|
||||
if attempt >= settings.ai_fallback_max_retries:
|
||||
break
|
||||
base = settings.ai_fallback_backoff_base_s * (2 ** attempt)
|
||||
delay = min(settings.ai_fallback_backoff_cap_s, base)
|
||||
delay += random.uniform(0, delay * settings.ai_fallback_backoff_jitter)
|
||||
time.sleep(delay)
|
||||
self._consecutive_failures += 1
|
||||
assert last_error is not None
|
||||
raise last_error
|
||||
@@ -0,0 +1,80 @@
|
||||
"""IMP-33 u3 — AI fallback prompt builder (fallback path only).
|
||||
|
||||
System+user prompt for the Anthropic client (u4). MDX is READ-ONLY
|
||||
(`feedback_ai_isolation_contract`); output is constrained to the u2
|
||||
schema; frame_id swap is forbidden (V4 rank-1 protected,
|
||||
`feedback_phase_z_spacing_direction`). Inputs per Stage 2 plan: V4
|
||||
result (route=ai_adaptation_required, cardinality), frame_contract,
|
||||
frame_visual HTML, figma_to_html_agent partial JSON, Internal Region,
|
||||
MDX text.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import Any
|
||||
|
||||
from src.phase_z2_ai_fallback.schema import FORBIDDEN_KINDS, ProposalKind
|
||||
|
||||
V4_ROUTE_AI_ADAPTATION = "ai_adaptation_required"
|
||||
|
||||
_ALLOWED_KINDS = ", ".join(sorted(k.value for k in ProposalKind))
|
||||
_FORBIDDEN_KINDS = ", ".join(sorted(FORBIDDEN_KINDS))
|
||||
|
||||
SYSTEM_PROMPT = (
|
||||
"You are an IMP-33 AI fallback adapter for Phase Z slide composition.\n"
|
||||
"STRICT RULES:\n"
|
||||
" 1. MDX text in the user payload is READ-ONLY. Do NOT rewrite, "
|
||||
"compress, or paraphrase MDX.\n"
|
||||
" 2. Output MUST be a single JSON object conforming to AiFallbackProposal.\n"
|
||||
f" 3. proposal_kind MUST be one of: {_ALLOWED_KINDS}.\n"
|
||||
f" 4. Do NOT propose any of: {_FORBIDDEN_KINDS}.\n"
|
||||
" 5. Do NOT change frame_id — V4 rank-1 frame is locked.\n"
|
||||
" 6. Keep declared frame slots (text/table/image/details) populated.\n"
|
||||
" 7. Respect Internal Region containment; place content units within "
|
||||
"the declared region only."
|
||||
)
|
||||
|
||||
|
||||
def build_ai_fallback_prompt(
|
||||
*,
|
||||
v4_result: dict[str, Any],
|
||||
frame_contract: dict[str, Any],
|
||||
frame_visual_html: str,
|
||||
figma_partial_json: dict[str, Any],
|
||||
internal_region: dict[str, Any],
|
||||
mdx_text: str,
|
||||
) -> dict[str, str]:
|
||||
"""Build system+user prompt strings for the fallback AI adapter.
|
||||
|
||||
Raises:
|
||||
ValueError: when ``v4_result.route`` is not
|
||||
``ai_adaptation_required`` — the fallback prompt MUST NOT be
|
||||
built outside this route (normal-path AI call count must
|
||||
remain 0; PZ-1).
|
||||
"""
|
||||
route = v4_result.get("route") or v4_result.get("imp05_route_hint")
|
||||
if route != V4_ROUTE_AI_ADAPTATION:
|
||||
raise ValueError(
|
||||
f"build_ai_fallback_prompt: v4_result.route={route!r} is not "
|
||||
f"{V4_ROUTE_AI_ADAPTATION!r}; fallback prompt MUST NOT be built "
|
||||
"outside the AI adaptation route."
|
||||
)
|
||||
user_payload = {
|
||||
"v4": {
|
||||
"route": route,
|
||||
"cardinality": v4_result.get("cardinality")
|
||||
or v4_result.get("cardinality_signature"),
|
||||
"label": v4_result.get("label"),
|
||||
"frame_id": v4_result.get("frame_id"),
|
||||
"rank": v4_result.get("rank"),
|
||||
},
|
||||
"frame_contract": frame_contract,
|
||||
"frame_visual_html": frame_visual_html,
|
||||
"figma_partial_json": figma_partial_json,
|
||||
"internal_region": internal_region,
|
||||
"mdx_text_READ_ONLY": mdx_text,
|
||||
}
|
||||
return {
|
||||
"system": SYSTEM_PROMPT,
|
||||
"user": json.dumps(user_payload, ensure_ascii=False),
|
||||
}
|
||||
@@ -0,0 +1,89 @@
|
||||
"""IMP-33 u7 — AI fallback router (fallback path only).
|
||||
|
||||
Composes the IMP-33 fallback flow:
|
||||
|
||||
1. flag gate (``settings.ai_fallback_enabled`` default OFF)
|
||||
2. V4 route gate (route must equal ``ai_adaptation_required``)
|
||||
3. cache read (u6 stub returns ``None`` until IMP-46 lands)
|
||||
4. build prompt (u3)
|
||||
5. call client (u4 ``request_proposal``)
|
||||
6. validate (u5 ``validate_proposal``)
|
||||
|
||||
Returns the validated ``AiFallbackProposal``. Save to cache is NOT
|
||||
performed here — it is caller-driven AFTER ``visual_check_passed=True``
|
||||
AND ``user_approved=True``, per the u6 IMP-46 gate. The router does not
|
||||
import ``save_proposal``; this is the structural guarantee that the
|
||||
router cannot persist a proposal before the caller's visual + user
|
||||
checks (`feedback_artifact_status_naming`).
|
||||
|
||||
Guardrails:
|
||||
|
||||
* PZ-1 — normal-path AI call count stays 0: flag-off OR route-mismatch
|
||||
short-circuits BEFORE the prompt builder or client are touched.
|
||||
* ``feedback_ai_isolation_contract`` — MDX READ-ONLY (u3 enforces in
|
||||
prompt; this module never reads or writes MDX).
|
||||
* ``feedback_phase_z_spacing_direction`` — V4 rank-1 protected (u5
|
||||
enforces; router only forwards the contract).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from src.config import settings
|
||||
from src.phase_z2_ai_fallback.cache import read_proposal
|
||||
from src.phase_z2_ai_fallback.client import AiFallbackClient
|
||||
from src.phase_z2_ai_fallback.prompts import (
|
||||
V4_ROUTE_AI_ADAPTATION,
|
||||
build_ai_fallback_prompt,
|
||||
)
|
||||
from src.phase_z2_ai_fallback.schema import AiFallbackProposal
|
||||
from src.phase_z2_ai_fallback.validate import validate_proposal
|
||||
|
||||
|
||||
def route_ai_fallback(
|
||||
*,
|
||||
cache_key: str,
|
||||
v4_result: dict[str, Any],
|
||||
frame_contract: dict[str, Any],
|
||||
frame_visual_html: str,
|
||||
figma_partial_json: dict[str, Any],
|
||||
internal_region: dict[str, Any],
|
||||
mdx_text: str,
|
||||
client: AiFallbackClient | None = None,
|
||||
) -> AiFallbackProposal | None:
|
||||
"""Route a fallback request through cache → prompt → client → validate.
|
||||
|
||||
Returns ``None`` when the master flag is OFF or when the V4 route is
|
||||
not ``ai_adaptation_required`` — both gates short-circuit BEFORE any
|
||||
prompt/client work, so the normal-path AI call count stays at 0
|
||||
(PZ-1).
|
||||
"""
|
||||
if not settings.ai_fallback_enabled:
|
||||
return None
|
||||
route = v4_result.get("route") or v4_result.get("imp05_route_hint")
|
||||
if route != V4_ROUTE_AI_ADAPTATION:
|
||||
return None
|
||||
cached = read_proposal(cache_key)
|
||||
if cached is not None:
|
||||
validate_proposal(
|
||||
cached,
|
||||
frame_contract=frame_contract,
|
||||
internal_region=internal_region,
|
||||
)
|
||||
return cached
|
||||
prompt = build_ai_fallback_prompt(
|
||||
v4_result=v4_result,
|
||||
frame_contract=frame_contract,
|
||||
frame_visual_html=frame_visual_html,
|
||||
figma_partial_json=figma_partial_json,
|
||||
internal_region=internal_region,
|
||||
mdx_text=mdx_text,
|
||||
)
|
||||
active_client = client if client is not None else AiFallbackClient()
|
||||
proposal = active_client.request_proposal(prompt)
|
||||
validate_proposal(
|
||||
proposal,
|
||||
frame_contract=frame_contract,
|
||||
internal_region=internal_region,
|
||||
)
|
||||
return proposal
|
||||
@@ -0,0 +1,50 @@
|
||||
"""IMP-33 u2 — AI fallback proposal schema.
|
||||
|
||||
Whitelisted proposal kinds (Stage 2 plan):
|
||||
- builder_options_patch : zone/frame builder option overrides
|
||||
- partial_overrides : Internal Region / Frame Slot content overrides
|
||||
- slot_mapping_proposal : restructuring proposal (content unit mapping)
|
||||
|
||||
Forbidden output forms (rejected by validator):
|
||||
- mdx_text (MDX read-only — `feedback_ai_isolation_contract`)
|
||||
- frame_id_change (V4 rank-1 protected — `feedback_phase_z_spacing_direction`)
|
||||
- raw_html (HTML structure is code-decided, not AI-generated)
|
||||
- raw_css (same)
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from enum import Enum
|
||||
from typing import Any
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field, field_validator
|
||||
|
||||
|
||||
class ProposalKind(str, Enum):
|
||||
BUILDER_OPTIONS_PATCH = "builder_options_patch"
|
||||
PARTIAL_OVERRIDES = "partial_overrides"
|
||||
SLOT_MAPPING_PROPOSAL = "slot_mapping_proposal"
|
||||
|
||||
|
||||
FORBIDDEN_KINDS: frozenset[str] = frozenset(
|
||||
{"mdx_text", "frame_id_change", "raw_html", "raw_css"}
|
||||
)
|
||||
|
||||
|
||||
class AiFallbackProposal(BaseModel):
|
||||
"""Single AI fallback proposal (output contract for u4 client)."""
|
||||
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
|
||||
proposal_kind: ProposalKind
|
||||
payload: dict[str, Any] = Field(default_factory=dict)
|
||||
rationale: str = ""
|
||||
|
||||
@field_validator("proposal_kind", mode="before")
|
||||
@classmethod
|
||||
def _reject_forbidden_kind(cls, value: Any) -> Any:
|
||||
if isinstance(value, str) and value in FORBIDDEN_KINDS:
|
||||
raise ValueError(
|
||||
f"proposal_kind={value!r} is forbidden (MDX/frame/raw HTML/CSS "
|
||||
"mutations are not permitted under IMP-33)."
|
||||
)
|
||||
return value
|
||||
@@ -0,0 +1,91 @@
|
||||
"""IMP-46 u1 — Frame transformation cache signature builder.
|
||||
|
||||
Deterministic SHA256 over the 8 declared structural axes:
|
||||
frame_id, v4_label, cardinality, source_shape,
|
||||
h3_count, char_count_bucket, layout_preset, zone_position
|
||||
|
||||
Guardrails:
|
||||
* No sample/section identifiers in the signature surface (no-hardcoding lock).
|
||||
* source_shape constrained to the bullet/paragraph/table/mixed enum.
|
||||
* char_count_bucket is the *bucket label*; numeric counts must be projected
|
||||
via :func:`bucket_char_count` before being fed to :func:`build_signature`.
|
||||
* Schema version is embedded in the hashed payload so a future axis change
|
||||
breaks the digest by design (cache invalidation on schema bump).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from enum import Enum
|
||||
|
||||
|
||||
SCHEMA_VERSION = 1
|
||||
|
||||
|
||||
class SourceShape(str, Enum):
|
||||
BULLET = "bullet"
|
||||
PARAGRAPH = "paragraph"
|
||||
TABLE = "table"
|
||||
MIXED = "mixed"
|
||||
|
||||
|
||||
_CHAR_COUNT_BUCKETS: tuple[tuple[int, str], ...] = (
|
||||
(50, "0-50"),
|
||||
(150, "51-150"),
|
||||
(400, "151-400"),
|
||||
(1000, "401-1000"),
|
||||
)
|
||||
_CHAR_COUNT_BUCKET_OVERFLOW = "1001+"
|
||||
CHAR_COUNT_BUCKET_LABELS: tuple[str, ...] = tuple(
|
||||
label for _, label in _CHAR_COUNT_BUCKETS
|
||||
) + (_CHAR_COUNT_BUCKET_OVERFLOW,)
|
||||
|
||||
|
||||
def bucket_char_count(char_count: int) -> str:
|
||||
"""Project a non-negative character count to its fixed bucket label."""
|
||||
if isinstance(char_count, bool) or not isinstance(char_count, int):
|
||||
raise TypeError("char_count must be a non-negative int")
|
||||
if char_count < 0:
|
||||
raise ValueError("char_count must be non-negative")
|
||||
for upper, label in _CHAR_COUNT_BUCKETS:
|
||||
if char_count <= upper:
|
||||
return label
|
||||
return _CHAR_COUNT_BUCKET_OVERFLOW
|
||||
|
||||
|
||||
def build_signature(
|
||||
*,
|
||||
frame_id: str,
|
||||
v4_label: str,
|
||||
cardinality: int | None,
|
||||
source_shape: SourceShape | str,
|
||||
h3_count: int,
|
||||
char_count_bucket: str,
|
||||
layout_preset: str,
|
||||
zone_position: str,
|
||||
) -> str:
|
||||
"""Return a deterministic SHA256 hex digest over the 8 declared axes."""
|
||||
if isinstance(source_shape, SourceShape):
|
||||
source_shape_value = source_shape.value
|
||||
elif isinstance(source_shape, str):
|
||||
source_shape_value = SourceShape(source_shape).value
|
||||
else:
|
||||
raise TypeError("source_shape must be SourceShape or str")
|
||||
if char_count_bucket not in CHAR_COUNT_BUCKET_LABELS:
|
||||
raise ValueError(
|
||||
f"char_count_bucket={char_count_bucket!r} is not a known bucket "
|
||||
f"label (expected one of {CHAR_COUNT_BUCKET_LABELS})"
|
||||
)
|
||||
payload = {
|
||||
"schema_version": SCHEMA_VERSION,
|
||||
"frame_id": frame_id,
|
||||
"v4_label": v4_label,
|
||||
"cardinality": cardinality,
|
||||
"source_shape": source_shape_value,
|
||||
"h3_count": h3_count,
|
||||
"char_count_bucket": char_count_bucket,
|
||||
"layout_preset": layout_preset,
|
||||
"zone_position": zone_position,
|
||||
}
|
||||
encoded = json.dumps(payload, sort_keys=True, ensure_ascii=False).encode("utf-8")
|
||||
return hashlib.sha256(encoded).hexdigest()
|
||||
@@ -0,0 +1,216 @@
|
||||
"""IMP-33 u8 + IMP-46 u4 — Step 12 AI repair wiring with structural cache key.
|
||||
|
||||
Phase Z Step 12 = slot_payload (the runtime "light_edit / restructure" surface
|
||||
where AI-assisted frame-aware adaptation is allowed per IMP-17 carve-out).
|
||||
This module is the only call site that pipes Phase Z composition units into
|
||||
``src.phase_z2_ai_fallback.router.route_ai_fallback``. One structural gate
|
||||
preserves the AI isolation contract:
|
||||
|
||||
* IMP-30 provisional gate — units with ``provisional=False`` are skipped
|
||||
before any route classification. AI repair is reserved for first-render
|
||||
invariant survivors (no rank-1 V4 evidence, recovered as provisional).
|
||||
|
||||
Per IMP-47B u1+u2, the ``reject`` V4 label routes to
|
||||
``ai_adaptation_required`` (no longer ``design_reference_only``) and is
|
||||
admitted to the AI repair path; the legacy "reject gate" short-circuit is
|
||||
removed. Any unit whose ``route_hint`` is not ``ai_adaptation_required``
|
||||
still falls through to the catch-all ``route_not_ai_adaptation:<hint>``
|
||||
skip — that single gate continues to enforce the AI=0 normal path.
|
||||
|
||||
Combined with the u7 router's flag-off + route-gate short-circuits, the
|
||||
default Phase Z run path performs zero AI calls (PZ-1). Save to cache is
|
||||
NOT performed here — that is the caller's responsibility AFTER
|
||||
``visual_check_passed=True`` AND ``user_approved=True`` (u6 IMP-46 gate).
|
||||
|
||||
IMP-46 u4 — structural cache key + fingerprints
|
||||
------------------------------------------------
|
||||
|
||||
The legacy ``cache_key`` was ``"{template_id}::{sorted(source_section_ids)}"``
|
||||
which leaked sample / section identity into the cache surface
|
||||
(no-hardcoding lock violation: structurally identical content with
|
||||
different MDX section ids would miss). u4 replaces it with
|
||||
``"{frame_id}::{signature_hash}"`` where ``signature_hash`` is the
|
||||
deterministic SHA256 over the 8 declared structural axes (see
|
||||
``src.phase_z2_ai_fallback.signature``). Per-unit signature inputs are
|
||||
read from unit attributes:
|
||||
|
||||
* ``cardinality`` (int | None) — also forwarded to ``v4_result``
|
||||
* ``layout_preset`` (str)
|
||||
* ``zone_position`` (str)
|
||||
* ``source_shape`` (str) — bullet / paragraph / table / mixed
|
||||
* ``h3_count`` (int)
|
||||
* ``char_count`` (int) — bucketed via ``bucket_char_count``
|
||||
|
||||
In parallel the three invalidation fingerprints
|
||||
(``contract_sha`` / ``partial_sha`` / ``catalog_sha``) are computed and
|
||||
attached to the record. The cache.py module remains a *comparator* — all
|
||||
fingerprint *computation* happens here (or via injected loaders) so the
|
||||
cache schema-agnostic contract is preserved. The router's existing
|
||||
``read_proposal(cache_key)`` continues to perform exact-match lookup only
|
||||
(fuzzy is deferred per Stage 2 plan); read-side fingerprint validation
|
||||
through the router is a follow-up axis.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from typing import Any, Callable, Iterable
|
||||
|
||||
from src.phase_z2_ai_fallback.router import route_ai_fallback
|
||||
from src.phase_z2_ai_fallback.signature import bucket_char_count, build_signature
|
||||
|
||||
|
||||
_AI_ADAPTATION_ROUTE = "ai_adaptation_required"
|
||||
|
||||
|
||||
def _sha256_of(payload: Any) -> str:
|
||||
"""Deterministic SHA256 hex digest over a JSON-serialisable payload."""
|
||||
encoded = json.dumps(payload, sort_keys=True, ensure_ascii=False).encode("utf-8")
|
||||
return hashlib.sha256(encoded).hexdigest()
|
||||
|
||||
|
||||
def gather_step12_ai_repair_proposals(
|
||||
units: Iterable[Any],
|
||||
*,
|
||||
route_for_label: Callable[[str | None], str | None],
|
||||
get_contract_fn: Callable[[str], dict | None],
|
||||
frame_visual_loader: Callable[[str], str],
|
||||
figma_partial_loader: Callable[[str], dict] | None = None,
|
||||
internal_region_lookup: Callable[[Any], dict] | None = None,
|
||||
mdx_text_loader: Callable[[Any], str] | None = None,
|
||||
catalog_sha_loader: Callable[[], str] | None = None,
|
||||
) -> list[dict]:
|
||||
"""Return one record per unit describing the Step 12 AI repair decision.
|
||||
|
||||
The record schema is stable across all gate decisions so the Step 12
|
||||
artifact consumer can rely on a single shape:
|
||||
|
||||
{
|
||||
"unit_index": int,
|
||||
"source_section_ids": list[str],
|
||||
"frame_template_id": str,
|
||||
"label": str | None,
|
||||
"route_hint": str | None,
|
||||
"provisional": bool,
|
||||
"ai_called": bool,
|
||||
"skip_reason": str | None,
|
||||
"proposal": dict | None,
|
||||
"error": str | None,
|
||||
"cache_key": str | None, # IMP-46 u4
|
||||
"fingerprints": dict | None, # IMP-46 u4
|
||||
}
|
||||
|
||||
``cache_key`` and ``fingerprints`` are populated only when the unit
|
||||
reaches the AI-eligible code path (provisional + ai_adaptation route).
|
||||
Skipped units retain ``None`` for both — the structural axes
|
||||
(layout_preset / zone_position / source_shape / h3_count / char_count)
|
||||
are not guaranteed to be set for non-AI paths.
|
||||
|
||||
``ai_called`` is True only when ``route_ai_fallback`` was invoked AND
|
||||
returned a proposal OR raised. Flag-off / route-mismatch returns
|
||||
``None`` from the router and is surfaced as ``ai_called=False`` with
|
||||
``skip_reason="router_short_circuit"`` so the caller can distinguish
|
||||
"router decided not to run" from "router ran and returned a proposal".
|
||||
"""
|
||||
records: list[dict] = []
|
||||
catalog_sha = (
|
||||
catalog_sha_loader() if catalog_sha_loader is not None else ""
|
||||
)
|
||||
for index, unit in enumerate(units):
|
||||
label = getattr(unit, "label", None)
|
||||
route_hint = route_for_label(label)
|
||||
record: dict = {
|
||||
"unit_index": index,
|
||||
"source_section_ids": list(getattr(unit, "source_section_ids", []) or []),
|
||||
"frame_template_id": getattr(unit, "frame_template_id", None),
|
||||
"label": label,
|
||||
"route_hint": route_hint,
|
||||
"provisional": bool(getattr(unit, "provisional", False)),
|
||||
"ai_called": False,
|
||||
"skip_reason": None,
|
||||
"proposal": None,
|
||||
"error": None,
|
||||
"cache_key": None,
|
||||
"fingerprints": None,
|
||||
}
|
||||
if not record["provisional"]:
|
||||
record["skip_reason"] = "not_provisional"
|
||||
records.append(record)
|
||||
continue
|
||||
if route_hint != _AI_ADAPTATION_ROUTE:
|
||||
record["skip_reason"] = f"route_not_ai_adaptation:{route_hint}"
|
||||
records.append(record)
|
||||
continue
|
||||
|
||||
template_id = record["frame_template_id"] or ""
|
||||
frame_contract = get_contract_fn(template_id) or {}
|
||||
frame_visual_html = frame_visual_loader(template_id)
|
||||
figma_partial_json = (
|
||||
figma_partial_loader(template_id) if figma_partial_loader is not None else {}
|
||||
)
|
||||
internal_region = (
|
||||
internal_region_lookup(unit) if internal_region_lookup is not None else {}
|
||||
)
|
||||
mdx_text = (
|
||||
mdx_text_loader(unit)
|
||||
if mdx_text_loader is not None
|
||||
else (getattr(unit, "raw_content", "") or "")
|
||||
)
|
||||
|
||||
frame_id_value = getattr(unit, "frame_id", "") or ""
|
||||
cardinality = getattr(unit, "cardinality", None)
|
||||
layout_preset = getattr(unit, "layout_preset", "") or ""
|
||||
zone_position = getattr(unit, "zone_position", "") or ""
|
||||
source_shape = getattr(unit, "source_shape", "paragraph") or "paragraph"
|
||||
h3_count = int(getattr(unit, "h3_count", 0) or 0)
|
||||
char_count = int(getattr(unit, "char_count", 0) or 0)
|
||||
char_count_bucket = bucket_char_count(char_count)
|
||||
signature_hash = build_signature(
|
||||
frame_id=frame_id_value,
|
||||
v4_label=label or "",
|
||||
cardinality=cardinality,
|
||||
source_shape=source_shape,
|
||||
h3_count=h3_count,
|
||||
char_count_bucket=char_count_bucket,
|
||||
layout_preset=layout_preset,
|
||||
zone_position=zone_position,
|
||||
)
|
||||
cache_key = f"{frame_id_value}::{signature_hash}"
|
||||
fingerprints = {
|
||||
"contract_sha": _sha256_of(frame_contract),
|
||||
"partial_sha": _sha256_of(figma_partial_json),
|
||||
"catalog_sha": catalog_sha,
|
||||
}
|
||||
record["cache_key"] = cache_key
|
||||
record["fingerprints"] = fingerprints
|
||||
|
||||
v4_result = {
|
||||
"route": route_hint,
|
||||
"label": label,
|
||||
"frame_id": getattr(unit, "frame_id", None),
|
||||
"rank": getattr(unit, "v4_rank", None),
|
||||
"cardinality": cardinality,
|
||||
}
|
||||
try:
|
||||
proposal = route_ai_fallback(
|
||||
cache_key=cache_key,
|
||||
v4_result=v4_result,
|
||||
frame_contract=frame_contract,
|
||||
frame_visual_html=frame_visual_html,
|
||||
figma_partial_json=figma_partial_json,
|
||||
internal_region=internal_region,
|
||||
mdx_text=mdx_text,
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001 — record + continue, no AI re-raise
|
||||
record["ai_called"] = True
|
||||
record["error"] = f"{type(exc).__name__}: {exc}"
|
||||
records.append(record)
|
||||
continue
|
||||
if proposal is None:
|
||||
record["skip_reason"] = "router_short_circuit"
|
||||
records.append(record)
|
||||
continue
|
||||
record["ai_called"] = True
|
||||
record["proposal"] = proposal.model_dump()
|
||||
records.append(record)
|
||||
return records
|
||||
@@ -0,0 +1,111 @@
|
||||
"""IMP-33 u9 — Step 17 AI repair wiring (BLOCKED until IMP-34 + IMP-35 land).
|
||||
|
||||
Phase Z Step 17 = retry / salvage cascade (see ``src.phase_z2_pipeline``
|
||||
section 11.7 ``_attempt_salvage_chain`` and the existing IMP-12 u8/u9
|
||||
deterministic chain at ``src/phase_z2_pipeline.py:1994`` and
|
||||
``src/phase_z2_pipeline.py:4948``).
|
||||
|
||||
Per IMP-17 carve-out (``docs/architecture/IMP-17-CARVE-OUT.md`` lines 16,
|
||||
40-44), AI repair at Step 17 is permitted ONLY after the full deterministic
|
||||
chain is exhausted AND popup escalation is exhausted AND a user-approved
|
||||
fallback budget remains. IMP-34 (zone resize + compact retry) and IMP-35
|
||||
(``details_popup_escalation``) are explicit prerequisites under the IMP-33
|
||||
out-of-scope contract — neither has landed yet. Therefore Step 17 AI repair
|
||||
is STRUCTURALLY BLOCKED at u9.
|
||||
|
||||
This module:
|
||||
|
||||
1. **SPECIFIES** the canonical overflow cascade order via
|
||||
:data:`OVERFLOW_CASCADE_ORDER` — ``deterministic`` → ``popup`` →
|
||||
``ai_repair`` → ``user_override``. Downstream Step 17 consumers can rely
|
||||
on this single source of truth.
|
||||
2. **KEEPS** Step 17 AI repair structurally blocked. The entry point
|
||||
:func:`gather_step17_ai_repair_proposals` does NOT import
|
||||
``route_ai_fallback`` (u7), does NOT instantiate ``AiFallbackClient`` (u4),
|
||||
and does NOT call any Anthropic API. Every unit is recorded with
|
||||
``skip_reason="step17_ai_blocked_imp_34_35_prerequisites_missing"`` so
|
||||
the caller can distinguish "blocked by carve-out gate" from any other
|
||||
skip path (e.g., u8 ``not_provisional`` / ``design_reference_only_no_ai``).
|
||||
|
||||
Once IMP-34 + IMP-35 land AND a user-approved fallback budget is granted,
|
||||
this module will gain the actual ``route_ai_fallback`` wiring guarded by
|
||||
the cascade-stage conjunction. Today the gate is closed.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from enum import Enum
|
||||
from typing import Any, Callable, Iterable
|
||||
|
||||
|
||||
class OverflowCascadeStage(str, Enum):
|
||||
"""Step 17 overflow cascade stages — canonical order (u9 single source of truth).
|
||||
|
||||
Members are ordered to match the AI isolation contract:
|
||||
|
||||
* ``DETERMINISTIC`` — IMP-12 u4/u5/u6 (``cross_zone_redistribute`` /
|
||||
``glue_compression`` / ``font_step_compression``) + IMP-12 terminal
|
||||
actions (``layout_adjust`` / ``frame_reselect``) + IMP-34
|
||||
(``zone resize + compact retry``, pending). No AI in any sub-stage.
|
||||
* ``POPUP`` — IMP-35 (``details_popup_escalation``, pending). Content
|
||||
popup escalation as the final deterministic resort before any AI.
|
||||
* ``AI_REPAIR`` — IMP-33 (this carve-out) + IMP-46 cache. Only reachable
|
||||
after DETERMINISTIC and POPUP are both exhausted AND user-approved
|
||||
fallback budget remains.
|
||||
* ``USER_OVERRIDE`` — explicit user override after all auto stages.
|
||||
"""
|
||||
|
||||
DETERMINISTIC = "deterministic"
|
||||
POPUP = "popup"
|
||||
AI_REPAIR = "ai_repair"
|
||||
USER_OVERRIDE = "user_override"
|
||||
|
||||
|
||||
OVERFLOW_CASCADE_ORDER: tuple[OverflowCascadeStage, ...] = (
|
||||
OverflowCascadeStage.DETERMINISTIC,
|
||||
OverflowCascadeStage.POPUP,
|
||||
OverflowCascadeStage.AI_REPAIR,
|
||||
OverflowCascadeStage.USER_OVERRIDE,
|
||||
)
|
||||
|
||||
|
||||
STEP17_AI_REPAIR_BLOCKED_REASON = (
|
||||
"step17_ai_blocked_imp_34_35_prerequisites_missing"
|
||||
)
|
||||
|
||||
|
||||
def gather_step17_ai_repair_proposals(
|
||||
units: Iterable[Any],
|
||||
*,
|
||||
route_for_label: Callable[[str | None], str | None],
|
||||
) -> list[dict]:
|
||||
"""Return one BLOCKED record per unit. No AI call is performed at u9.
|
||||
|
||||
The record schema mirrors :func:`src.phase_z2_ai_fallback.step12
|
||||
.gather_step12_ai_repair_proposals` so the Step 17 artifact consumer can
|
||||
reuse the same shape, with one addition: ``cascade_stage`` pins the
|
||||
stage this record belongs to (always ``ai_repair`` here).
|
||||
|
||||
Per Stage 2 contract (IMP-33 u9): Step 17 AI repair is blocked behind
|
||||
IMP-34 + IMP-35. Every unit returns with
|
||||
``skip_reason=STEP17_AI_REPAIR_BLOCKED_REASON`` and ``ai_called=False``.
|
||||
"""
|
||||
records: list[dict] = []
|
||||
for index, unit in enumerate(units):
|
||||
label = getattr(unit, "label", None)
|
||||
record: dict = {
|
||||
"unit_index": index,
|
||||
"source_section_ids": list(
|
||||
getattr(unit, "source_section_ids", []) or []
|
||||
),
|
||||
"frame_template_id": getattr(unit, "frame_template_id", None),
|
||||
"label": label,
|
||||
"route_hint": route_for_label(label),
|
||||
"provisional": bool(getattr(unit, "provisional", False)),
|
||||
"cascade_stage": OverflowCascadeStage.AI_REPAIR.value,
|
||||
"ai_called": False,
|
||||
"skip_reason": STEP17_AI_REPAIR_BLOCKED_REASON,
|
||||
"proposal": None,
|
||||
"error": None,
|
||||
}
|
||||
records.append(record)
|
||||
return records
|
||||
@@ -0,0 +1,83 @@
|
||||
"""IMP-33 u5 — AI fallback proposal validator (fallback path only).
|
||||
|
||||
Defence-in-depth layer between the u4 client output (already u2-schema-valid)
|
||||
and the caller. Adds the four Stage 2 guards that u2 cannot express purely at
|
||||
the schema level:
|
||||
|
||||
1. builder-options whitelist (BUILDER_OPTIONS_PATCH may only touch keys
|
||||
already declared in ``frame_contract.payload.builder_options``).
|
||||
2. dropped-slot guard (PARTIAL_OVERRIDES / SLOT_MAPPING_PROPOSAL must keep
|
||||
every declared ``sub_zones[*].id`` populated — text/table/image/details
|
||||
slots cannot disappear; `feedback_ai_isolation_contract`).
|
||||
3. frame-swap guard (no ``frame_id`` mutation inside payload — V4 rank-1
|
||||
protected; `feedback_phase_z_spacing_direction`).
|
||||
4. Internal Region containment (``payload.region_id`` must match the
|
||||
declared Internal Region id when present).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from src.phase_z2_ai_fallback.schema import AiFallbackProposal, ProposalKind
|
||||
|
||||
|
||||
class AiFallbackValidationError(ValueError):
|
||||
"""Raised when a proposal violates an IMP-33 u5 guard."""
|
||||
|
||||
|
||||
_SLOT_KINDS = (ProposalKind.PARTIAL_OVERRIDES, ProposalKind.SLOT_MAPPING_PROPOSAL)
|
||||
|
||||
|
||||
def validate_proposal(
|
||||
proposal: AiFallbackProposal,
|
||||
*,
|
||||
frame_contract: dict[str, Any],
|
||||
internal_region: dict[str, Any] | None = None,
|
||||
) -> None:
|
||||
"""Validate an AI fallback proposal against the active frame contract.
|
||||
|
||||
Raises ``AiFallbackValidationError`` on any guard violation. Returns
|
||||
``None`` on success — caller is responsible for downstream application.
|
||||
"""
|
||||
AiFallbackProposal.model_validate(proposal.model_dump())
|
||||
|
||||
payload = proposal.payload
|
||||
frame_id = frame_contract.get("frame_id")
|
||||
if "frame_id" in payload and payload["frame_id"] != frame_id:
|
||||
raise AiFallbackValidationError(
|
||||
f"frame-swap guard: payload.frame_id={payload['frame_id']!r} "
|
||||
f"differs from contract frame_id={frame_id!r}; V4 rank-1 is locked."
|
||||
)
|
||||
|
||||
if proposal.proposal_kind is ProposalKind.BUILDER_OPTIONS_PATCH:
|
||||
declared = (frame_contract.get("payload") or {}).get("builder_options") or {}
|
||||
unknown = set(payload.keys()) - set(declared.keys())
|
||||
if unknown:
|
||||
raise AiFallbackValidationError(
|
||||
f"builder whitelist: keys {sorted(unknown)} not in "
|
||||
f"frame_contract.payload.builder_options {sorted(declared)}."
|
||||
)
|
||||
|
||||
if proposal.proposal_kind in _SLOT_KINDS:
|
||||
declared_slot_ids = [z.get("id") for z in (frame_contract.get("sub_zones") or [])]
|
||||
slots = payload.get("slots")
|
||||
if not isinstance(slots, dict):
|
||||
raise AiFallbackValidationError(
|
||||
"dropped-slot guard: PARTIAL_OVERRIDES / SLOT_MAPPING_PROPOSAL "
|
||||
"payload MUST include a 'slots' mapping."
|
||||
)
|
||||
missing = [sid for sid in declared_slot_ids if sid not in slots]
|
||||
if missing:
|
||||
raise AiFallbackValidationError(
|
||||
f"dropped-slot guard: declared slots {missing} are absent "
|
||||
"from payload.slots (text/table/image/details must remain populated)."
|
||||
)
|
||||
|
||||
region_id = payload.get("region_id")
|
||||
if region_id is not None and internal_region is not None:
|
||||
declared_region_id = internal_region.get("id")
|
||||
if region_id != declared_region_id:
|
||||
raise AiFallbackValidationError(
|
||||
f"Internal Region containment: payload.region_id={region_id!r} "
|
||||
f"differs from internal_region.id={declared_region_id!r}."
|
||||
)
|
||||
@@ -368,6 +368,15 @@ class CompositionUnit:
|
||||
# 0 길이 = "no_non_reject_v4_candidate" 신호 (Step 9 application_plan input).
|
||||
v4_candidates: list = field(default_factory=list)
|
||||
|
||||
# IMP-30 u2 — provisional first-render flag. True when the V4Match
|
||||
# backing this unit was synthesized via lookup_v4_match_with_fallback
|
||||
# (allow_provisional=True) after chain_exhausted, or when u3 inserts
|
||||
# a last-resort provisional fill for an uncovered section. Carried as
|
||||
# data (not re-derived from label/selection_path downstream) so the
|
||||
# render path / status / zone template can surface "needs adaptation"
|
||||
# uniformly. Default False keeps non-provisional units byte-identical.
|
||||
provisional: bool = False
|
||||
|
||||
|
||||
# ─── Heading Tree ──────────────────────────────────────────────
|
||||
|
||||
@@ -490,6 +499,7 @@ def collect_candidates(sections, v4_lookup_fn, v4_label_to_status: dict,
|
||||
raw_content=s.raw_content,
|
||||
title=s.title,
|
||||
v4_candidates=_v4_cands(s.section_id),
|
||||
provisional=getattr(match, "provisional", False),
|
||||
)
|
||||
_apply_capacity_fit(c, capacity_fit_fn)
|
||||
candidates.append(c)
|
||||
@@ -524,6 +534,7 @@ def collect_candidates(sections, v4_lookup_fn, v4_label_to_status: dict,
|
||||
raw_content=merged_raw,
|
||||
title=pid,
|
||||
v4_candidates=_v4_cands(pid),
|
||||
provisional=getattr(parent_match, "provisional", False),
|
||||
)
|
||||
_apply_capacity_fit(c_pm, capacity_fit_fn)
|
||||
candidates.append(c_pm)
|
||||
@@ -624,6 +635,10 @@ def collect_candidates(sections, v4_lookup_fn, v4_label_to_status: dict,
|
||||
notes=notes,
|
||||
# rep_child 의 V4 후보 list (rep_match 와 같은 출처, frame_* 와 일관).
|
||||
v4_candidates=_v4_cands(rep_child.section_id),
|
||||
# IMP-30 u2 — rep_match drives frame selection so its provisional
|
||||
# flag flows here. If a non-rep child match is provisional but the
|
||||
# rep is not, this unit is not provisional (the rep frame is real).
|
||||
provisional=getattr(rep_match, "provisional", False),
|
||||
)
|
||||
_apply_capacity_fit(c_inf, capacity_fit_fn)
|
||||
candidates.append(c_inf)
|
||||
@@ -670,7 +685,13 @@ def score_candidate(c: CompositionUnit) -> CompositionUnit:
|
||||
|
||||
# ─── Selection ─────────────────────────────────────────────────
|
||||
|
||||
def select_composition_units(candidates, allowed_statuses: set[str]) -> list[CompositionUnit]:
|
||||
def select_composition_units(
|
||||
candidates,
|
||||
allowed_statuses: set[str],
|
||||
*,
|
||||
all_section_ids: Optional[list[str]] = None,
|
||||
allow_provisional_fill: bool = False,
|
||||
) -> list[CompositionUnit]:
|
||||
"""Greedy non-overlapping selection by score, with coverage tiebreak.
|
||||
|
||||
1. 모든 candidate 점수 매김
|
||||
@@ -685,6 +706,27 @@ def select_composition_units(candidates, allowed_statuses: set[str]) -> list[Com
|
||||
|
||||
auto_selectable=False candidate 는 자동 선택 X. debug 의 candidates_summary 에는 남음.
|
||||
UI/editor layer 에서 사용자가 별도 처리 가능 (현 v0 범위 X).
|
||||
|
||||
IMP-30 u3 — last-resort provisional fill (opt-in via allow_provisional_fill):
|
||||
After the normal greedy pass, sections in ``all_section_ids`` that are
|
||||
still uncovered are filled with the highest-score *provisional*
|
||||
candidate (``c.provisional == True``) that includes at least one
|
||||
uncovered section and does not collide with already-covered ones. A
|
||||
provisional candidate's backing V4Match was synthesized via
|
||||
``lookup_v4_match_with_fallback(allow_provisional=True)`` (IMP-30 u1)
|
||||
after chain_exhausted; its ``phase_z_status`` is therefore typically
|
||||
*outside* ``allowed_statuses`` (extract_matched_zone / fallback_candidate),
|
||||
which is why it gets filtered out of the normal greedy pass. The fill
|
||||
preserves first-render invariant for sections whose rank-1~3 are all
|
||||
restructure/reject. Default ``allow_provisional_fill=False`` keeps
|
||||
pre-u3 behavior byte-identical (IMP-05 regression guard).
|
||||
|
||||
Args:
|
||||
candidates: full candidate pool from collect_candidates().
|
||||
allowed_statuses: phase_z_status set considered auto-renderable.
|
||||
all_section_ids: ordered section id list (only consulted when
|
||||
allow_provisional_fill=True; required for coverage check).
|
||||
allow_provisional_fill: opt-in for last-resort provisional fill.
|
||||
"""
|
||||
scored = [score_candidate(c) for c in candidates]
|
||||
viable = [
|
||||
@@ -701,6 +743,28 @@ def select_composition_units(candidates, allowed_statuses: set[str]) -> list[Com
|
||||
selected.append(c)
|
||||
covered.update(c.source_section_ids)
|
||||
|
||||
# IMP-30 u3 — last-resort provisional fill (opt-in, default off).
|
||||
# Honors first-render invariant by surfacing chain_exhausted sections as
|
||||
# provisional zones instead of dropping them. Skip reasons on
|
||||
# non-provisional filtered candidates are preserved (not mutated here).
|
||||
if allow_provisional_fill and all_section_ids:
|
||||
uncovered = {sid for sid in all_section_ids if sid not in covered}
|
||||
if uncovered:
|
||||
provisional_pool = [
|
||||
c for c in scored
|
||||
if c.provisional
|
||||
and any(sid in uncovered for sid in c.source_section_ids)
|
||||
]
|
||||
provisional_pool.sort(
|
||||
key=lambda c: (c.score, len(c.source_section_ids)),
|
||||
reverse=True,
|
||||
)
|
||||
for c in provisional_pool:
|
||||
if any(sid in covered for sid in c.source_section_ids):
|
||||
continue
|
||||
selected.append(c)
|
||||
covered.update(c.source_section_ids)
|
||||
|
||||
return selected
|
||||
|
||||
|
||||
@@ -740,7 +804,9 @@ def select_layout_preset(units: list[CompositionUnit]) -> Optional[str]:
|
||||
def plan_composition(sections, v4_lookup_fn, v4_label_to_status: dict,
|
||||
allowed_statuses: set[str],
|
||||
capacity_fit_fn=None,
|
||||
v4_candidates_lookup_fn=None) -> tuple[list[CompositionUnit], Optional[str], dict]:
|
||||
v4_candidates_lookup_fn=None,
|
||||
*,
|
||||
allow_provisional_fill: bool = False) -> tuple[list[CompositionUnit], Optional[str], dict]:
|
||||
"""Composition planner v0.2 entry.
|
||||
|
||||
v0.2 변경 :
|
||||
@@ -753,6 +819,14 @@ def plan_composition(sections, v4_lookup_fn, v4_label_to_status: dict,
|
||||
logic 변화 X — 단일 frame_template_id / frame_id / label / confidence 는 그대로.
|
||||
runtime 결과 무변. Step 9 application_plan input 위한 schema 확장.
|
||||
|
||||
IMP-30 u3 — last-resort provisional fill (opt-in, default off):
|
||||
``allow_provisional_fill`` is plumbed to select_composition_units().
|
||||
When True, uncovered sections receive a provisional fill from candidates
|
||||
whose backing V4Match was synthesized via ``allow_provisional=True``
|
||||
(IMP-30 u1). ``_candidate_state`` returns ``selected_provisional`` for
|
||||
those filled units so the debug summary distinguishes greedy selections
|
||||
from provisional fills. Default False keeps IMP-05 behavior identical.
|
||||
|
||||
v0.1 / v0.1.1 동작 (유지) :
|
||||
- parent_merged_inferred candidate 생성 (parent V4 없어도)
|
||||
- review 개념 X. auto_selectable + filter_reasons 만으로 자동 결정
|
||||
@@ -771,11 +845,22 @@ def plan_composition(sections, v4_lookup_fn, v4_label_to_status: dict,
|
||||
)
|
||||
scored_all = [score_candidate(c) for c in candidates]
|
||||
|
||||
units = select_composition_units(candidates, allowed_statuses)
|
||||
units = select_composition_units(
|
||||
candidates,
|
||||
allowed_statuses,
|
||||
all_section_ids=[s.section_id for s in sections] if allow_provisional_fill else None,
|
||||
allow_provisional_fill=allow_provisional_fill,
|
||||
)
|
||||
preset = select_layout_preset(units)
|
||||
|
||||
def _candidate_state(c: CompositionUnit) -> str:
|
||||
if c in units:
|
||||
# IMP-30 u3 — provisional-fill units surface as a distinct state so
|
||||
# downstream debug consumers can tell greedy selection apart from
|
||||
# last-resort fill. unit.provisional flows from u1 (V4Match
|
||||
# synthesis) → u2 (CompositionUnit propagation).
|
||||
if c.provisional:
|
||||
return "selected_provisional"
|
||||
return "selected"
|
||||
if c.phase_z_status not in allowed_statuses:
|
||||
return "filtered_status" # V4 label → status not auto-renderable
|
||||
|
||||
@@ -32,6 +32,7 @@ import yaml
|
||||
|
||||
PROJECT_ROOT = Path(__file__).parent.parent
|
||||
CATALOG_PATH = PROJECT_ROOT / "templates" / "phase_z2" / "catalog" / "frame_contracts.yaml"
|
||||
V4_FALLBACK_POLICY_PATH = PROJECT_ROOT / "templates" / "phase_z2" / "catalog" / "v4_fallback_policy.yaml"
|
||||
|
||||
|
||||
class FitError(Exception):
|
||||
@@ -57,6 +58,44 @@ def get_contract(template_id: str) -> dict | None:
|
||||
return load_frame_contracts().get(template_id)
|
||||
|
||||
|
||||
# ─── V4 fallback policy loading (IMP-38) ──────────────────────────
|
||||
|
||||
_V4_FALLBACK_POLICY_CACHE: dict | None = None
|
||||
|
||||
_V4_FALLBACK_POLICY_DEFAULT: dict = {
|
||||
"policy_type": "static",
|
||||
"usable_threshold": 1,
|
||||
"default_max_rank": 3,
|
||||
"extended_max_rank": 3, # graceful: yaml 없을 시 확장 X (byte-identical to pre-IMP-38)
|
||||
}
|
||||
|
||||
|
||||
def load_v4_fallback_policy() -> dict:
|
||||
"""IMP-38 V4 fallback policy loader (separate yaml, catalog 오염 방지).
|
||||
|
||||
Returns dict with keys: policy_type, usable_threshold, default_max_rank, extended_max_rank.
|
||||
|
||||
Codex #1 권장: frame_contracts.yaml top-level 오염 회피 (별 yaml).
|
||||
Codex #3 LOCK: load_frame_contracts() shape 변경 X (이 함수는 별 cache).
|
||||
|
||||
Graceful fallback:
|
||||
yaml 파일 없을 시 → _V4_FALLBACK_POLICY_DEFAULT (default_max_rank=3, extended=3)
|
||||
→ backward compat byte-identical to pre-IMP-38 behavior.
|
||||
|
||||
Returns:
|
||||
dict — 정책 키 (정책 yaml 의 superset 가능, 알 수 없는 키는 무시 권장).
|
||||
"""
|
||||
global _V4_FALLBACK_POLICY_CACHE
|
||||
if _V4_FALLBACK_POLICY_CACHE is None:
|
||||
if V4_FALLBACK_POLICY_PATH.exists():
|
||||
loaded = yaml.safe_load(V4_FALLBACK_POLICY_PATH.read_text(encoding="utf-8")) or {}
|
||||
# merge with default (yaml 키 부분 누락 시 default 로 fall through)
|
||||
_V4_FALLBACK_POLICY_CACHE = {**_V4_FALLBACK_POLICY_DEFAULT, **loaded}
|
||||
else:
|
||||
_V4_FALLBACK_POLICY_CACHE = dict(_V4_FALLBACK_POLICY_DEFAULT)
|
||||
return _V4_FALLBACK_POLICY_CACHE
|
||||
|
||||
|
||||
# ─── Source-shape splitters ──────────────────────────────────────
|
||||
|
||||
def _split_top_bullets(content: str) -> list[tuple[str, list[str]]]:
|
||||
|
||||
+1013
-142
File diff suppressed because it is too large
Load Diff
+22
-1
@@ -120,11 +120,30 @@ def plan_zone_ratio_retry(
|
||||
continue
|
||||
|
||||
# rule 4-(d) 현재 height > min_height
|
||||
# IMP-34 u1: donor capacity bounded by measured empty space
|
||||
# (clientHeight - scrollHeight from Step 14) when both fields are present,
|
||||
# falling back to static contract slack when absent. Prevents the donor
|
||||
# from being over-allocated when it is already full but not overflowing.
|
||||
height = zones_before.get(pos)
|
||||
min_h = zone_min_by_pos.get(pos)
|
||||
if height is None or min_h is None:
|
||||
continue
|
||||
slack = height - min_h
|
||||
static_slack = height - min_h
|
||||
client_h = zinfo.get("clientHeight")
|
||||
scroll_h = zinfo.get("scrollHeight")
|
||||
if (
|
||||
isinstance(client_h, (int, float))
|
||||
and isinstance(scroll_h, (int, float))
|
||||
and not isinstance(client_h, bool)
|
||||
and not isinstance(scroll_h, bool)
|
||||
):
|
||||
measured_empty_px = max(0, int(client_h) - int(scroll_h))
|
||||
slack = min(static_slack, measured_empty_px)
|
||||
slack_bound_source = "measured_bound"
|
||||
else:
|
||||
measured_empty_px = None
|
||||
slack = static_slack
|
||||
slack_bound_source = "static_fallback"
|
||||
if slack <= 0:
|
||||
continue
|
||||
|
||||
@@ -134,6 +153,8 @@ def plan_zone_ratio_retry(
|
||||
"min_height": min_h,
|
||||
"slack": slack,
|
||||
"capacity_fit_status": cap_status,
|
||||
"measured_empty_px": measured_empty_px,
|
||||
"slack_bound_source": slack_bound_source,
|
||||
})
|
||||
|
||||
# rule 4-(f) 여러 후보면 slack 가장 큰 것부터
|
||||
|
||||
+4
-17
@@ -36,6 +36,7 @@ from src.image_utils import get_image_sizes, embed_images
|
||||
from src.space_allocator import calculate_container_specs
|
||||
from src.slide_measurer import measure_rendered_heights, capture_slide_screenshot
|
||||
from src.config import settings
|
||||
from src.json_utils import parse_json as _parse_json
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -1182,6 +1183,7 @@ async def generate_slide(
|
||||
yield {"event": "progress", "data": "3/7 슬라이드 HTML 생성 중..."}
|
||||
|
||||
async def stage_2(context: PipelineContext) -> dict:
|
||||
# [legacy Phase R'/Q example — INTEGRATION-AUDIT-01 §10.4]
|
||||
# Phase X-BX': Type B는 code_assembled 직접 사용, Sonnet 재구성 스킵
|
||||
if context.analysis.layout_template in ("B", "B'", "B''"):
|
||||
from src.block_assembler import assemble_slide_html_final
|
||||
@@ -1190,6 +1192,7 @@ async def generate_slide(
|
||||
logger.info(f"[Stage 2] Type B: slide-base + 블록 (font_scale={fs:.1f})")
|
||||
return {"generated_html": generated}
|
||||
|
||||
# [legacy Phase R'/Q example — INTEGRATION-AUDIT-01 §10.4]
|
||||
# Type A: 기존 Sonnet 재구성 코드 그대로
|
||||
from src.content_verifier import generate_with_retry
|
||||
|
||||
@@ -1998,6 +2001,7 @@ async def _apply_adjustments(
|
||||
block["detail_target"] = True
|
||||
if "data" in block:
|
||||
del block["data"]
|
||||
# [legacy Phase R'/Q example — INTEGRATION-AUDIT-01 §10.4]
|
||||
block["reason"] = f"재구성: {detail}"
|
||||
logger.info(
|
||||
f"조정: {area} → kei_restructure (detail_target)"
|
||||
@@ -2077,20 +2081,3 @@ def _convert_kei_judgment(
|
||||
new_adjs.append(adj)
|
||||
|
||||
review_result["adjustments"] = new_adjs
|
||||
|
||||
|
||||
def _parse_json(text: str) -> dict[str, Any] | None:
|
||||
"""텍스트에서 JSON을 추출한다."""
|
||||
patterns = [
|
||||
r"```json\s*(.*?)```",
|
||||
r"```\s*(.*?)```",
|
||||
r"(\{.*\})",
|
||||
]
|
||||
for pattern in patterns:
|
||||
match = re.search(pattern, text, re.DOTALL)
|
||||
if match:
|
||||
try:
|
||||
return json.loads(match.group(1).strip())
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
return None
|
||||
|
||||
+39
-49
@@ -13,86 +13,76 @@ from collections import OrderedDict
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import yaml
|
||||
from jinja2 import Environment, FileSystemLoader
|
||||
|
||||
from src import catalog as _catalog_mod
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
TEMPLATES_DIR = Path(__file__).parent.parent / "templates"
|
||||
STATIC_DIR = Path(__file__).parent.parent / "static"
|
||||
CATALOG_PATH = TEMPLATES_DIR / "catalog.yaml"
|
||||
|
||||
# 카테고리 검색 순서
|
||||
BLOCK_CATEGORIES = ["headers", "cards", "tables", "visuals", "emphasis", "media"]
|
||||
|
||||
# catalog.yaml에서 id → template 경로 매핑 로드 (BF-10: mtime 체크로 자동 갱신)
|
||||
# id → template 경로 매핑 (IMP-27: src.catalog 공유 로더 위임, renderer-local projection cache)
|
||||
_CATALOG_MAP: dict[str, str] | None = None
|
||||
_CATALOG_MTIME: float = 0.0
|
||||
_CATALOG_MAP_MTIME: float = 0.0
|
||||
|
||||
# Phase R: variant별 template 경로 캐시 (renderer-local projection)
|
||||
_CATALOG_VARIANT_MAP: dict[str, str] | None = None
|
||||
_CATALOG_VARIANT_MAP_MTIME: float = 0.0
|
||||
|
||||
|
||||
def _load_catalog_map() -> dict[str, str]:
|
||||
"""catalog.yaml에서 블록 id → template 경로 매핑을 로드한다.
|
||||
"""블록 id → template 경로 projection (IMP-27: src.catalog 공유 로더 위임).
|
||||
|
||||
파일 수정시간(mtime)을 확인하여, 변경 시에만 재로드한다.
|
||||
catalog 파일 읽기와 mtime 캐싱은 ``src.catalog`` 가 단독 소유. 본 함수는
|
||||
그 결과를 ``id → template`` 형태로 변환한 renderer-local projection 캐시만
|
||||
유지하며, projection 무효화는 ``src.catalog.get_catalog_mtime()`` 키잉.
|
||||
"""
|
||||
global _CATALOG_MAP, _CATALOG_MTIME
|
||||
global _CATALOG_MAP, _CATALOG_MAP_MTIME
|
||||
|
||||
current_mtime = CATALOG_PATH.stat().st_mtime if CATALOG_PATH.exists() else 0.0
|
||||
blocks = _catalog_mod.load_blocks()
|
||||
current_mtime = _catalog_mod.get_catalog_mtime()
|
||||
|
||||
if _CATALOG_MAP is not None and _CATALOG_MTIME == current_mtime:
|
||||
return _CATALOG_MAP # 파일 변경 없음 → 캐시 재사용
|
||||
if _CATALOG_MAP is not None and _CATALOG_MAP_MTIME == current_mtime:
|
||||
return _CATALOG_MAP
|
||||
|
||||
# 변경 감지 또는 첫 로드 → 새로 읽기
|
||||
_CATALOG_MTIME = current_mtime
|
||||
_CATALOG_MAP_MTIME = current_mtime
|
||||
_CATALOG_MAP = {}
|
||||
if CATALOG_PATH.exists():
|
||||
try:
|
||||
with open(CATALOG_PATH, encoding="utf-8") as f:
|
||||
catalog = yaml.safe_load(f)
|
||||
for block in catalog.get("blocks", []):
|
||||
block_id = block.get("id", "")
|
||||
template = block.get("template", "")
|
||||
if block_id and template:
|
||||
_CATALOG_MAP[block_id] = template
|
||||
logger.info(f"catalog.yaml 로드: {len(_CATALOG_MAP)}개 블록 매핑")
|
||||
except Exception as e:
|
||||
logger.warning(f"catalog.yaml 로드 실패: {e}")
|
||||
else:
|
||||
logger.warning(f"catalog.yaml 미발견: {CATALOG_PATH}")
|
||||
for block in blocks:
|
||||
block_id = block.get("id", "")
|
||||
template = block.get("template", "")
|
||||
if block_id and template:
|
||||
_CATALOG_MAP[block_id] = template
|
||||
logger.info(f"catalog.yaml 로드: {len(_CATALOG_MAP)}개 블록 매핑")
|
||||
|
||||
return _CATALOG_MAP
|
||||
|
||||
|
||||
# Phase R: variant별 template 경로 캐시
|
||||
_CATALOG_VARIANT_MAP: dict[str, str] | None = None
|
||||
|
||||
|
||||
def _load_catalog_map_with_variants() -> dict[str, str]:
|
||||
"""catalog.yaml에서 variant별 template 경로 매핑을 로드한다.
|
||||
"""variant별 template 경로 projection (IMP-27: src.catalog 공유 로더 위임).
|
||||
|
||||
키: "block_id--variant_id" → 값: template 경로
|
||||
키: "block_id--variant_id" → 값: template 경로.
|
||||
"""
|
||||
global _CATALOG_VARIANT_MAP
|
||||
global _CATALOG_VARIANT_MAP, _CATALOG_VARIANT_MAP_MTIME
|
||||
|
||||
# _load_catalog_map이 이미 캐시 관리하므로 같은 mtime 사용
|
||||
_load_catalog_map() # 캐시 갱신 보장
|
||||
blocks = _catalog_mod.load_blocks()
|
||||
current_mtime = _catalog_mod.get_catalog_mtime()
|
||||
|
||||
if _CATALOG_VARIANT_MAP is not None and _CATALOG_MTIME == (CATALOG_PATH.stat().st_mtime if CATALOG_PATH.exists() else 0.0):
|
||||
if _CATALOG_VARIANT_MAP is not None and _CATALOG_VARIANT_MAP_MTIME == current_mtime:
|
||||
return _CATALOG_VARIANT_MAP
|
||||
|
||||
_CATALOG_VARIANT_MAP_MTIME = current_mtime
|
||||
_CATALOG_VARIANT_MAP = {}
|
||||
if CATALOG_PATH.exists():
|
||||
try:
|
||||
with open(CATALOG_PATH, encoding="utf-8") as f:
|
||||
catalog = yaml.safe_load(f)
|
||||
for block in catalog.get("blocks", []):
|
||||
block_id = block.get("id", "")
|
||||
for variant in block.get("variants", []):
|
||||
vid = variant.get("id", "default")
|
||||
vtemplate = variant.get("template", "")
|
||||
if vid != "default" and vtemplate:
|
||||
_CATALOG_VARIANT_MAP[f"{block_id}--{vid}"] = vtemplate
|
||||
except Exception as e:
|
||||
logger.warning(f"catalog variant 로드 실패: {e}")
|
||||
for block in blocks:
|
||||
block_id = block.get("id", "")
|
||||
for variant in block.get("variants", []):
|
||||
vid = variant.get("id", "default")
|
||||
vtemplate = variant.get("template", "")
|
||||
if vid != "default" and vtemplate:
|
||||
_CATALOG_VARIANT_MAP[f"{block_id}--{vid}"] = vtemplate
|
||||
|
||||
return _CATALOG_VARIANT_MAP
|
||||
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,45 @@
|
||||
# IMP-38 V4 max_rank 정책 — separate yaml (catalog 오염 방지)
|
||||
#
|
||||
# 도입 배경:
|
||||
# 기존 `lookup_v4_match_with_fallback(max_rank=3)` hardcoded → rank 4~32 의 등록 frame 도달 못함
|
||||
# mdx05-2 같이 V4 rank 1~9 가 catalog 미등록 + rank 10~ 등록 case → chain_exhausted → unit 생성 X
|
||||
#
|
||||
# 4 round 합의 (IMP-38 #67):
|
||||
# - Codex #1: frame_contracts.yaml 오염 회피 → 별 yaml 파일 (이 파일)
|
||||
# - Codex #2: 3 변수 분리 (configured / judgments / catalog count)
|
||||
# - Codex #3: effective_extended_ceiling = min(configured, len(judgments_full32))
|
||||
#
|
||||
# 적용 path: src/phase_z2_mapper.py 의 load_v4_fallback_policy() loader
|
||||
# + src/phase_z2_pipeline.py 의 lookup_v4_match_with_fallback() 동적 max_rank logic
|
||||
|
||||
policy_type: dynamic_usable_count_based
|
||||
|
||||
# usable_threshold N:
|
||||
# rank 1~default_max_rank 중 "usable" predicate 충족 frame 수 >= N → default_max_rank 유지
|
||||
# < N → extended_max_rank 로 확장
|
||||
usable_threshold: 1
|
||||
|
||||
# default_max_rank:
|
||||
# normal case (usable_count >= threshold) 의 fallback chain 길이
|
||||
# mdx03 같이 rank 1 use_as_is 매칭 잘 되는 case 보호
|
||||
default_max_rank: 3
|
||||
|
||||
# extended_max_rank:
|
||||
# usable_count < threshold case 의 확장 ceiling
|
||||
# mdx05-2 같이 rank 1~9 미등록 case 처리
|
||||
# ★ 실제 effective_extended_ceiling = min(extended_max_rank, len(judgments_full32))
|
||||
# (Codex #2 정정: yaml ceiling 무력화 방지 + V4 schema 범위 초과 방지)
|
||||
extended_max_rank: 32
|
||||
|
||||
# usable predicate (3-tier):
|
||||
# (a) phase_z_status in MVP1_ALLOWED_STATUSES (matched_zone / adapt_matched_zone)
|
||||
# (b) get_contract(template_id) is not None (catalog 등록)
|
||||
# (c) capacity_fit ok (raw_content 제공 시만 — optional)
|
||||
|
||||
# 의미 신뢰 vs catalog presence trade-off:
|
||||
# N=1 = 가장 보수 (rank 1 usable 시 확장 X — mdx03 정상 case 보호)
|
||||
# default_max_rank=3 = 의미 신뢰 범위 (V4 rank 1~3)
|
||||
# extended_max_rank=32 = catalog presence fallback (rank 4~32)
|
||||
|
||||
# graceful fallback (yaml 없을 시):
|
||||
# loader 가 default {default_max_rank: 3, extended_max_rank: 3} 로 fall through (backward compat)
|
||||
@@ -0,0 +1,38 @@
|
||||
# Phase Z Families — WIP Marker
|
||||
|
||||
**Status:** intentionally untracked, uncontracted, out-of-scope for runtime matcher.
|
||||
**Closes audit follow-up:** INTEGRATION-AUDIT-01-REPORT.md §10.2 F-2 (option c).
|
||||
**Gate for promote/remove:** Gitea issue #42 (`IMP-04b Catalog extension to 32 frames`).
|
||||
|
||||
## Baseline lock (2026-05-19)
|
||||
|
||||
| Surface | Count | Notes |
|
||||
|---|---|---|
|
||||
| `git ls-files templates/phase_z2/families/*.html` | 11 | tracked family templates |
|
||||
| `templates/phase_z2/catalog/frame_contracts.yaml` top-level keys | 11 | 1:1 with tracked basenames |
|
||||
| `Get-ChildItem templates/phase_z2/families/*.html` | 13 | tracked 11 + WIP 2 (below) |
|
||||
|
||||
Active contracted family count = **11**. Tracked basenames ↔ `frame_contracts.yaml` top-level keys are set-equal. Drift between disk (13) and contracted (11) is fully explained by the 2 WIP files below.
|
||||
|
||||
## WIP family templates (uncontracted)
|
||||
|
||||
| File | Figma frame | Status |
|
||||
|---|---|---|
|
||||
|
||||
> **#42 IMP-04b u3 (2026-05-21)** — frame 23 (`1171281203`) partial absorbed → `frame_contracts.yaml::app_sw_package_vs_solution`. Counts in `## Baseline lock (2026-05-19)` are historical (post-u3: 11 tracked + 1 WIP / 12 contract).
|
||||
> **#42 IMP-04b u4 (2026-05-21)** — frame 9 (`1171281180`) partial absorbed → `frame_contracts.yaml::pre_construction_model_info_stacked`. WIP family table now empty (post-u4: 13 tracked / 13 contract).
|
||||
|
||||
These files are partials authored during Phase Z-2 MVP-1.5b exploration. They are **not** part of the contracted Phase Z runtime catalog and must not be enumerated by frame selection, matcher, or any Stage 3 pipeline surface.
|
||||
|
||||
## Rules
|
||||
|
||||
- Adding a file to `templates/phase_z2/families/*.html` without a matching `frame_contracts.yaml` entry is **only** permitted if it is named here as WIP.
|
||||
- `tests/test_family_contract_baseline.py` enforces this invariant: tracked families ↔ `frame_contracts.yaml` keys must be set-equal, modulo the WIP allowlist parsed from this file.
|
||||
- Promoting a WIP file (add `frame_contracts.yaml` entry + register with matcher) or removing it must happen under issue #42 or a follow-up issue, not silently.
|
||||
|
||||
## References
|
||||
|
||||
- `docs/architecture/INTEGRATION-AUDIT-01-REPORT.md` §10.2 F-2 (audit finding closed by issue #52, option c)
|
||||
- `docs/architecture/IMP-18-SVG-GAP-REPORT.md` L28, L51 (count basis corrected to `11 contracted + 2 WIP`)
|
||||
- Gitea issue #52 (this reconciliation)
|
||||
- Gitea issue #42 (pre-flight gate for promote/remove)
|
||||
@@ -114,6 +114,43 @@
|
||||
min-height: 0;
|
||||
}
|
||||
|
||||
/* ── IMP-30 u5 : provisional zone marker (first-render invariant) ──
|
||||
When V4 rank-1 candidate falls outside MVP1_ALLOWED_STATUSES (chain_exhausted)
|
||||
the pipeline still renders the rank-1 frame so the first-render invariant
|
||||
holds, but the zone is tagged `provisional` so the user/AI can adapt later
|
||||
(IMP-31). Visual contract:
|
||||
- dashed amber border + striped wash → "needs adaptation" at a glance
|
||||
- inline badge top-right → text label for non-color-perceiving readers
|
||||
MDX content is preserved as-is; no shrink, no rewrite. */
|
||||
.zone--provisional {
|
||||
outline: 2px dashed #b8860b;
|
||||
outline-offset: -2px;
|
||||
background-image: repeating-linear-gradient(
|
||||
45deg,
|
||||
rgba(184, 134, 11, 0.04) 0,
|
||||
rgba(184, 134, 11, 0.04) 8px,
|
||||
transparent 8px,
|
||||
transparent 16px
|
||||
);
|
||||
}
|
||||
.zone--provisional .zone__needs-adaptation-badge {
|
||||
position: absolute;
|
||||
top: 4px;
|
||||
right: 4px;
|
||||
z-index: 10;
|
||||
padding: 2px 6px;
|
||||
background: #b8860b;
|
||||
color: #fff;
|
||||
font-size: 9px;
|
||||
font-weight: 700;
|
||||
line-height: 1.2;
|
||||
letter-spacing: 0.04em;
|
||||
border-radius: 2px;
|
||||
text-transform: uppercase;
|
||||
pointer-events: none;
|
||||
box-shadow: 0 1px 2px rgba(0, 0, 0, 0.15);
|
||||
}
|
||||
|
||||
/* ── Frame-family text layout contract (shared, reusable) ──
|
||||
feedback-1 (mvp1.5b_test7): visible improvement 강화.
|
||||
Stronger hanging indent + breathing line spacing + visible hierarchy. */
|
||||
@@ -264,7 +301,8 @@
|
||||
<div class="slide-body">
|
||||
<div class="layout-{{ layout_preset }}">
|
||||
{% for zone in zones %}
|
||||
<div class="zone" data-zone-position="{{ zone.position }}" data-template-id="{{ zone.template_id }}" style="grid-area: {{ zone.position }};">
|
||||
<div class="zone{% if zone.provisional %} zone--provisional{% endif %}" data-zone-position="{{ zone.position }}" data-template-id="{{ zone.template_id }}"{% if zone.provisional %} data-provisional="1"{% endif %} style="grid-area: {{ zone.position }};">
|
||||
{% if zone.provisional %}<span class="zone__needs-adaptation-badge" aria-label="needs user or AI adaptation">needs adaptation</span>{% endif %}
|
||||
{{ zone.partial_html | safe }}
|
||||
</div>
|
||||
{% endfor %}
|
||||
|
||||
+175
@@ -0,0 +1,175 @@
|
||||
# CLAUDE.md — 매칭 시스템 작업 컨텍스트
|
||||
|
||||
이 파일은 Claude (AI) 가 `tests/` 디렉토리에서 작업할 때 참고하는 컨텍스트입니다.
|
||||
프로젝트 루트의 [../CLAUDE.md](../CLAUDE.md) 와 함께 사용.
|
||||
|
||||
## 작업 디렉토리
|
||||
|
||||
- 메인 작업 디렉토리: `tests/matching/`
|
||||
- 데이터 / 보고서 파일도 같은 위치
|
||||
- 실행 시 항상 `tests/matching/` 에서 (스크립트 내부 상대 경로 의존)
|
||||
|
||||
## 시스템 개요
|
||||
|
||||
MDX 콘텐츠 ↔ Figma Frame 32 개를 매칭하는 4 단계 파이프라인 (V1~V4).
|
||||
상세는 `README.md` / `PLAN.md` / `PROGRESS.md` 참조.
|
||||
|
||||
## 절대 규칙
|
||||
|
||||
### 1. 하드코딩 금지 (사용자 강조 사항)
|
||||
- 결과물을 직접 고치지 말고 **프로세스/코드를 고쳐라**
|
||||
- 임의 데이터 삽입 금지 (예: DECK 04 의 "제목·A 라벨·B 라벨·행 데이터" placeholder 사용 금지)
|
||||
- 모든 표시값은 실제 코드 결과 (yaml / 함수 출력) 에서 가져와야 함
|
||||
|
||||
### 2. 사용자 직접 수정 보존
|
||||
- 사용자가 HTML 파일을 직접 편집한 경우 **반드시 pipeline 코드에 반영** 후 재생성
|
||||
- 코드만 고치고 재실행하면 사용자 수정이 사라짐
|
||||
- 변경 시: 사용자 수정 8 개 모두 코드에 반영 → 재실행
|
||||
|
||||
### 3. 정직한 코드 동작 표시
|
||||
- 임원 보고용 deck 라도 **코드의 한계를 솔직히 표시**
|
||||
- 예: "MDX 자동 분석 결과 — 정책/요구사항 (사람이 보면 행렬형 비교)" 같은 표기
|
||||
- "이 축은 사실 frame 매칭에 영향 없음" 같은 ablation 결과는 임원용에는 빼지만, 내부 문서 (PROGRESS.md) 에는 명시
|
||||
|
||||
### 4. 임원 보고용 톤
|
||||
- 영문 enum 코드 (`policy_requirements`) 직접 노출 금지 — 한글 (정책/요구사항) 우선
|
||||
- 매칭 키워드는 5~10 개 + "등 N 개" 로 축약
|
||||
- 디자인: 그라데이션 / 화려한 카드 금지. 단순 표 + 흑백 + 강조 색 1~2 가지
|
||||
- 정의 / 부연 설명 최소화 (def 는 한 줄, 길게 풀어 쓰지 말 것)
|
||||
|
||||
## 명명 규칙
|
||||
|
||||
### 파이프라인 스크립트
|
||||
```
|
||||
pipeline_<숫자>_<이름>.py
|
||||
```
|
||||
- 01~07: 입력 추출 + 전처리 + 키워드
|
||||
- 08: V2 (semantic) / V3 (structure r2~r5) / V4 (template_fit, r1/r2)
|
||||
- 09: V2 진단
|
||||
- 10: Holdout 라벨링 / 평가
|
||||
- 11: templates_v1 감사
|
||||
- 12: templates_v2 생성 (r1, r2, r3, final, final_r2, promote_frame13)
|
||||
- 13: meeting docs / samples
|
||||
- 14: single sample
|
||||
- 15: bm25 / idf / logistic regression 비교
|
||||
- 16: deck 페이지 생성 (DECK 1~7)
|
||||
- 17: V4 full32 (32 frame 전체 평가)
|
||||
- 18: V4 slot 축 ablation
|
||||
|
||||
### 결과 파일
|
||||
```
|
||||
<단계>_<설명>_result.yaml
|
||||
```
|
||||
- `mdx_matching_result.yaml` — V1
|
||||
- `v2_semantic_rerank_result.yaml` — V2
|
||||
- `v3_structure_rerank_r5_result.yaml` — V3 (최종 r5)
|
||||
- `v4_full32_result.yaml` — V4 (32 frame 전체)
|
||||
- `structure_ontology_v2_final_r2.yaml` — Frame 32 DB
|
||||
|
||||
### 보고서
|
||||
```
|
||||
DECK_<번호>_<이름>.html — 임원 보고용 A4 페이지
|
||||
ATTACH_<번호>_<이름>.html — 부속 자료
|
||||
<NAME>_REPORT.html / .md — 분석 보고서
|
||||
```
|
||||
|
||||
## 자주 쓰는 명령어
|
||||
|
||||
```bash
|
||||
cd tests/matching/
|
||||
|
||||
# 매칭 시스템 전체 재실행
|
||||
python pipeline_06_2_mdx_matching.py
|
||||
python pipeline_08_v2_semantic_rerank.py
|
||||
python pipeline_08_v3_r5_structure_rerank.py
|
||||
python pipeline_17_v4_full32.py
|
||||
|
||||
# 보고서 재생성
|
||||
python pipeline_16_deck_4pages.py # DECK 1~7
|
||||
|
||||
# Ablation / 검증
|
||||
python pipeline_15_logistic_regression.py
|
||||
python pipeline_18_slot_axis_ablation.py
|
||||
```
|
||||
|
||||
## 파이프라인 핵심 가중치
|
||||
|
||||
### V1 키워드 매칭 (Logistic Regression 학습)
|
||||
```
|
||||
matching_score = 0.414 × 핵심 + 0.320 × 세트 + 0.265 × 연관
|
||||
```
|
||||
|
||||
### V3 구조 매칭
|
||||
```
|
||||
total = 0.40 × 레이아웃 일치 + 0.35 × 콘텐츠 성격 + 0.25 × 시각 의도
|
||||
```
|
||||
|
||||
### V4 종합 판정
|
||||
```
|
||||
confidence = 0.25 × anchor + 0.20 × cardinality + 0.20 × relation
|
||||
+ 0.15 × slot + 0.20 × content − penalty
|
||||
|
||||
라벨 임계값:
|
||||
≥ 0.90 → use_as_is (그대로 사용)
|
||||
≥ 0.75 → light_edit (가벼운 편집)
|
||||
≥ 0.60 → restructure (구조 재배치)
|
||||
< 0.60 → reject (사용 불가)
|
||||
```
|
||||
|
||||
## 데이터 소스
|
||||
|
||||
| 데이터 | 위치 | 용도 |
|
||||
|---|---|---|
|
||||
| Figma 텍스트 | `figma_to_html_agent/blocks/*/texts.md` | 32 frame 텍스트 추출 |
|
||||
| BEPS 마스터 | (별도 위치) | 키워드 보강용 |
|
||||
| MDX 검증 구간 | (`pipeline_01_extract_nodes.py` 의 `MDX_SECTIONS`) | 정답 매칭 검증 |
|
||||
| Frame 이미지 | `data/figma_previews/<프레임번호>.png` | DECK 시각화 |
|
||||
|
||||
## 테스트 픽스처 컨벤션 (F-5, INTEGRATION-AUDIT-01 §10.5.1)
|
||||
|
||||
테스트 데이터 / 샘플 참조의 정식 위치 규약. `tests/` 안에서만 적용되고 `src/**` 프로덕션 경로에는 적용되지 않음.
|
||||
|
||||
| 경로 | 상태 | 용도 | 비고 |
|
||||
|---|---|---|---|
|
||||
| `tests/phase_z2/fixtures/` | **존재 (정식)** | Phase Z 회귀 YAML 픽스처 | `test_fixtures_loader.py` 가 로드. 서브디렉토리 : `build_layout_css/`, `retry_gate/`. |
|
||||
| `tests/fixtures/` (루트) | **없음 (현재 미생성)** | 비-Phase-Z / 비-YAML 픽스처 미래 후보 | 샘플 인벤토리가 `tests/phase_z2/test_*.py` 인라인으로 감당 못 할 때만 별도 이슈로 신설. |
|
||||
| `samples/mdx_batch/**` , `samples/mdx/**` | 존재 | 통합 스모크 입력 | `tests/**` 에서만 참조 가능. `src/**` 런타임 경로 하드코딩 금지. |
|
||||
|
||||
규칙 :
|
||||
|
||||
- 테스트 코드에서는 `samples/mdx_batch/02.mdx` 같은 샘플 MDX 를 직접 참조해도 됨 (예 : `tests/phase_z2/test_pz2_vu_integration.py`). `src/**` 런타임 입력은 절대 샘플 파일명 / 콘텐츠를 핀하지 말 것.
|
||||
- 새 YAML 회귀 픽스처는 `tests/phase_z2/fixtures/` 아래 새 서브디렉토리로 추가. 루트 `tests/fixtures/` 신설은 금지 (별도 이슈 필요).
|
||||
- `src/**` 안에 등장하는 "BIM" / "건설산업 DX" / "재구성" 같은 sample-like 리터럴은 INTEGRATION-AUDIT-01 §10.4 (F-4) 에서 의도된 docstring / glossary / 예시 dict 로 분류 완료. annotation marker 가 붙어 있으면 의도된 example. 새 sample 리터럴을 `src/**` 에 도입하지 말 것.
|
||||
- 본 컨벤션의 anchor 정의는 `docs/architecture/INTEGRATION-AUDIT-01-REPORT.md` §10.5.1. 변경 시 anchor 부터 갱신.
|
||||
|
||||
## 자주 헷갈리는 것
|
||||
|
||||
### 영문 enum vs 한글 매핑
|
||||
- 코드 / yaml: 영문 enum (`comparative_matrix`, `cycle_interrelation`)
|
||||
- 보고서 표시: 한글 (`행렬형 비교`, `순환/상호 관계`)
|
||||
- DECK 05 의 키워드 사전 표는 양쪽 다 표시 (사용자 매칭 가능)
|
||||
|
||||
### 항목수 vs 슬롯 후보 개수
|
||||
- **동일** — `item_count = len(slot_candidates)` (표 / subsections / bullets 어떤 형태든)
|
||||
- V4 의 cardinality 축과 slot.within 부분은 **같은 신호의 중복 가중** (ablation 으로 확인)
|
||||
|
||||
### V3 vs V4 구조 점수
|
||||
- V3 = layout family + content_affinity + structure_intent (3 축)
|
||||
- V4 = anchor + cardinality + relation + slot + content (5 축)
|
||||
- **다른 모델**. V3 점수와 V4 confidence 는 별도 계산
|
||||
|
||||
## 사용자가 강조한 피드백
|
||||
|
||||
- "코드로 돌린 결과물이지 임의 데이터 아님" — 모든 표시 정직
|
||||
- "임원 보고용이야" — 부정적 부연 / 디테일 산식 빼기
|
||||
- "한가지만 해" — 한 번에 한 가지만 변경
|
||||
- "모든 변경은 pipeline 코드에 반영" — HTML 직접 수정은 일시적
|
||||
|
||||
## 진행 중 발견된 약점
|
||||
|
||||
`PROGRESS.md` 의 "발견된 약점" 표 참조. 8 개 모두 Phase E 작업 대상.
|
||||
|
||||
가장 시급:
|
||||
1. **02-2.2 매칭 실패** (E.5)
|
||||
2. **MDX 분석 LLM 화** (E.1, E.2)
|
||||
3. **슬롯 의미 매핑** (E.3, E.4)
|
||||
@@ -0,0 +1,90 @@
|
||||
# IMP-47A — mdx03 frontend stabilization manual e2e
|
||||
|
||||
Scope: frontend-only. Backend pipeline must NOT be modified during this test.
|
||||
Path under test: `mdx=03` (default sample loaded on page open).
|
||||
|
||||
## Preconditions
|
||||
|
||||
- Backend running on `http://localhost:8001` (`uvicorn src.main:app --port 8001`).
|
||||
- Frontend dev server running (`cd Front && npm run dev`).
|
||||
- Working tree at IMP-47A Stage 3 HEAD (u1+u2+u3 applied to `Front/client/src/components/SlideCanvas.tsx`, `Front/client/src/services/designAgentApi.ts`, `Front/client/src/pages/Home.tsx`).
|
||||
- Browser opens `http://localhost:5173/?mdx=03` (or default route, which auto-loads `mdx=03`).
|
||||
|
||||
## Section 1 — iframe rendering (axis 1)
|
||||
|
||||
Goal: verify u1 sandbox change lets `slide_base.html` script apply `html.embedded` class so the slide is not clipped inside the iframe.
|
||||
|
||||
Steps:
|
||||
1. Open the app; wait for `mdx=03` auto-load and initial `final.html` render.
|
||||
2. Click the "슬라이드 플랜 생성하기" button; wait for `run "<id>" 완료` toast.
|
||||
3. Open browser DevTools → Elements; locate the iframe inside `SlideCanvas`; switch context to the iframe document.
|
||||
4. Confirm `<html class="embedded">` is present (not just `<html>`).
|
||||
5. Confirm the rendered slide content fills the 1280×720 frame with no top padding offset and no clipping at the bottom.
|
||||
|
||||
Pass: `html.embedded` class present AND no visible vertical shift/clipping.
|
||||
Fail signal: iframe content pushed downward, footer cut off, or `html` lacks `embedded` class (means script never ran → sandbox regression).
|
||||
|
||||
## Section 2 — multi-source frame candidates (axis 2)
|
||||
|
||||
Goal: verify u2 3-source merge surfaces candidates from `candidate_evidence`, `v4_all_judgments`, and `v4_candidates` with deterministic dedup and cap.
|
||||
|
||||
Steps:
|
||||
1. After Section 1 success, click any zone in the canvas; right panel switches to the "frame" tab.
|
||||
2. In the frame candidate list, count visible candidates. Expect ≤ `TOP_N_FRAMES` (=6).
|
||||
3. Open DevTools → Network → reload `/api/run/<id>`; inspect the JSON response and confirm at least two of `candidate_evidence`, `v4_all_judgments`, `v4_candidates` are non-empty.
|
||||
4. Cross-check: union of `template_id ?? id ?? frame_id` keys from all three arrays (deduped, capped at 6) equals the UI list count and order.
|
||||
5. Confirm the order respects LABEL_PRIORITY (`use_as_is` < `light_edit` < `restructure` < `reject`) then descending confidence.
|
||||
|
||||
Pass: union/dedup/cap/order all match.
|
||||
Fail signal: only candidates from a single source visible, duplicates by template_id, more than 6 items, or order violates LABEL_PRIORITY.
|
||||
|
||||
## Section 3 — frame / layout override regeneration (axis 3)
|
||||
|
||||
Goal: verify u3 5-dep `handleGenerate` callback delivers the latest override state to backend (no stale closure).
|
||||
|
||||
Steps:
|
||||
1. From Section 2, pick a non-default frame candidate (one whose label is not `use_as_is`); click "이 프레임 적용".
|
||||
2. Confirm bottom-left button transforms to "선택대로 재생성하기" with amber pulse dot (hasPendingChanges = true).
|
||||
3. Without any extra clicks, click "선택대로 재생성하기".
|
||||
4. Open DevTools → Network → inspect the POST `/api/pipeline` body; confirm `overrides.frames` contains the chosen `{unit_id: frame_id}` mapping.
|
||||
5. After success toast, confirm new `run_id` differs from the previous run, and the rendered iframe reflects the chosen frame (frame DOM class / id matches selection).
|
||||
6. Repeat with a layout-card "적용하기" → pending overlay enters → "선택대로 재생성하기"; confirm POST body includes `overrides.layout` with the chosen preset id.
|
||||
|
||||
Pass: every override (frames / layout / zoneSections / zoneGeometries when applicable) reaches backend on first click; new `run_id` returned.
|
||||
Fail signal: POST body lacks the override, or backend re-renders with the previous selection (stale closure regression).
|
||||
|
||||
## Section 4 — pending overlay enter / cancel / clear (axis 4)
|
||||
|
||||
Goal: verify pendingLayout overlay enters on "적용하기", exits on "취소", and auto-clears on successful regenerate.
|
||||
|
||||
Steps:
|
||||
1. Click any non-current layout card "적용하기" button.
|
||||
2. Confirm amber dashed overlay appears over `.slide-body` area with `PENDING BODY LAYOUT` label and chosen layout id.
|
||||
3. Confirm "취소" button (top-right of canvas) is visible.
|
||||
4. Click "취소"; confirm overlay disappears, `userSelection` resets, and `hasPendingChanges` indicator clears.
|
||||
5. Re-enter pending mode (apply a layout again), this time click "선택대로 재생성하기"; confirm overlay disappears on success (Home.tsx clears `pendingLayout` + `hasPendingChanges` before pipeline call) and the new `final.html` renders inline (no overlay).
|
||||
|
||||
Pass: enter → cancel → re-enter → regenerate cycle leaves no overlay, no stuck pending state, no leftover `hasPendingChanges` flag.
|
||||
Fail signal: overlay persists after regenerate, button stays in amber state, or cancel does not restore the canvas.
|
||||
|
||||
## Section 5 — mdx03 end-to-end pass (axis 5)
|
||||
|
||||
Goal: smoke run combining Sections 1–4 in one session to validate mdx03 demo path.
|
||||
|
||||
Steps:
|
||||
1. Fresh reload `http://localhost:5173/?mdx=03`.
|
||||
2. Click "슬라이드 플랜 생성하기" → wait for run completion → confirm iframe renders cleanly (Section 1 pass).
|
||||
3. Click any zone → inspect 2+ candidates in frame list (Section 2 pass).
|
||||
4. Apply a non-default frame → "선택대로 재생성하기" → new run id + iframe reflects override (Section 3 pass).
|
||||
5. Apply a non-default layout → confirm overlay → "선택대로 재생성하기" → overlay clears (Section 4 pass).
|
||||
6. Capture: run_id chain, final.html path under `data/runs/<run_id>/final.html`, and DOM screenshot of the rendered iframe content.
|
||||
|
||||
Pass: all five sections green, no console errors, no toast errors, three distinct `run_id`s produced across the session.
|
||||
Fail signal: any earlier section regression OR backend pipeline failure (out-of-scope for IMP-47A — log separately).
|
||||
|
||||
## Out of scope (not tested here)
|
||||
|
||||
- AI fallback activation in `Step 12` (`light_edit` / `restructure`) — IMP-47B.
|
||||
- Frame cache (#62 / IMP-46).
|
||||
- mdx04 / mdx05 path-specific axes.
|
||||
- Automated Playwright e2e replacement — future work.
|
||||
@@ -0,0 +1,492 @@
|
||||
"""P5 (2026-05-20) — Dormant trigger guard tests (issue #58, unit u4).
|
||||
|
||||
Covers the L3 dormant trigger layer end-to-end:
|
||||
|
||||
- u1 — docs/architecture/DORMANT-TRIGGERS.yaml schema + content for the
|
||||
IMP-16 / IMP-17 / IMP-18 / IMP-19 / IMP-20 axes.
|
||||
- u2 — scripts/check_dormant_triggers.py file-pattern + content-pattern
|
||||
matching, manual-evidence skip, followup-linked skip,
|
||||
false-positive guards, exit-0 standalone invocation.
|
||||
- u3 — orchestrator._check_dormant_triggers() helper fail-open contract
|
||||
and the Stage 4→5 _audit_mode() bypass predicate.
|
||||
- u5 — DORMANT-TRIGGERS.yaml self-documenting header
|
||||
(governance doc cross-reference test runs after u5 lands).
|
||||
|
||||
Each test names the IMP-# trigger it exercises (scope-qualified verification
|
||||
per the work-principles lock).
|
||||
|
||||
Run: pytest -q tests/orchestrator_unit/test_dormant_triggers.py
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib
|
||||
import json
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
import yaml
|
||||
|
||||
ROOT = Path(__file__).parent.parent.parent
|
||||
sys.path.insert(0, str(ROOT))
|
||||
sys.path.insert(0, str(ROOT / "scripts"))
|
||||
|
||||
REGISTRY_PATH = ROOT / "docs" / "architecture" / "DORMANT-TRIGGERS.yaml"
|
||||
GOVERNANCE_PATH = ROOT / "docs" / "architecture" / "PROJECT-INTENT-AND-GOVERNANCE.md"
|
||||
CHECKER_PATH = ROOT / "scripts" / "check_dormant_triggers.py"
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────
|
||||
# Shared fixtures
|
||||
# ─────────────────────────────────────────────────────────────────
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def registry_entries() -> list[dict]:
|
||||
"""Parsed registry — used across schema + content tests."""
|
||||
with REGISTRY_PATH.open("r", encoding="utf-8") as f:
|
||||
data = yaml.safe_load(f)
|
||||
assert isinstance(data, list), "registry root must be a YAML list"
|
||||
return data
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def chkmod():
|
||||
"""Fresh import of the standalone checker module (function-scoped so
|
||||
monkeypatched module attributes do not leak across tests)."""
|
||||
if "check_dormant_triggers" in sys.modules:
|
||||
return importlib.reload(sys.modules["check_dormant_triggers"])
|
||||
import check_dormant_triggers as m # noqa: WPS433 — runtime import by design
|
||||
return m
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def orch():
|
||||
"""Orchestrator module under test (u3 helper)."""
|
||||
import orchestrator as m # noqa: WPS433
|
||||
return m
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────
|
||||
# u1 — registry yaml schema + content
|
||||
# ─────────────────────────────────────────────────────────────────
|
||||
|
||||
class TestRegistrySchema:
|
||||
"""u1 — DORMANT-TRIGGERS.yaml exists, parses, and shapes the 5 dormant axes."""
|
||||
|
||||
def test_registry_yaml_parses_with_pyyaml(self):
|
||||
"""Schema sanity (covers all of IMP-16/17/18/19/20): file parses as YAML list."""
|
||||
assert REGISTRY_PATH.exists(), f"registry missing: {REGISTRY_PATH}"
|
||||
with REGISTRY_PATH.open("r", encoding="utf-8") as f:
|
||||
data = yaml.safe_load(f)
|
||||
assert isinstance(data, list), "registry root must be a YAML list"
|
||||
|
||||
def test_registry_has_5_entries_4_active_plus_1_followup_linked(self, registry_entries):
|
||||
"""Stage 1/2 scope-lock: IMP-16 / IMP-17 / IMP-18 / IMP-19 + IMP-20 followup-linked."""
|
||||
assert len(registry_entries) == 5
|
||||
issues = sorted(e["issue"] for e in registry_entries)
|
||||
assert issues == [16, 17, 18, 19, 20]
|
||||
|
||||
def test_registry_required_fields_present(self, registry_entries):
|
||||
"""Every entry has issue / title / doc / status / trigger / on_trigger (covers IMP-16~20)."""
|
||||
for e in registry_entries:
|
||||
assert isinstance(e.get("issue"), int)
|
||||
assert isinstance(e.get("title"), str) and e["title"]
|
||||
assert isinstance(e.get("doc"), str) and e["doc"]
|
||||
assert "status" in e
|
||||
assert isinstance(e.get("trigger"), dict)
|
||||
assert isinstance(e.get("on_trigger"), dict)
|
||||
trig = e["trigger"]
|
||||
assert "description" in trig
|
||||
assert isinstance(trig.get("manual_evidence_required"), bool)
|
||||
|
||||
|
||||
class TestImp16Entry:
|
||||
"""IMP-16 (issue 16) — active watch on src/** reverse-path adapter."""
|
||||
|
||||
def test_imp16_active_src_glob_and_reverse_path_content_pattern(self, registry_entries):
|
||||
"""IMP-16 trigger: src/**/*.py glob + reverse_path / html_to_slide_mdx content."""
|
||||
e = next(x for x in registry_entries if x["issue"] == 16)
|
||||
assert e["status"] == "documented:dormant"
|
||||
assert not e.get("followup_issue"), "IMP-16 is active, not followup-linked"
|
||||
t = e["trigger"]
|
||||
assert t["manual_evidence_required"] is False
|
||||
assert any("src/**" in p for p in t["file_patterns"])
|
||||
cps = t["content_patterns"]
|
||||
assert any("reverse_path" in p or "html_to_slide_mdx" in p for p in cps)
|
||||
|
||||
|
||||
class TestImp17Entry:
|
||||
"""IMP-17 (issue 17) — manual-evidence gate (3-cond User-GO AND)."""
|
||||
|
||||
def test_imp17_manual_evidence_required_true(self, registry_entries):
|
||||
"""IMP-17 trigger gate: manual_evidence_required=true (User GO + B4 + IMP-04/05 live)."""
|
||||
e = next(x for x in registry_entries if x["issue"] == 17)
|
||||
assert e["trigger"]["manual_evidence_required"] is True
|
||||
|
||||
|
||||
class TestImp18Entry:
|
||||
"""IMP-18 (issue 18) — SVG partial under templates/phase_z2/."""
|
||||
|
||||
def test_imp18_active_watch_on_phase_z2_templates_with_svg_content(self, registry_entries):
|
||||
"""IMP-18 trigger: templates/phase_z2/{families,frames}/*.html + SVG signature."""
|
||||
e = next(x for x in registry_entries if x["issue"] == 18)
|
||||
assert e["trigger"]["manual_evidence_required"] is False
|
||||
fps = e["trigger"]["file_patterns"]
|
||||
assert any("templates/phase_z2/" in p for p in fps)
|
||||
cps = e["trigger"]["content_patterns"]
|
||||
assert any("svg" in p.lower() or "viewBox" in p for p in cps)
|
||||
|
||||
|
||||
class TestImp19Entry:
|
||||
"""IMP-19 (issue 19) — manual-evidence gate (IMP-09 owner sign-off)."""
|
||||
|
||||
def test_imp19_manual_evidence_required_true(self, registry_entries):
|
||||
"""IMP-19 trigger gate: manual_evidence_required=true (failing-case + IMP-09 sign-off)."""
|
||||
e = next(x for x in registry_entries if x["issue"] == 19)
|
||||
assert e["trigger"]["manual_evidence_required"] is True
|
||||
|
||||
|
||||
class TestImp20Entry:
|
||||
"""IMP-20 (issue 20) — followup-linked to open issue #55, note-only."""
|
||||
|
||||
def test_imp20_followup_linked_to_55_with_note_only_action(self, registry_entries):
|
||||
"""IMP-20 status: followup_issue=55, on_trigger.action=note_only (no checker watch)."""
|
||||
e = next(x for x in registry_entries if x["issue"] == 20)
|
||||
assert e.get("followup_issue") == 55
|
||||
assert e["on_trigger"]["action"] == "note_only"
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────
|
||||
# u2 — check_dormant_triggers.py
|
||||
# ─────────────────────────────────────────────────────────────────
|
||||
|
||||
class TestCheckerMatching:
|
||||
|
||||
def test_checker_clean_tree_no_alerts_covers_imp16_18(self, chkmod, monkeypatch):
|
||||
"""False-positive guard: empty change surface → no IMP-16 / IMP-18 alerts."""
|
||||
monkeypatch.setattr(chkmod, "collect_changed_files", lambda: [])
|
||||
entries = chkmod.load_registry()
|
||||
alerts = [a for a in (chkmod.check_entry(e, []) for e in entries) if a]
|
||||
assert alerts == []
|
||||
|
||||
def test_checker_imp16_alert_on_src_reverse_path_adapter(
|
||||
self, chkmod, monkeypatch, tmp_path
|
||||
):
|
||||
"""IMP-16 positive: src/foo/adapter.py with `reverse_path` content → alert."""
|
||||
fake_path = "src/foo/adapter.py"
|
||||
(tmp_path / "src" / "foo").mkdir(parents=True)
|
||||
(tmp_path / fake_path).write_text("def reverse_path(): pass\n", encoding="utf-8")
|
||||
monkeypatch.setattr(chkmod, "REPO_ROOT", tmp_path)
|
||||
entries = chkmod.load_registry()
|
||||
imp16 = next(e for e in entries if e["issue"] == 16)
|
||||
result = chkmod.check_entry(imp16, [fake_path])
|
||||
assert result is not None
|
||||
assert result["issue"] == 16
|
||||
assert fake_path in result["match"]["files"]
|
||||
|
||||
def test_checker_imp16_no_alert_on_tests_path_false_positive_guard(
|
||||
self, chkmod, monkeypatch, tmp_path
|
||||
):
|
||||
"""False-positive guard: IMP-16 must NOT fire on tests/foo.py even with matching content."""
|
||||
fake_path = "tests/foo.py"
|
||||
(tmp_path / "tests").mkdir()
|
||||
(tmp_path / fake_path).write_text("def reverse_path(): pass\n", encoding="utf-8")
|
||||
monkeypatch.setattr(chkmod, "REPO_ROOT", tmp_path)
|
||||
entries = chkmod.load_registry()
|
||||
imp16 = next(e for e in entries if e["issue"] == 16)
|
||||
assert chkmod.check_entry(imp16, [fake_path]) is None
|
||||
|
||||
def test_checker_imp18_alert_on_phase_z2_family_svg(
|
||||
self, chkmod, monkeypatch, tmp_path
|
||||
):
|
||||
"""IMP-18 positive: templates/phase_z2/families/new.html with <svg viewBox> → alert."""
|
||||
fake_path = "templates/phase_z2/families/new_partial.html"
|
||||
(tmp_path / "templates" / "phase_z2" / "families").mkdir(parents=True)
|
||||
(tmp_path / fake_path).write_text(
|
||||
'<svg viewBox="0 0 100 100"></svg>\n', encoding="utf-8"
|
||||
)
|
||||
monkeypatch.setattr(chkmod, "REPO_ROOT", tmp_path)
|
||||
entries = chkmod.load_registry()
|
||||
imp18 = next(e for e in entries if e["issue"] == 18)
|
||||
result = chkmod.check_entry(imp18, [fake_path])
|
||||
assert result is not None
|
||||
assert result["issue"] == 18
|
||||
|
||||
def test_checker_imp18_flat_glob_boundary_nested_path_skipped(
|
||||
self, chkmod, monkeypatch, tmp_path
|
||||
):
|
||||
"""False-positive guard: IMP-18 flat glob `families/*.html` skips nested family path."""
|
||||
fake_path = "templates/phase_z2/families/nested/inner.html"
|
||||
(tmp_path / "templates" / "phase_z2" / "families" / "nested").mkdir(parents=True)
|
||||
(tmp_path / fake_path).write_text(
|
||||
'<svg viewBox="0 0 100 100"></svg>\n', encoding="utf-8"
|
||||
)
|
||||
monkeypatch.setattr(chkmod, "REPO_ROOT", tmp_path)
|
||||
entries = chkmod.load_registry()
|
||||
imp18 = next(e for e in entries if e["issue"] == 18)
|
||||
assert chkmod.check_entry(imp18, [fake_path]) is None
|
||||
|
||||
def test_checker_skips_imp17_manual_evidence(self, chkmod):
|
||||
"""Guardrail: IMP-17 manual_evidence_required skips even with broad change surface."""
|
||||
entries = chkmod.load_registry()
|
||||
imp17 = next(e for e in entries if e["issue"] == 17)
|
||||
assert chkmod.check_entry(imp17, ["src/foo.py", "anything.html"]) is None
|
||||
|
||||
def test_checker_skips_imp19_manual_evidence(self, chkmod):
|
||||
"""Guardrail: IMP-19 manual_evidence_required skips even with broad change surface."""
|
||||
entries = chkmod.load_registry()
|
||||
imp19 = next(e for e in entries if e["issue"] == 19)
|
||||
assert chkmod.check_entry(imp19, ["src/foo.py"]) is None
|
||||
|
||||
def test_checker_skips_imp20_followup_linked(self, chkmod):
|
||||
"""Guardrail: IMP-20 followup_issue=55 skips (open issue #55 owns the watch)."""
|
||||
entries = chkmod.load_registry()
|
||||
imp20 = next(e for e in entries if e["issue"] == 20)
|
||||
assert chkmod.check_entry(imp20, ["anything", "src/a.py"]) is None
|
||||
|
||||
def test_checker_standalone_invocation_exit_0(self):
|
||||
"""Guardrail: standalone `python scripts/check_dormant_triggers.py` exits 0 always.
|
||||
|
||||
Exercises the IMP-16~20 informational-only contract — checker never blocks
|
||||
the orchestrator regardless of working-tree state.
|
||||
"""
|
||||
r = subprocess.run(
|
||||
[sys.executable, str(CHECKER_PATH)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
encoding="utf-8",
|
||||
errors="replace",
|
||||
cwd=str(ROOT),
|
||||
timeout=30,
|
||||
)
|
||||
assert r.returncode == 0, f"checker exit={r.returncode} stderr={r.stderr!r}"
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────
|
||||
# u3 — orchestrator helper + Stage 4→5 hook
|
||||
# ─────────────────────────────────────────────────────────────────
|
||||
|
||||
class TestOrchestratorDormantHook:
|
||||
|
||||
def test_orchestrator_helper_returns_list_on_clean_tree(self, orch):
|
||||
"""u3 helper returns list[dict] (possibly empty) — never raises (IMP-16~20 informational)."""
|
||||
result = orch._check_dormant_triggers()
|
||||
assert isinstance(result, list)
|
||||
|
||||
def test_orchestrator_helper_fail_open_on_subprocess_error(self, orch, monkeypatch):
|
||||
"""u3 helper: subprocess raise → [] (fail-open, no false positives across all IMPs)."""
|
||||
def boom(*a, **kw):
|
||||
raise RuntimeError("subprocess unavailable")
|
||||
monkeypatch.setattr(orch.subprocess, "run", boom)
|
||||
assert orch._check_dormant_triggers() == []
|
||||
|
||||
def test_orchestrator_helper_fail_open_on_nonzero_exit(self, orch, monkeypatch):
|
||||
"""u3 helper: subprocess returncode != 0 → [] (fail-open)."""
|
||||
class _Err:
|
||||
returncode = 1
|
||||
stdout = ""
|
||||
stderr = "boom"
|
||||
monkeypatch.setattr(orch.subprocess, "run", lambda *a, **kw: _Err())
|
||||
assert orch._check_dormant_triggers() == []
|
||||
|
||||
def test_orchestrator_helper_fail_open_on_missing_alerts_file(
|
||||
self, orch, monkeypatch, tmp_path
|
||||
):
|
||||
"""u3 helper: subprocess OK but alert file absent → [] (fail-open)."""
|
||||
class _OK:
|
||||
returncode = 0
|
||||
stdout = ""
|
||||
stderr = ""
|
||||
monkeypatch.setattr(orch.subprocess, "run", lambda *a, **kw: _OK())
|
||||
monkeypatch.setattr(orch, "ORCH_DIR", tmp_path)
|
||||
assert orch._check_dormant_triggers() == []
|
||||
|
||||
def test_orchestrator_helper_parses_alerts_payload(self, orch, monkeypatch, tmp_path):
|
||||
"""u3 helper: reads `alerts` list out of .orchestrator/dormant_alerts.json payload."""
|
||||
class _OK:
|
||||
returncode = 0
|
||||
stdout = ""
|
||||
stderr = ""
|
||||
monkeypatch.setattr(orch.subprocess, "run", lambda *a, **kw: _OK())
|
||||
monkeypatch.setattr(orch, "ORCH_DIR", tmp_path)
|
||||
payload = {
|
||||
"alerts": [
|
||||
{"issue": 16, "title": "IMP-16 sample", "on_trigger": {"action": "create_runtime_issue"}},
|
||||
],
|
||||
}
|
||||
(tmp_path / "dormant_alerts.json").write_text(
|
||||
json.dumps(payload), encoding="utf-8"
|
||||
)
|
||||
result = orch._check_dormant_triggers()
|
||||
assert isinstance(result, list)
|
||||
assert len(result) == 1
|
||||
assert result[0]["issue"] == 16
|
||||
|
||||
def test_orchestrator_helper_handles_non_list_alerts_payload(
|
||||
self, orch, monkeypatch, tmp_path
|
||||
):
|
||||
"""u3 helper: malformed `alerts` (non-list) → [] (fail-open, IMP-16~20 informational)."""
|
||||
class _OK:
|
||||
returncode = 0
|
||||
stdout = ""
|
||||
stderr = ""
|
||||
monkeypatch.setattr(orch.subprocess, "run", lambda *a, **kw: _OK())
|
||||
monkeypatch.setattr(orch, "ORCH_DIR", tmp_path)
|
||||
(tmp_path / "dormant_alerts.json").write_text(
|
||||
json.dumps({"alerts": "oops"}), encoding="utf-8"
|
||||
)
|
||||
assert orch._check_dormant_triggers() == []
|
||||
|
||||
def test_stage_4_to_5_hook_predicate_audit_bypass(self, orch):
|
||||
"""u3 Stage 4→5 hook guard: _audit_mode(title) True → dormant checker bypassed.
|
||||
|
||||
Mirrors P4a placement — the dormant hook condition is
|
||||
`sid == "test-verify" and not _audit_mode(title)`. This test asserts
|
||||
the gating predicate for the IMP-16/18 active watches.
|
||||
"""
|
||||
assert orch._audit_mode("[INTEGRATION-AUDIT-02] cumulative review") is True
|
||||
assert orch._audit_mode("[AUDIT-ONLY] doc consistency") is True
|
||||
assert orch._audit_mode("[P5][DORMANT-TRIGGER-GUARD] hook") is False
|
||||
assert orch._audit_mode("IMP-16 U2 wiring") is False
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────
|
||||
# u3 — Stage 4→5 hook integration (focused static assertions on run_stage)
|
||||
# ─────────────────────────────────────────────────────────────────
|
||||
|
||||
class TestRunStageDormantHookIntegration:
|
||||
"""u3 — Stage 4→5 hook must be wired into ``orchestrator.run_stage``.
|
||||
|
||||
Codex #5 rewind: testing ``_audit_mode()`` in isolation is insufficient.
|
||||
These focused static-source assertions on ``inspect.getsource(run_stage)``
|
||||
fail if the dormant hook is silently removed or its audit-bypass predicate
|
||||
is weakened — guarding the IMP-16/17/18/19/20 informational alert wiring.
|
||||
"""
|
||||
|
||||
def _run_stage_src(self, orch) -> str:
|
||||
import inspect
|
||||
return inspect.getsource(orch.run_stage)
|
||||
|
||||
def test_run_stage_invokes_check_dormant_triggers_on_test_verify_non_audit(self, orch):
|
||||
"""u3 contract: run_stage body calls _check_dormant_triggers().
|
||||
|
||||
Exercises the Stage 4 (test-verify) non-audit invocation path for
|
||||
IMP-16/18 active watches. If the call disappears from run_stage,
|
||||
this test fails (catches the regression that Codex #5 flagged).
|
||||
"""
|
||||
src = self._run_stage_src(orch)
|
||||
assert "_check_dormant_triggers()" in src, (
|
||||
"run_stage must invoke _check_dormant_triggers() — the L3 wiring "
|
||||
"for IMP-16/17/18/19/20 dormant alerts. Silent removal breaks the "
|
||||
"Stage 4→5 informational hook."
|
||||
)
|
||||
|
||||
def test_run_stage_dormant_hook_gated_by_test_verify_and_non_audit(self, orch):
|
||||
"""u3 contract: dormant hook is gated by both sid==test-verify AND not _audit_mode(title).
|
||||
|
||||
The predicate is what makes audit-only Stage 4 bypass the IMP-16~20
|
||||
checker (Stage 1 scope-lock guardrail). Removing either conjunct
|
||||
would silently re-enable the checker on audit-only issues whose
|
||||
change surface is restricted to audit-report docs.
|
||||
"""
|
||||
import re
|
||||
src = self._run_stage_src(orch)
|
||||
m = re.search(
|
||||
r'sid\s*==\s*"test-verify"\s+and\s+not\s+_audit_mode\(title\)\s*:\s*\n'
|
||||
r'\s+alerts\s*=\s*_check_dormant_triggers\(\)',
|
||||
src,
|
||||
)
|
||||
assert m is not None, (
|
||||
"Stage 4→5 dormant hook must be gated by "
|
||||
'`sid == "test-verify" and not _audit_mode(title):` immediately '
|
||||
"before `alerts = _check_dormant_triggers()`. Audit-only bypass "
|
||||
"and Stage 4 placement are part of the IMP-16~20 contract."
|
||||
)
|
||||
|
||||
def test_run_stage_dormant_hook_is_informational_no_continue(self, orch):
|
||||
"""u3 contract: dormant hook block must NOT contain a `continue` statement.
|
||||
|
||||
IMP-16~20 alerts are informational only (Stage 1 guardrail). The hook
|
||||
block between the _check_dormant_triggers() call and the next
|
||||
Stage 4 PASS log line ("YES (evidence verified)") must not short-
|
||||
circuit Stage 5 entry via `continue`. Comments mentioning the word
|
||||
"continue" are allowed (they document the contract).
|
||||
"""
|
||||
import re
|
||||
src = self._run_stage_src(orch)
|
||||
start = src.find("_check_dormant_triggers()")
|
||||
assert start >= 0
|
||||
end = src.find("YES (evidence verified)", start)
|
||||
assert end > start, (
|
||||
"could not locate Stage 4 PASS log line after dormant hook — "
|
||||
"run_stage shape may have shifted; re-examine integration."
|
||||
)
|
||||
hook_block = src[start:end]
|
||||
# Strip line comments before checking for the `continue` keyword as
|
||||
# an actual statement — the source intentionally documents the
|
||||
# informational-only contract with a `# Never continue — ...` comment.
|
||||
stripped_lines = []
|
||||
for line in hook_block.splitlines():
|
||||
code = line.split("#", 1)[0]
|
||||
stripped_lines.append(code)
|
||||
code_only = "\n".join(stripped_lines)
|
||||
assert not re.search(r'(^|\s)continue\b', code_only), (
|
||||
"dormant hook block contains a `continue` statement — IMP-16~20 "
|
||||
"alerts must never block Stage 5 entry (informational-only contract)."
|
||||
)
|
||||
|
||||
def test_run_stage_dormant_hook_positioned_before_stage_pass_return(self, orch):
|
||||
"""u3 contract: dormant hook is positioned within the Stage 4 YES PASS path.
|
||||
|
||||
Verifies the hook sits between the P4a audit commit-scope guard and
|
||||
the Stage 4 success log+return (i.e., on the PASS path), not in an
|
||||
unreachable branch. Protects against accidental relocation during
|
||||
future refactors.
|
||||
"""
|
||||
src = self._run_stage_src(orch)
|
||||
hook_pos = src.find("_check_dormant_triggers()")
|
||||
pass_log_pos = src.find("YES (evidence verified)", hook_pos)
|
||||
return_true_pos = src.find("return True", hook_pos)
|
||||
assert hook_pos >= 0
|
||||
assert 0 < pass_log_pos - hook_pos < 4000, (
|
||||
"dormant hook is not adjacent to the Stage 4 PASS log line — "
|
||||
"run_stage shape may have shifted away from PASS-path placement."
|
||||
)
|
||||
assert return_true_pos > pass_log_pos, (
|
||||
"Stage 4 `return True` should follow the PASS log line; "
|
||||
"dormant hook must precede both."
|
||||
)
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────
|
||||
# u5 — registry self-documentation + governance cross-reference
|
||||
# ─────────────────────────────────────────────────────────────────
|
||||
|
||||
class TestRegistryHeaderAndGovernanceRef:
|
||||
|
||||
def test_registry_header_explains_schema_and_l3_purpose(self):
|
||||
"""u5 acceptance: yaml header explains schema + L3 informational-only purpose.
|
||||
|
||||
The header is the durable self-documentation surface that anchors the
|
||||
IMP-16/17/18/19/20 registry (per Stage 1 exit unit u5).
|
||||
"""
|
||||
text = REGISTRY_PATH.read_text(encoding="utf-8")
|
||||
assert "Schema" in text or "schema" in text
|
||||
assert "dormant" in text.lower()
|
||||
assert "L3" in text or "machine-readable" in text.lower()
|
||||
assert "Guardrails" in text or "informational" in text.lower()
|
||||
|
||||
@pytest.mark.skipif(
|
||||
not GOVERNANCE_PATH.exists()
|
||||
or "DORMANT-TRIGGERS.yaml" not in GOVERNANCE_PATH.read_text(encoding="utf-8"),
|
||||
reason="u5 governance-doc reference line not yet appended (passes after u5 lands).",
|
||||
)
|
||||
def test_governance_doc_references_registry(self):
|
||||
"""u5 deliverable: PROJECT-INTENT-AND-GOVERNANCE.md cites DORMANT-TRIGGERS.yaml as L3.
|
||||
|
||||
Runs after u5 lands — IMP-16~20 registry surfaces in the governance
|
||||
anti-patterns row so future maintainers find it without rediscovery.
|
||||
"""
|
||||
text = GOVERNANCE_PATH.read_text(encoding="utf-8")
|
||||
assert "DORMANT-TRIGGERS.yaml" in text
|
||||
@@ -4,6 +4,10 @@ Stage 1 finding: line 564 previously referenced a non-existent ID ("IMP-31").
|
||||
The legitimate slot is IMP-17 (Gitea #17, carve-out — AI fallback only, normal path 밖).
|
||||
Line 565 (IMP-29 frontend zone-level override) must remain untouched.
|
||||
|
||||
Anchor re-pin (2026-05-20, IMP-30 u1 follow-up): V4Match.provisional field added at
|
||||
src/phase_z2_pipeline.py:179-184 shifted the route-hint table down by six lines.
|
||||
Pinned line numbers updated from 564/565 → 570/571 to track the actual anchor location.
|
||||
|
||||
Run: pytest -q tests/orchestrator_unit/test_imp17_comment_anchor.py
|
||||
"""
|
||||
from pathlib import Path
|
||||
@@ -16,14 +20,14 @@ def _lines() -> list[str]:
|
||||
return PIPELINE.read_text(encoding="utf-8").splitlines()
|
||||
|
||||
|
||||
def test_line_564_references_imp17_not_imp31():
|
||||
line = _lines()[563] # 1-indexed line 564
|
||||
assert "restructure" in line, f"line 564 anchor drifted: {line!r}"
|
||||
assert "IMP-17" in line, f"line 564 must reference IMP-17 (carve-out): {line!r}"
|
||||
assert "IMP-31" not in line, f"line 564 must not reference non-existent IMP-31: {line!r}"
|
||||
def test_line_570_references_imp17_not_imp31():
|
||||
line = _lines()[569] # 1-indexed line 570
|
||||
assert "restructure" in line, f"line 570 anchor drifted: {line!r}"
|
||||
assert "IMP-17" in line, f"line 570 must reference IMP-17 (carve-out): {line!r}"
|
||||
assert "IMP-31" not in line, f"line 570 must not reference non-existent IMP-31: {line!r}"
|
||||
|
||||
|
||||
def test_line_565_still_references_imp29():
|
||||
line = _lines()[564] # 1-indexed line 565
|
||||
assert "reject" in line, f"line 565 anchor drifted: {line!r}"
|
||||
assert "IMP-29" in line, f"line 565 must still reference IMP-29 frontend override: {line!r}"
|
||||
def test_line_571_still_references_imp29():
|
||||
line = _lines()[570] # 1-indexed line 571
|
||||
assert "reject" in line, f"line 571 anchor drifted: {line!r}"
|
||||
assert "IMP-29" in line, f"line 571 must still reference IMP-29 frontend override: {line!r}"
|
||||
|
||||
@@ -97,6 +97,107 @@ Addressing [Codex #2] findings ...
|
||||
assert detect_agent("[Codex#1] hello") == "codex"
|
||||
assert detect_agent("[Claude#5] hi") == "claude"
|
||||
|
||||
def test_audit_anchor_preface_breaks_detection(self):
|
||||
"""P5 (2026-05-20) — regression: AUDIT-ONLY mode 의 'Audit anchor:' preface 가
|
||||
첫 줄에 박히면 detect_agent 는 None 반환 (P0-1 strict 의도된 동작).
|
||||
이게 #56 (INTEGRATION-AUDIT-02) 의 Stage 4 Round #14 infinite loop 의 직접 원인.
|
||||
해결책 = detect_agent 완화 X, AUDIT_ONLY_NOTE 가 agent header 를 first line 으로 강제."""
|
||||
body_anchor_first = (
|
||||
"Audit anchor: This audit verifies pipeline contracts...\n"
|
||||
"It does not implement runtime code.\n"
|
||||
"\n"
|
||||
"[Codex #14] Stage 4 (test-verify) Round #14 - INTEGRATION-AUDIT-02\n"
|
||||
"\n"
|
||||
"Verdict: PASS. Stage 3 satisfies all criteria.\n"
|
||||
"FINAL_CONSENSUS: YES\n"
|
||||
)
|
||||
assert detect_agent(body_anchor_first) is None, (
|
||||
"audit anchor preface as first line MUST cause detect_agent None "
|
||||
"(P0-1 strict). Fix path: comment format, not detect_agent."
|
||||
)
|
||||
|
||||
def test_audit_anchor_after_header_works(self):
|
||||
"""P5 (2026-05-20) — 올바른 format: agent header first line, anchor line 2+."""
|
||||
body_header_first = (
|
||||
"[Codex #14] Stage 4 (test-verify) Round #14 - INTEGRATION-AUDIT-02\n"
|
||||
"\n"
|
||||
"Audit anchor: This audit verifies pipeline contracts...\n"
|
||||
"\n"
|
||||
"Verdict: PASS.\n"
|
||||
"FINAL_CONSENSUS: YES\n"
|
||||
)
|
||||
assert detect_agent(body_header_first) == "codex"
|
||||
|
||||
# P5b (2026-05-20) — Stage 2 compact-plan first-line conflict regression.
|
||||
# #24 IMP-24 K6: Codex r1~r3 가 첫 줄을 '=== IMPLEMENTATION_UNITS ===' 로 시작 →
|
||||
# detect_agent None → orchestrator silent loop. fix path = comment format strict,
|
||||
# NOT detect_agent 완화 (P0-1 강화 그대로 유지).
|
||||
|
||||
def test_implementation_units_first_line_breaks_detection(self):
|
||||
"""=== IMPLEMENTATION_UNITS === 가 첫 줄이면 detect_agent None (P0-1 strict 정상 동작)."""
|
||||
body = (
|
||||
"=== IMPLEMENTATION_UNITS ===\n"
|
||||
"- id: u1\n"
|
||||
" summary: ...\n"
|
||||
" files:\n"
|
||||
" - docs/architecture/PHASE-Q-AUDIT.md\n"
|
||||
" tests:\n"
|
||||
" - pytest -q tests\n"
|
||||
" estimate_lines: 1\n"
|
||||
"\n"
|
||||
"FINAL_CONSENSUS: YES\n"
|
||||
)
|
||||
assert detect_agent(body) is None, (
|
||||
"=== IMPLEMENTATION_UNITS === as first line MUST cause detect_agent None "
|
||||
"(P0-1 strict). Fix path: enforce agent header first-line in prompt, not relax detect_agent."
|
||||
)
|
||||
|
||||
def test_compact_plan_with_header_first_works(self):
|
||||
"""올바른 Stage 2 compact format: [Codex #N] 첫 줄 → === IMPLEMENTATION_UNITS === 둘째 줄+."""
|
||||
body = (
|
||||
"[Codex #4] Stage 2 simulation-plan review - IMP-24 K6\n"
|
||||
"\n"
|
||||
"=== IMPLEMENTATION_UNITS ===\n"
|
||||
"- id: u1\n"
|
||||
" summary: ...\n"
|
||||
" tests:\n"
|
||||
" - pytest -q tests\n"
|
||||
"\n"
|
||||
"FINAL_CONSENSUS: YES\n"
|
||||
)
|
||||
assert detect_agent(body) == "codex"
|
||||
|
||||
def test_markdown_prefix_breaks_detection(self):
|
||||
"""P5b — `## [Codex #N]` 같은 markdown header prefix 도 detect_agent None.
|
||||
(#21 Stage 4 에서 관찰된 latent silent loop 원인.)"""
|
||||
body_hash = "## [Codex #1] Stage 4 test-verify Round #1\n\nVerdict: PASS\n"
|
||||
body_emoji = "📌 **[Claude #1] Stage 2 plan**\n\nbody\n"
|
||||
body_bold = "**[Codex #1] Stage 4**\n\nbody\n"
|
||||
assert detect_agent(body_hash) is None
|
||||
assert detect_agent(body_emoji) is None
|
||||
assert detect_agent(body_bold) is None
|
||||
|
||||
|
||||
class TestRulesAndCompactPlanFirstLineContract:
|
||||
"""P5b (2026-05-20) — RULES 와 COMPACT_PLAN_RULE 둘 다 first-line agent header
|
||||
rule 을 명시해야 함. wording 검증."""
|
||||
|
||||
def test_rules_has_first_line_strict(self):
|
||||
from orchestrator import RULES
|
||||
# RULES 안에 first-line strict + 모든 stage 적용 명시 있어야 함.
|
||||
assert "FIRST non-empty line" in RULES
|
||||
assert "[Claude #N]" in RULES and "[Codex #N]" in RULES
|
||||
# P5b OVERRIDES 키워드 — body rule 들이 first-line rule 보다 우선하지 않음을 강조
|
||||
assert "OVERRIDES" in RULES or "overrides" in RULES.lower()
|
||||
|
||||
def test_compact_plan_rule_carves_out_first_line(self):
|
||||
from orchestrator import COMPACT_PLAN_RULE
|
||||
# "body" 는 first-line agent header 다음부터 시작한다고 명시
|
||||
assert "FIRST non-empty line" in COMPACT_PLAN_RULE or "first-line agent header" in COMPACT_PLAN_RULE
|
||||
# "after the first-line" 같은 carve-out wording 검증
|
||||
body_lower = COMPACT_PLAN_RULE.lower()
|
||||
assert "after the first" in body_lower or "after the agent header" in body_lower
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────
|
||||
# parse_consensus — YES/NO + rewind_target
|
||||
|
||||
@@ -0,0 +1,180 @@
|
||||
"""IMP-34 R1 u2 — plan_zone_ratio_retry measured-empty donor capacity bound.
|
||||
|
||||
Stage 2 contract (unit u2) regression coverage for u1 planner change at
|
||||
`src/phase_z2_retry.py:122-157`:
|
||||
|
||||
axis 1: absent measured fields → static_fallback (regression)
|
||||
axis 2: measured_empty < static_slack → measured_bound applied
|
||||
axis 3: measured_empty >= static_slack → static_slack honored
|
||||
axis 4: measured_empty == 0 → donor excluded (slack<=0 gate)
|
||||
axis 5: donor filter + telemetry source fields → eligibility + bool guards
|
||||
|
||||
u1 adds an additive measured-empty bound on donor capacity in
|
||||
`plan_zone_ratio_retry`. Step 14 schema unchanged; reuses `clientHeight` /
|
||||
`scrollHeight` already in `overflow["zones"]`. Static fallback preserves
|
||||
prior behavior when measured fields are absent.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from src.phase_z2_retry import plan_zone_ratio_retry
|
||||
|
||||
|
||||
_ROUTER_ACTIVE = {"router_active": True}
|
||||
|
||||
|
||||
def _classification(target_pos: str, excess_y: float) -> dict:
|
||||
return {
|
||||
"classifications": [
|
||||
{
|
||||
"proposed_action": "zone_ratio_retry",
|
||||
"zone_position": target_pos,
|
||||
"inputs": {"excess_y": excess_y},
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
|
||||
def _zone(position: str, height_px: int, min_height_px: int,
|
||||
fit_status: str | None = "ok") -> dict:
|
||||
return {
|
||||
"position": position,
|
||||
"height_px": height_px,
|
||||
"min_height_px": min_height_px,
|
||||
"composition_rationale": {
|
||||
"capacity_fit": {"fit_status": fit_status},
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _ozone(position: str, *, client_h=None, scroll_h=None,
|
||||
overflowed: bool = False, clipped_inner: bool = False) -> dict:
|
||||
z: dict = {"position": position, "overflowed": overflowed,
|
||||
"clipped_inner": clipped_inner}
|
||||
if client_h is not None:
|
||||
z["clientHeight"] = client_h
|
||||
if scroll_h is not None:
|
||||
z["scrollHeight"] = scroll_h
|
||||
return z
|
||||
|
||||
|
||||
def test_axis1_absent_measured_fields_uses_static_fallback():
|
||||
"""Step 14 zones without clientHeight/scrollHeight → planner falls back to
|
||||
static slack (regression: pre-u1 behavior preserved)."""
|
||||
debug_zones = [
|
||||
_zone("top", height_px=200, min_height_px=180),
|
||||
_zone("bottom", height_px=400, min_height_px=200), # static_slack=200
|
||||
]
|
||||
overflow = {"zones": [_ozone("bottom")]} # no clientHeight/scrollHeight
|
||||
plan = plan_zone_ratio_retry(
|
||||
debug_zones=debug_zones,
|
||||
overflow=overflow,
|
||||
fit_classification=_classification("top", excess_y=20.0),
|
||||
router_decision=_ROUTER_ACTIVE,
|
||||
)
|
||||
assert plan is not None and plan["feasible"] is True
|
||||
donors = plan["donor_candidates_considered"]
|
||||
assert len(donors) == 1
|
||||
assert donors[0]["position"] == "bottom"
|
||||
assert donors[0]["slack"] == 200 # static slack unchanged
|
||||
assert donors[0]["measured_empty_px"] is None
|
||||
assert donors[0]["slack_bound_source"] == "static_fallback"
|
||||
|
||||
|
||||
def test_axis2_measured_empty_less_than_static_slack_bound_applied():
|
||||
"""measured_empty (30) < static_slack (200) → donor capacity bounded to 30,
|
||||
telemetry reports measured_bound."""
|
||||
debug_zones = [
|
||||
_zone("top", height_px=200, min_height_px=180),
|
||||
_zone("bottom", height_px=400, min_height_px=200), # static_slack=200
|
||||
]
|
||||
# clientHeight 400, scrollHeight 370 → measured_empty=30
|
||||
overflow = {"zones": [_ozone("bottom", client_h=400, scroll_h=370)]}
|
||||
plan = plan_zone_ratio_retry(
|
||||
debug_zones=debug_zones,
|
||||
overflow=overflow,
|
||||
fit_classification=_classification("top", excess_y=20.0),
|
||||
router_decision=_ROUTER_ACTIVE,
|
||||
)
|
||||
assert plan["feasible"] is True
|
||||
donor = plan["donor_candidates_considered"][0]
|
||||
assert donor["slack"] == 30 # bound by measured_empty
|
||||
assert donor["measured_empty_px"] == 30
|
||||
assert donor["slack_bound_source"] == "measured_bound"
|
||||
# target_added_px = ceil(20)+4 = 24, donor slack=30 covers it
|
||||
assert plan["aggregate_slack_available"] == 30
|
||||
assert plan["donor_reduced_px"] == 24
|
||||
|
||||
|
||||
def test_axis3_measured_empty_ge_static_slack_static_honored():
|
||||
"""measured_empty (300) >= static_slack (50) → static_slack wins via min(),
|
||||
but telemetry still reports measured_bound (both fields present)."""
|
||||
debug_zones = [
|
||||
_zone("top", height_px=200, min_height_px=180),
|
||||
_zone("bottom", height_px=250, min_height_px=200), # static_slack=50
|
||||
]
|
||||
# clientHeight 250, scrollHeight -50 → measured_empty=max(0, 300)=300
|
||||
# (extreme value to prove min() honors static_slack)
|
||||
overflow = {"zones": [_ozone("bottom", client_h=250, scroll_h=-50)]}
|
||||
plan = plan_zone_ratio_retry(
|
||||
debug_zones=debug_zones,
|
||||
overflow=overflow,
|
||||
fit_classification=_classification("top", excess_y=20.0),
|
||||
router_decision=_ROUTER_ACTIVE,
|
||||
)
|
||||
assert plan["feasible"] is True
|
||||
donor = plan["donor_candidates_considered"][0]
|
||||
assert donor["slack"] == 50 # static_slack honored via min()
|
||||
assert donor["measured_empty_px"] == 300
|
||||
assert donor["slack_bound_source"] == "measured_bound"
|
||||
|
||||
|
||||
def test_axis4_measured_empty_zero_excludes_donor():
|
||||
"""measured_empty == 0 (donor full) → slack<=0 gate excludes the donor
|
||||
(planner avoids over-allocating a visually-full sibling)."""
|
||||
debug_zones = [
|
||||
_zone("top", height_px=200, min_height_px=180),
|
||||
_zone("bottom", height_px=400, min_height_px=200), # static_slack=200
|
||||
]
|
||||
# clientHeight==scrollHeight → measured_empty=0
|
||||
overflow = {"zones": [_ozone("bottom", client_h=400, scroll_h=400)]}
|
||||
plan = plan_zone_ratio_retry(
|
||||
debug_zones=debug_zones,
|
||||
overflow=overflow,
|
||||
fit_classification=_classification("top", excess_y=20.0),
|
||||
router_decision=_ROUTER_ACTIVE,
|
||||
)
|
||||
assert plan is not None and plan["feasible"] is False
|
||||
assert plan["donor_candidates_considered"] == []
|
||||
assert "no donor candidates eligible" in plan["failure_reason"]
|
||||
# zones_after preserves zones_before — revert-friendly
|
||||
assert plan["zones_after"]["bottom"] == 400
|
||||
|
||||
|
||||
def test_axis5_donor_filter_preservation_and_telemetry_bool_guard():
|
||||
"""Donor eligibility filter (overflowed / clipped_inner) precedes the new
|
||||
measured bound, and bool values must NOT be treated as numeric measured
|
||||
fields (Python isinstance(True, int) is True without an explicit guard)."""
|
||||
debug_zones = [
|
||||
_zone("top", height_px=200, min_height_px=180),
|
||||
_zone("middle", height_px=400, min_height_px=200), # static_slack=200
|
||||
_zone("bottom", height_px=400, min_height_px=200), # static_slack=200
|
||||
]
|
||||
overflow = {"zones": [
|
||||
# middle is itself overflowed → must be filtered out regardless of measured fields
|
||||
_ozone("middle", client_h=400, scroll_h=350, overflowed=True),
|
||||
# bottom uses bool measured fields → must fall back to static_fallback (not crash, not measured_bound)
|
||||
_ozone("bottom", client_h=True, scroll_h=False),
|
||||
]}
|
||||
plan = plan_zone_ratio_retry(
|
||||
debug_zones=debug_zones,
|
||||
overflow=overflow,
|
||||
fit_classification=_classification("top", excess_y=20.0),
|
||||
router_decision=_ROUTER_ACTIVE,
|
||||
)
|
||||
assert plan["feasible"] is True
|
||||
donors = plan["donor_candidates_considered"]
|
||||
# middle filtered by overflowed=True; only bottom remains
|
||||
assert [d["position"] for d in donors] == ["bottom"]
|
||||
assert donors[0]["slack"] == 200 # static slack (bools rejected as measured)
|
||||
assert donors[0]["measured_empty_px"] is None
|
||||
assert donors[0]["slack_bound_source"] == "static_fallback"
|
||||
@@ -0,0 +1,154 @@
|
||||
"""IMP-33 u10 — AST isolation guard for the AI fallback package.
|
||||
|
||||
Structural defence: parse every ``*.py`` file under
|
||||
``src/phase_z2_ai_fallback/`` and assert that none of them imports a
|
||||
Phase Q runtime module, the Kei API client, or any ``phase_z2_*`` runtime
|
||||
module (e.g. ``phase_z2_pipeline``). Even if a future patch wires such a
|
||||
module by accident, this AST scan catches it before runtime and protects
|
||||
the PZ-1 invariant (normal-path AI call count = 0).
|
||||
|
||||
Allowed imports inside the fallback package:
|
||||
|
||||
* Standard library modules.
|
||||
* ``anthropic`` (u4 client) and ``pydantic`` (u2 schema).
|
||||
* ``src.config`` (u1 settings — single source of truth for policy knobs).
|
||||
* Other modules inside ``src.phase_z2_ai_fallback`` (intra-package).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import ast
|
||||
import pathlib
|
||||
|
||||
import pytest
|
||||
|
||||
PACKAGE_ROOT = pathlib.Path(__file__).resolve().parents[2] / "src" / "phase_z2_ai_fallback"
|
||||
|
||||
_ALLOWED_SRC_PREFIXES: tuple[str, ...] = (
|
||||
"src.config",
|
||||
"src.phase_z2_ai_fallback",
|
||||
)
|
||||
|
||||
_ALLOWED_TOP_LEVEL: frozenset[str] = frozenset(
|
||||
{
|
||||
"anthropic",
|
||||
"pydantic",
|
||||
"__future__",
|
||||
"ast",
|
||||
"dataclasses",
|
||||
"enum",
|
||||
"hashlib",
|
||||
"json",
|
||||
"pathlib",
|
||||
"random",
|
||||
"time",
|
||||
"typing",
|
||||
}
|
||||
)
|
||||
|
||||
_FORBIDDEN_PHASE_Q_MODULES: frozenset[str] = frozenset(
|
||||
{
|
||||
"src.pipeline",
|
||||
"src.pipeline_v2",
|
||||
"src.block_assembler",
|
||||
"src.block_assembler_b2",
|
||||
"src.block_matcher_tfidf",
|
||||
"src.block_reference",
|
||||
"src.block_search",
|
||||
"src.block_selector",
|
||||
"src.content_editor",
|
||||
"src.design_director",
|
||||
"src.html_generator",
|
||||
"src.html_validator",
|
||||
"src.renderer",
|
||||
"src.mdx_normalizer",
|
||||
"src.fit_verifier",
|
||||
"src.slide_measurer",
|
||||
"src.space_allocator",
|
||||
"src.kei_client",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def _module_files() -> list[pathlib.Path]:
|
||||
return sorted(p for p in PACKAGE_ROOT.glob("*.py") if p.name != "__pycache__")
|
||||
|
||||
|
||||
def _imported_names(tree: ast.AST) -> list[str]:
|
||||
names: list[str] = []
|
||||
for node in ast.walk(tree):
|
||||
if isinstance(node, ast.Import):
|
||||
for alias in node.names:
|
||||
names.append(alias.name)
|
||||
elif isinstance(node, ast.ImportFrom):
|
||||
if node.module is not None:
|
||||
names.append(node.module)
|
||||
return names
|
||||
|
||||
|
||||
def _parse(path: pathlib.Path) -> ast.AST:
|
||||
return ast.parse(path.read_text(encoding="utf-8"), filename=str(path))
|
||||
|
||||
|
||||
def _is_allowed(name: str) -> bool:
|
||||
for prefix in _ALLOWED_SRC_PREFIXES:
|
||||
if name == prefix or name.startswith(prefix + "."):
|
||||
return True
|
||||
top = name.split(".", 1)[0]
|
||||
return top in _ALLOWED_TOP_LEVEL
|
||||
|
||||
|
||||
def test_fallback_package_root_exists() -> None:
|
||||
assert PACKAGE_ROOT.is_dir(), (
|
||||
f"fallback package root not found at {PACKAGE_ROOT!s}; module path "
|
||||
"is locked by IMP-31-GATE-AUDIT (src/phase_z2_ai_fallback/)."
|
||||
)
|
||||
files = _module_files()
|
||||
assert files, f"no .py modules found under {PACKAGE_ROOT!s}"
|
||||
|
||||
|
||||
def test_fallback_package_imports_are_whitelisted() -> None:
|
||||
violations: list[tuple[str, str]] = []
|
||||
for path in _module_files():
|
||||
for name in _imported_names(_parse(path)):
|
||||
if not _is_allowed(name):
|
||||
violations.append((path.name, name))
|
||||
assert not violations, (
|
||||
"fallback package imports outside the IMP-33 whitelist "
|
||||
f"(Phase Q / Kei / phase_z2_* runtime forbidden): {violations}"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("forbidden_module", sorted(_FORBIDDEN_PHASE_Q_MODULES))
|
||||
def test_fallback_package_forbids_phase_q_and_kei_imports(forbidden_module: str) -> None:
|
||||
for path in _module_files():
|
||||
for name in _imported_names(_parse(path)):
|
||||
top2 = ".".join(name.split(".")[:2])
|
||||
assert top2 != forbidden_module and name != forbidden_module, (
|
||||
f"{path.name} imports forbidden module {name!r}; "
|
||||
f"{forbidden_module!r} is a Phase Q / Kei runtime module and "
|
||||
"must not be reachable from the AI fallback package."
|
||||
)
|
||||
|
||||
|
||||
def test_fallback_package_forbids_phase_z2_pipeline_imports() -> None:
|
||||
for path in _module_files():
|
||||
for name in _imported_names(_parse(path)):
|
||||
assert not name.startswith("src.phase_z2_pipeline"), (
|
||||
f"{path.name} imports {name!r}; the Phase Z2 pipeline runtime "
|
||||
"module must not be reachable from the AI fallback package "
|
||||
"(PZ-1: normal-path AI=0)."
|
||||
)
|
||||
|
||||
|
||||
def test_fallback_package_forbids_other_phase_z2_runtime_imports() -> None:
|
||||
violations: list[tuple[str, str]] = []
|
||||
for path in _module_files():
|
||||
for name in _imported_names(_parse(path)):
|
||||
if name.startswith("src.phase_z2_") and not name.startswith(
|
||||
"src.phase_z2_ai_fallback"
|
||||
):
|
||||
violations.append((path.name, name))
|
||||
assert not violations, (
|
||||
"fallback package imports another phase_z2_* runtime module; "
|
||||
f"violations: {violations}"
|
||||
)
|
||||
@@ -0,0 +1,508 @@
|
||||
"""IMP-46 u2 — Persistent JSON cache backend tests.
|
||||
|
||||
Scope (Stage 2 plan, u2):
|
||||
|
||||
* Replaced ``NotImplementedError`` marker with a real persistent backend
|
||||
at ``data/frame_cache/{frame_id}/{signature_hash}.json``.
|
||||
* Preserved IMP-33 u6 dual write gate: ``visual_check_passed`` AND
|
||||
``user_approved`` BOTH required (loud :class:`AiFallbackCacheGateError`
|
||||
before any filesystem touch).
|
||||
* Round-trip every :class:`ProposalKind`; round-trip ``slide_css`` None
|
||||
*and* set; missing or corrupt files miss silently.
|
||||
* Fingerprint *comparison* is u3; here we only check that the field is
|
||||
persisted.
|
||||
|
||||
All filesystem writes are scoped to ``tmp_path`` via
|
||||
``monkeypatch.setattr`` on the module-level :data:`CACHE_ROOT`, so the
|
||||
production directory is never touched by these tests.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import pathlib
|
||||
|
||||
import pytest
|
||||
|
||||
from src.phase_z2_ai_fallback import cache as cache_mod
|
||||
from src.phase_z2_ai_fallback.cache import (
|
||||
AiFallbackCacheGateError,
|
||||
KEY_DELIMITER,
|
||||
SCHEMA_VERSION,
|
||||
read_proposal,
|
||||
save_proposal,
|
||||
)
|
||||
from src.phase_z2_ai_fallback.schema import AiFallbackProposal, ProposalKind
|
||||
|
||||
|
||||
_FRAME_ID = "1171281190"
|
||||
_SIG_HASH = "a" * 64 # SHA256-shaped placeholder; cache is shape-agnostic.
|
||||
_KEY = f"{_FRAME_ID}{KEY_DELIMITER}{_SIG_HASH}"
|
||||
|
||||
|
||||
def _proposal(
|
||||
kind: ProposalKind = ProposalKind.BUILDER_OPTIONS_PATCH,
|
||||
payload: dict | None = None,
|
||||
) -> AiFallbackProposal:
|
||||
return AiFallbackProposal(
|
||||
proposal_kind=kind,
|
||||
payload=payload if payload is not None else {"item_parser": "bullet_v2"},
|
||||
rationale="u2-test",
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _isolated_cache_root(tmp_path: pathlib.Path, monkeypatch: pytest.MonkeyPatch):
|
||||
"""Redirect the cache root to an isolated tmp directory for every test."""
|
||||
monkeypatch.setattr(cache_mod, "CACHE_ROOT", tmp_path / "frame_cache")
|
||||
yield tmp_path / "frame_cache"
|
||||
|
||||
|
||||
# -- read_proposal --------------------------------------------------------
|
||||
|
||||
|
||||
def test_read_proposal_returns_none_for_missing_file():
|
||||
assert read_proposal(_KEY) is None
|
||||
|
||||
|
||||
def test_read_proposal_rejects_empty_key():
|
||||
with pytest.raises(ValueError):
|
||||
read_proposal("")
|
||||
|
||||
|
||||
def test_read_proposal_rejects_non_string_key():
|
||||
with pytest.raises(ValueError):
|
||||
read_proposal(None) # type: ignore[arg-type]
|
||||
|
||||
|
||||
def test_read_proposal_returns_none_for_legacy_key_format():
|
||||
"""Router back-compat: pre-u4 cache_key (no '::') misses silently."""
|
||||
assert read_proposal("frame:1171281190:cardinality:many") is None
|
||||
|
||||
|
||||
def test_read_proposal_returns_none_for_corrupt_json(_isolated_cache_root: pathlib.Path):
|
||||
path = _isolated_cache_root / _FRAME_ID / f"{_SIG_HASH}.json"
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text("{not valid json", encoding="utf-8")
|
||||
assert read_proposal(_KEY) is None
|
||||
|
||||
|
||||
def test_read_proposal_returns_none_for_non_dict_root(_isolated_cache_root: pathlib.Path):
|
||||
path = _isolated_cache_root / _FRAME_ID / f"{_SIG_HASH}.json"
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text("[]", encoding="utf-8")
|
||||
assert read_proposal(_KEY) is None
|
||||
|
||||
|
||||
def test_read_proposal_returns_none_when_payload_proposal_missing(
|
||||
_isolated_cache_root: pathlib.Path,
|
||||
):
|
||||
path = _isolated_cache_root / _FRAME_ID / f"{_SIG_HASH}.json"
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(json.dumps({"schema_version": 1}), encoding="utf-8")
|
||||
assert read_proposal(_KEY) is None
|
||||
|
||||
|
||||
def test_read_proposal_returns_none_for_forbidden_proposal_kind(
|
||||
_isolated_cache_root: pathlib.Path,
|
||||
):
|
||||
path = _isolated_cache_root / _FRAME_ID / f"{_SIG_HASH}.json"
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"schema_version": 1,
|
||||
"proposal": {"proposal_kind": "mdx_text", "payload": {}, "rationale": ""},
|
||||
"slide_css": None,
|
||||
"fingerprints": {},
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
assert read_proposal(_KEY) is None
|
||||
|
||||
|
||||
# -- save_proposal: write gates -------------------------------------------
|
||||
|
||||
|
||||
def test_save_rejects_when_visual_check_failed():
|
||||
with pytest.raises(AiFallbackCacheGateError) as exc:
|
||||
save_proposal(
|
||||
_KEY, _proposal(), visual_check_passed=False, user_approved=True
|
||||
)
|
||||
assert "visual_check_passed" in str(exc.value)
|
||||
|
||||
|
||||
def test_save_rejects_when_user_not_approved():
|
||||
with pytest.raises(AiFallbackCacheGateError) as exc:
|
||||
save_proposal(
|
||||
_KEY, _proposal(), visual_check_passed=True, user_approved=False
|
||||
)
|
||||
assert "user_approved" in str(exc.value)
|
||||
|
||||
|
||||
def test_save_rejects_when_both_gates_false():
|
||||
with pytest.raises(AiFallbackCacheGateError):
|
||||
save_proposal(
|
||||
_KEY, _proposal(), visual_check_passed=False, user_approved=False
|
||||
)
|
||||
|
||||
|
||||
def test_save_gate_violation_does_not_touch_filesystem(
|
||||
_isolated_cache_root: pathlib.Path,
|
||||
):
|
||||
with pytest.raises(AiFallbackCacheGateError):
|
||||
save_proposal(
|
||||
_KEY, _proposal(), visual_check_passed=False, user_approved=True
|
||||
)
|
||||
# Cache root may or may not exist depending on fixture order, but the
|
||||
# frame_id directory must NOT exist when the gate rejects the write.
|
||||
assert not (_isolated_cache_root / _FRAME_ID).exists()
|
||||
|
||||
|
||||
def test_save_rejects_empty_key():
|
||||
with pytest.raises(ValueError):
|
||||
save_proposal(
|
||||
"", _proposal(), visual_check_passed=True, user_approved=True
|
||||
)
|
||||
|
||||
|
||||
def test_save_rejects_non_proposal_object():
|
||||
with pytest.raises(TypeError):
|
||||
save_proposal(
|
||||
_KEY,
|
||||
{"proposal_kind": "builder_options_patch"}, # type: ignore[arg-type]
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
)
|
||||
|
||||
|
||||
def test_save_rejects_legacy_key_format():
|
||||
"""Writes must use the structural ``frame_id::signature_hash`` form."""
|
||||
with pytest.raises(ValueError):
|
||||
save_proposal(
|
||||
"frame:1171281190:cardinality:many",
|
||||
_proposal(),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
)
|
||||
|
||||
|
||||
def test_save_rejects_slide_css_non_string():
|
||||
with pytest.raises(TypeError):
|
||||
save_proposal(
|
||||
_KEY,
|
||||
_proposal(),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
slide_css=123, # type: ignore[arg-type]
|
||||
)
|
||||
|
||||
|
||||
def test_save_rejects_fingerprints_non_dict():
|
||||
with pytest.raises(TypeError):
|
||||
save_proposal(
|
||||
_KEY,
|
||||
_proposal(),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
fingerprints=["contract_sha", "abc"], # type: ignore[arg-type]
|
||||
)
|
||||
|
||||
|
||||
def test_gate_error_is_not_notimplementederror():
|
||||
"""The persistent backend no longer raises ``NotImplementedError`` —
|
||||
callers must distinguish gate violation from absent persistence."""
|
||||
assert not issubclass(AiFallbackCacheGateError, NotImplementedError)
|
||||
|
||||
|
||||
# -- save_proposal: persistence + round-trip ------------------------------
|
||||
|
||||
|
||||
def test_save_creates_parent_directories(_isolated_cache_root: pathlib.Path):
|
||||
assert not (_isolated_cache_root / _FRAME_ID).exists()
|
||||
save_proposal(
|
||||
_KEY, _proposal(), visual_check_passed=True, user_approved=True
|
||||
)
|
||||
assert (_isolated_cache_root / _FRAME_ID / f"{_SIG_HASH}.json").is_file()
|
||||
|
||||
|
||||
def test_save_returns_resolved_path(_isolated_cache_root: pathlib.Path):
|
||||
path = save_proposal(
|
||||
_KEY, _proposal(), visual_check_passed=True, user_approved=True
|
||||
)
|
||||
assert path == _isolated_cache_root / _FRAME_ID / f"{_SIG_HASH}.json"
|
||||
|
||||
|
||||
def test_save_payload_includes_schema_version(_isolated_cache_root: pathlib.Path):
|
||||
path = save_proposal(
|
||||
_KEY, _proposal(), visual_check_passed=True, user_approved=True
|
||||
)
|
||||
data = json.loads(path.read_text(encoding="utf-8"))
|
||||
assert data["schema_version"] == SCHEMA_VERSION
|
||||
|
||||
|
||||
def test_save_payload_includes_proposal_dump(_isolated_cache_root: pathlib.Path):
|
||||
proposal = _proposal(payload={"item_parser": "pillar_item"})
|
||||
path = save_proposal(
|
||||
_KEY, proposal, visual_check_passed=True, user_approved=True
|
||||
)
|
||||
data = json.loads(path.read_text(encoding="utf-8"))
|
||||
assert data["proposal"] == proposal.model_dump(mode="json")
|
||||
|
||||
|
||||
def test_round_trip_default_slide_css_is_none(_isolated_cache_root: pathlib.Path):
|
||||
path = save_proposal(
|
||||
_KEY, _proposal(), visual_check_passed=True, user_approved=True
|
||||
)
|
||||
data = json.loads(path.read_text(encoding="utf-8"))
|
||||
assert data["slide_css"] is None
|
||||
assert data["fingerprints"] == {}
|
||||
|
||||
|
||||
def test_round_trip_with_slide_css_set(_isolated_cache_root: pathlib.Path):
|
||||
css = ".slide { padding: 40px; }"
|
||||
path = save_proposal(
|
||||
_KEY,
|
||||
_proposal(),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
slide_css=css,
|
||||
)
|
||||
data = json.loads(path.read_text(encoding="utf-8"))
|
||||
assert data["slide_css"] == css
|
||||
|
||||
|
||||
def test_round_trip_with_fingerprints(_isolated_cache_root: pathlib.Path):
|
||||
fingerprints = {
|
||||
"contract_sha": "c" * 64,
|
||||
"partial_sha": "p" * 64,
|
||||
"catalog_sha": "x" * 64,
|
||||
}
|
||||
path = save_proposal(
|
||||
_KEY,
|
||||
_proposal(),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
fingerprints=fingerprints,
|
||||
)
|
||||
data = json.loads(path.read_text(encoding="utf-8"))
|
||||
assert data["fingerprints"] == fingerprints
|
||||
|
||||
|
||||
def test_read_returns_proposal_after_save(_isolated_cache_root: pathlib.Path):
|
||||
original = _proposal(payload={"key": "value"})
|
||||
save_proposal(
|
||||
_KEY, original, visual_check_passed=True, user_approved=True
|
||||
)
|
||||
loaded = read_proposal(_KEY)
|
||||
assert loaded is not None
|
||||
assert loaded.proposal_kind == original.proposal_kind
|
||||
assert loaded.payload == original.payload
|
||||
assert loaded.rationale == original.rationale
|
||||
|
||||
|
||||
@pytest.mark.parametrize("kind", list(ProposalKind))
|
||||
def test_round_trip_all_proposal_kinds(
|
||||
kind: ProposalKind, _isolated_cache_root: pathlib.Path
|
||||
):
|
||||
"""Every whitelisted ProposalKind survives save → read unchanged."""
|
||||
if kind is ProposalKind.PARTIAL_OVERRIDES:
|
||||
payload = {"slots": {"pillar_1": "alpha"}}
|
||||
elif kind is ProposalKind.SLOT_MAPPING_PROPOSAL:
|
||||
payload = {"mapping": [{"from": "a", "to": "b"}]}
|
||||
else:
|
||||
payload = {"item_parser": "bullet_v2"}
|
||||
save_proposal(
|
||||
_KEY,
|
||||
_proposal(kind=kind, payload=payload),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
)
|
||||
loaded = read_proposal(_KEY)
|
||||
assert loaded is not None
|
||||
assert loaded.proposal_kind is kind
|
||||
assert loaded.payload == payload
|
||||
|
||||
|
||||
def test_save_overwrites_existing_entry(_isolated_cache_root: pathlib.Path):
|
||||
save_proposal(
|
||||
_KEY,
|
||||
_proposal(payload={"v": 1}),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
)
|
||||
save_proposal(
|
||||
_KEY,
|
||||
_proposal(payload={"v": 2}),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
)
|
||||
loaded = read_proposal(_KEY)
|
||||
assert loaded is not None
|
||||
assert loaded.payload == {"v": 2}
|
||||
|
||||
|
||||
def test_file_layout_uses_frame_id_directory(_isolated_cache_root: pathlib.Path):
|
||||
"""Storage layout = ``frame_id/`` directory, ``signature_hash.json`` file."""
|
||||
other_frame_key = f"{_FRAME_ID}_other{KEY_DELIMITER}{_SIG_HASH}"
|
||||
save_proposal(
|
||||
_KEY, _proposal(), visual_check_passed=True, user_approved=True
|
||||
)
|
||||
save_proposal(
|
||||
other_frame_key,
|
||||
_proposal(),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
)
|
||||
assert (_isolated_cache_root / _FRAME_ID / f"{_SIG_HASH}.json").is_file()
|
||||
assert (
|
||||
_isolated_cache_root / f"{_FRAME_ID}_other" / f"{_SIG_HASH}.json"
|
||||
).is_file()
|
||||
|
||||
|
||||
def test_different_signature_hashes_isolated(_isolated_cache_root: pathlib.Path):
|
||||
"""Two distinct signature hashes under the same frame_id never collide."""
|
||||
key_a = f"{_FRAME_ID}{KEY_DELIMITER}{'a' * 64}"
|
||||
key_b = f"{_FRAME_ID}{KEY_DELIMITER}{'b' * 64}"
|
||||
save_proposal(
|
||||
key_a,
|
||||
_proposal(payload={"sig": "a"}),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
)
|
||||
save_proposal(
|
||||
key_b,
|
||||
_proposal(payload={"sig": "b"}),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
)
|
||||
loaded_a = read_proposal(key_a)
|
||||
loaded_b = read_proposal(key_b)
|
||||
assert loaded_a is not None and loaded_a.payload == {"sig": "a"}
|
||||
assert loaded_b is not None and loaded_b.payload == {"sig": "b"}
|
||||
|
||||
|
||||
def test_parse_key_rejects_triple_delimiter():
|
||||
"""Two ``::`` markers (extra delimiter inside signature) is rejected."""
|
||||
assert (
|
||||
read_proposal(
|
||||
f"{_FRAME_ID}{KEY_DELIMITER}{_SIG_HASH}{KEY_DELIMITER}extra"
|
||||
)
|
||||
is None
|
||||
)
|
||||
|
||||
|
||||
# -- IMP-46 u5: auto_cache gate (2^3 truth table) -------------------------
|
||||
#
|
||||
# Three booleans: visual_check_passed (V), user_approved (U), auto_cache (A).
|
||||
# Contract: V=True AND (U=True OR A=True) -> persist; else gate-raise.
|
||||
# V is never bypassable; A=True only relaxes U=False.
|
||||
|
||||
_GATE_TRUTH_TABLE = [
|
||||
# (V, U, A, expect_persist)
|
||||
(False, False, False, False),
|
||||
(False, False, True, False),
|
||||
(False, True, False, False),
|
||||
(False, True, True, False),
|
||||
(True, False, False, False),
|
||||
(True, False, True, True),
|
||||
(True, True, False, True),
|
||||
(True, True, True, True),
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("v,u,a,expect_persist", _GATE_TRUTH_TABLE)
|
||||
def test_save_gate_truth_table(
|
||||
v: bool,
|
||||
u: bool,
|
||||
a: bool,
|
||||
expect_persist: bool,
|
||||
_isolated_cache_root: pathlib.Path,
|
||||
) -> None:
|
||||
"""IMP-46 u5 — exhaustive 2^3 enumeration of (V, U, A) -> {persist, raise}."""
|
||||
if expect_persist:
|
||||
path = save_proposal(
|
||||
_KEY,
|
||||
_proposal(payload={"v": int(v), "u": int(u), "a": int(a)}),
|
||||
visual_check_passed=v,
|
||||
user_approved=u,
|
||||
auto_cache=a,
|
||||
)
|
||||
assert path.is_file(), f"truth row (V={v}, U={u}, A={a}) must persist"
|
||||
else:
|
||||
with pytest.raises(AiFallbackCacheGateError):
|
||||
save_proposal(
|
||||
_KEY,
|
||||
_proposal(),
|
||||
visual_check_passed=v,
|
||||
user_approved=u,
|
||||
auto_cache=a,
|
||||
)
|
||||
# Gate violations must never touch the filesystem (parent dir absent).
|
||||
assert not (_isolated_cache_root / _FRAME_ID).exists(), (
|
||||
f"truth row (V={v}, U={u}, A={a}) leaked a directory"
|
||||
)
|
||||
|
||||
|
||||
def test_auto_cache_default_off_preserves_dual_gate_semantics(
|
||||
_isolated_cache_root: pathlib.Path,
|
||||
) -> None:
|
||||
"""Calling save_proposal without ``auto_cache`` keeps the IMP-46 u2 behaviour."""
|
||||
with pytest.raises(AiFallbackCacheGateError) as exc:
|
||||
save_proposal(
|
||||
_KEY, _proposal(), visual_check_passed=True, user_approved=False
|
||||
)
|
||||
assert "user_approved" in str(exc.value)
|
||||
assert not (_isolated_cache_root / _FRAME_ID).exists()
|
||||
|
||||
|
||||
def test_auto_cache_cannot_bypass_visual_check() -> None:
|
||||
"""``visual_check_passed=False`` raises even with ``auto_cache=True``."""
|
||||
with pytest.raises(AiFallbackCacheGateError) as exc:
|
||||
save_proposal(
|
||||
_KEY,
|
||||
_proposal(),
|
||||
visual_check_passed=False,
|
||||
user_approved=True,
|
||||
auto_cache=True,
|
||||
)
|
||||
assert "visual_check_passed" in str(exc.value)
|
||||
|
||||
|
||||
def test_auto_cache_bypass_user_approved_persists(
|
||||
_isolated_cache_root: pathlib.Path,
|
||||
) -> None:
|
||||
"""``auto_cache=True`` with ``user_approved=False`` persists the proposal."""
|
||||
path = save_proposal(
|
||||
_KEY,
|
||||
_proposal(payload={"bypass": "user"}),
|
||||
visual_check_passed=True,
|
||||
user_approved=False,
|
||||
auto_cache=True,
|
||||
)
|
||||
assert path.is_file()
|
||||
loaded = read_proposal(_KEY)
|
||||
assert loaded is not None
|
||||
assert loaded.payload == {"bypass": "user"}
|
||||
|
||||
|
||||
def test_auto_cache_rejects_non_bool() -> None:
|
||||
"""``auto_cache`` must be a bool (loud TypeError, symmetric with other kwargs)."""
|
||||
with pytest.raises(TypeError):
|
||||
save_proposal(
|
||||
_KEY,
|
||||
_proposal(),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
auto_cache="yes", # type: ignore[arg-type]
|
||||
)
|
||||
|
||||
|
||||
def test_auto_cache_is_keyword_only() -> None:
|
||||
"""``auto_cache`` must be passed by keyword (positional rejected)."""
|
||||
import inspect
|
||||
|
||||
sig = inspect.signature(save_proposal)
|
||||
param = sig.parameters["auto_cache"]
|
||||
assert param.kind is inspect.Parameter.KEYWORD_ONLY
|
||||
assert param.default is False
|
||||
@@ -0,0 +1,347 @@
|
||||
"""IMP-46 u3 — Fingerprint-based cache invalidation tests.
|
||||
|
||||
Scope (Stage 2 plan, u3):
|
||||
|
||||
* ``save_proposal`` persists ``fingerprints`` verbatim (u2 already covers
|
||||
the round-trip; this suite re-asserts the read-side comparator).
|
||||
* ``read_proposal`` accepts an optional ``fingerprints`` kwarg. When
|
||||
supplied, the stored dict must equal the supplied dict EXACTLY (strict
|
||||
equality). Mismatch — including missing keys, extra keys, or value
|
||||
drift — returns ``None``.
|
||||
* Default ``fingerprints=None`` performs no comparison (back-compat for
|
||||
legacy callers).
|
||||
* Fingerprint *computation* stays outside ``cache.py`` — these tests
|
||||
treat the three declared shas (``contract_sha`` / ``partial_sha`` /
|
||||
``catalog_sha``) as opaque hex strings, never recomputing them. The
|
||||
cache layer is a content-addressed *comparator*, not a content
|
||||
*hasher*.
|
||||
|
||||
All filesystem writes are scoped to ``tmp_path`` via
|
||||
``monkeypatch.setattr`` on the module-level :data:`CACHE_ROOT`.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import pathlib
|
||||
|
||||
import pytest
|
||||
|
||||
from src.phase_z2_ai_fallback import cache as cache_mod
|
||||
from src.phase_z2_ai_fallback.cache import (
|
||||
KEY_DELIMITER,
|
||||
read_proposal,
|
||||
save_proposal,
|
||||
)
|
||||
from src.phase_z2_ai_fallback.schema import AiFallbackProposal, ProposalKind
|
||||
|
||||
|
||||
_FRAME_ID = "1171281190"
|
||||
_SIG_HASH = "f" * 64
|
||||
_KEY = f"{_FRAME_ID}{KEY_DELIMITER}{_SIG_HASH}"
|
||||
|
||||
_FINGERPRINTS_BASELINE: dict[str, str] = {
|
||||
"contract_sha": "c" * 64,
|
||||
"partial_sha": "p" * 64,
|
||||
"catalog_sha": "x" * 64,
|
||||
}
|
||||
|
||||
|
||||
def _proposal(payload: dict | None = None) -> AiFallbackProposal:
|
||||
return AiFallbackProposal(
|
||||
proposal_kind=ProposalKind.BUILDER_OPTIONS_PATCH,
|
||||
payload=payload if payload is not None else {"item_parser": "bullet_v2"},
|
||||
rationale="u3-test",
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _isolated_cache_root(tmp_path: pathlib.Path, monkeypatch: pytest.MonkeyPatch):
|
||||
monkeypatch.setattr(cache_mod, "CACHE_ROOT", tmp_path / "frame_cache")
|
||||
yield tmp_path / "frame_cache"
|
||||
|
||||
|
||||
# -- save side: fingerprints persisted verbatim ---------------------------
|
||||
|
||||
|
||||
def test_save_persists_fingerprints_verbatim(
|
||||
_isolated_cache_root: pathlib.Path,
|
||||
):
|
||||
path = save_proposal(
|
||||
_KEY,
|
||||
_proposal(),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
fingerprints=_FINGERPRINTS_BASELINE,
|
||||
)
|
||||
stored = json.loads(path.read_text(encoding="utf-8"))["fingerprints"]
|
||||
assert stored == _FINGERPRINTS_BASELINE
|
||||
|
||||
|
||||
# -- read side: back-compat (no fingerprints kwarg) -----------------------
|
||||
|
||||
|
||||
def test_read_without_fingerprints_kwarg_returns_proposal(
|
||||
_isolated_cache_root: pathlib.Path,
|
||||
):
|
||||
"""Legacy read path (no kwarg) skips invalidation — round-trip succeeds."""
|
||||
save_proposal(
|
||||
_KEY,
|
||||
_proposal(),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
fingerprints=_FINGERPRINTS_BASELINE,
|
||||
)
|
||||
loaded = read_proposal(_KEY)
|
||||
assert loaded is not None
|
||||
assert loaded.payload == {"item_parser": "bullet_v2"}
|
||||
|
||||
|
||||
def test_read_without_fingerprints_kwarg_ignores_stored_mismatch(
|
||||
_isolated_cache_root: pathlib.Path,
|
||||
):
|
||||
"""A caller that has not adopted fingerprint-aware lookup must still
|
||||
see the proposal — invalidation only kicks in when explicitly asked."""
|
||||
save_proposal(
|
||||
_KEY,
|
||||
_proposal(),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
fingerprints={"contract_sha": "old"},
|
||||
)
|
||||
loaded = read_proposal(_KEY)
|
||||
assert loaded is not None
|
||||
|
||||
|
||||
# -- read side: matching fingerprints -------------------------------------
|
||||
|
||||
|
||||
def test_read_with_matching_fingerprints_returns_proposal(
|
||||
_isolated_cache_root: pathlib.Path,
|
||||
):
|
||||
save_proposal(
|
||||
_KEY,
|
||||
_proposal(),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
fingerprints=_FINGERPRINTS_BASELINE,
|
||||
)
|
||||
loaded = read_proposal(_KEY, fingerprints=dict(_FINGERPRINTS_BASELINE))
|
||||
assert loaded is not None
|
||||
assert loaded.proposal_kind is ProposalKind.BUILDER_OPTIONS_PATCH
|
||||
|
||||
|
||||
def test_read_with_empty_fingerprints_matches_empty_stored(
|
||||
_isolated_cache_root: pathlib.Path,
|
||||
):
|
||||
"""Both sides empty is an exact match, not a special-case None."""
|
||||
save_proposal(
|
||||
_KEY,
|
||||
_proposal(),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
# default fingerprints=None → stored as {}
|
||||
)
|
||||
loaded = read_proposal(_KEY, fingerprints={})
|
||||
assert loaded is not None
|
||||
|
||||
|
||||
# -- read side: invalidation on mismatch ----------------------------------
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"drifted_axis",
|
||||
["contract_sha", "partial_sha", "catalog_sha"],
|
||||
)
|
||||
def test_read_invalidates_on_single_axis_drift(
|
||||
drifted_axis: str, _isolated_cache_root: pathlib.Path
|
||||
):
|
||||
save_proposal(
|
||||
_KEY,
|
||||
_proposal(),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
fingerprints=_FINGERPRINTS_BASELINE,
|
||||
)
|
||||
supplied = dict(_FINGERPRINTS_BASELINE)
|
||||
supplied[drifted_axis] = "deadbeef" * 8 # 64-char distinct value
|
||||
assert read_proposal(_KEY, fingerprints=supplied) is None
|
||||
|
||||
|
||||
def test_read_invalidates_when_caller_supplies_extra_key(
|
||||
_isolated_cache_root: pathlib.Path,
|
||||
):
|
||||
"""Strict equality — extra key on caller side is a mismatch."""
|
||||
save_proposal(
|
||||
_KEY,
|
||||
_proposal(),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
fingerprints=_FINGERPRINTS_BASELINE,
|
||||
)
|
||||
supplied = dict(_FINGERPRINTS_BASELINE)
|
||||
supplied["future_axis_sha"] = "z" * 64
|
||||
assert read_proposal(_KEY, fingerprints=supplied) is None
|
||||
|
||||
|
||||
def test_read_invalidates_when_caller_supplies_subset(
|
||||
_isolated_cache_root: pathlib.Path,
|
||||
):
|
||||
"""Strict equality — subset on caller side is a mismatch."""
|
||||
save_proposal(
|
||||
_KEY,
|
||||
_proposal(),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
fingerprints=_FINGERPRINTS_BASELINE,
|
||||
)
|
||||
subset = {"contract_sha": _FINGERPRINTS_BASELINE["contract_sha"]}
|
||||
assert read_proposal(_KEY, fingerprints=subset) is None
|
||||
|
||||
|
||||
def test_read_invalidates_when_entry_saved_without_fingerprints(
|
||||
_isolated_cache_root: pathlib.Path,
|
||||
):
|
||||
"""A pre-invalidation cache entry (empty stored fingerprints) MUST NOT
|
||||
satisfy a fingerprint-aware lookup — caller demands proof of freshness."""
|
||||
save_proposal(
|
||||
_KEY,
|
||||
_proposal(),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
# default fingerprints=None → stored as {}
|
||||
)
|
||||
assert read_proposal(_KEY, fingerprints=_FINGERPRINTS_BASELINE) is None
|
||||
|
||||
|
||||
def test_read_invalidates_when_stored_fingerprints_not_dict(
|
||||
_isolated_cache_root: pathlib.Path,
|
||||
):
|
||||
"""Hand-corrupted payload (fingerprints serialized as non-dict) → None."""
|
||||
path = _isolated_cache_root / _FRAME_ID / f"{_SIG_HASH}.json"
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"schema_version": 1,
|
||||
"proposal": _proposal().model_dump(mode="json"),
|
||||
"slide_css": None,
|
||||
"fingerprints": ["contract_sha", "c" * 64],
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
assert read_proposal(_KEY, fingerprints=_FINGERPRINTS_BASELINE) is None
|
||||
|
||||
|
||||
def test_read_invalidates_when_stored_fingerprints_field_missing(
|
||||
_isolated_cache_root: pathlib.Path,
|
||||
):
|
||||
"""Legacy payload (no ``fingerprints`` field at all) → None when caller
|
||||
demands fingerprint comparison."""
|
||||
path = _isolated_cache_root / _FRAME_ID / f"{_SIG_HASH}.json"
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"schema_version": 1,
|
||||
"proposal": _proposal().model_dump(mode="json"),
|
||||
"slide_css": None,
|
||||
# fingerprints field deliberately omitted
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
assert read_proposal(_KEY, fingerprints={"contract_sha": "c" * 64}) is None
|
||||
|
||||
|
||||
def test_read_with_matching_fingerprints_still_loses_to_missing_file():
|
||||
"""File missing takes precedence over fingerprint check — no false hit."""
|
||||
assert read_proposal(_KEY, fingerprints=_FINGERPRINTS_BASELINE) is None
|
||||
|
||||
|
||||
def test_read_with_matching_fingerprints_still_loses_to_corrupt_json(
|
||||
_isolated_cache_root: pathlib.Path,
|
||||
):
|
||||
path = _isolated_cache_root / _FRAME_ID / f"{_SIG_HASH}.json"
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text("{not valid json", encoding="utf-8")
|
||||
assert read_proposal(_KEY, fingerprints=_FINGERPRINTS_BASELINE) is None
|
||||
|
||||
|
||||
# -- read side: input validation symmetry with save -----------------------
|
||||
|
||||
|
||||
def test_read_rejects_non_dict_fingerprints():
|
||||
with pytest.raises(TypeError):
|
||||
read_proposal(_KEY, fingerprints=["contract_sha", "c" * 64]) # type: ignore[arg-type]
|
||||
|
||||
|
||||
def test_read_rejects_non_dict_fingerprints_string():
|
||||
with pytest.raises(TypeError):
|
||||
read_proposal(_KEY, fingerprints="contract_sha=c" * 8) # type: ignore[arg-type]
|
||||
|
||||
|
||||
def test_read_rejects_non_dict_fingerprints_int():
|
||||
with pytest.raises(TypeError):
|
||||
read_proposal(_KEY, fingerprints=42) # type: ignore[arg-type]
|
||||
|
||||
|
||||
# -- isolation: cache.py never computes fingerprints ----------------------
|
||||
|
||||
|
||||
def test_cache_module_has_no_fingerprint_computer():
|
||||
"""Guardrail: cache.py is a *comparator*, not a *hasher*. The three
|
||||
declared shas are computed outside this module (step 12 / pipeline
|
||||
glue). Adding a fingerprint computer here would leak Phase Z runtime
|
||||
knowledge into the cache layer and violate AI isolation."""
|
||||
public_surface = [
|
||||
name
|
||||
for name in dir(cache_mod)
|
||||
if not name.startswith("_") and callable(getattr(cache_mod, name))
|
||||
]
|
||||
forbidden_substrings = ("hash", "sha", "fingerprint")
|
||||
leaks = [
|
||||
name
|
||||
for name in public_surface
|
||||
if any(sub in name.lower() for sub in forbidden_substrings)
|
||||
]
|
||||
assert leaks == [], (
|
||||
f"cache.py public surface leaks fingerprint computation: {leaks}; "
|
||||
"computation must live outside cache.py per IMP-46 u3 contract."
|
||||
)
|
||||
|
||||
|
||||
# -- isolation across distinct fingerprint sets ---------------------------
|
||||
|
||||
|
||||
def test_distinct_fingerprint_sets_isolated_per_signature(
|
||||
_isolated_cache_root: pathlib.Path,
|
||||
):
|
||||
"""Two entries under different signature hashes keep their own
|
||||
fingerprints; reading one with the other's fingerprints misses."""
|
||||
key_a = f"{_FRAME_ID}{KEY_DELIMITER}{'a' * 64}"
|
||||
key_b = f"{_FRAME_ID}{KEY_DELIMITER}{'b' * 64}"
|
||||
fps_a = {"contract_sha": "a" * 64}
|
||||
fps_b = {"contract_sha": "b" * 64}
|
||||
save_proposal(
|
||||
key_a,
|
||||
_proposal(payload={"sig": "a"}),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
fingerprints=fps_a,
|
||||
)
|
||||
save_proposal(
|
||||
key_b,
|
||||
_proposal(payload={"sig": "b"}),
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
fingerprints=fps_b,
|
||||
)
|
||||
# Crossed lookups miss.
|
||||
assert read_proposal(key_a, fingerprints=fps_b) is None
|
||||
assert read_proposal(key_b, fingerprints=fps_a) is None
|
||||
# Aligned lookups hit.
|
||||
a_hit = read_proposal(key_a, fingerprints=fps_a)
|
||||
b_hit = read_proposal(key_b, fingerprints=fps_b)
|
||||
assert a_hit is not None and a_hit.payload == {"sig": "a"}
|
||||
assert b_hit is not None and b_hit.payload == {"sig": "b"}
|
||||
@@ -0,0 +1,93 @@
|
||||
"""IMP-46 u6 — repository layout coverage for the persistent frame cache.
|
||||
|
||||
This module is a *layout* contract test, not a runtime test. It asserts the
|
||||
files committed to source control that make ``data/frame_cache/`` exist on a
|
||||
fresh checkout while keeping cached JSON payloads ignored by git:
|
||||
|
||||
* ``data/frame_cache/.gitkeep`` is tracked (so the cache root exists for a
|
||||
fresh clone before any AI fallback run materialises payloads).
|
||||
* ``.gitignore`` ignores ``data/*`` broadly, re-includes the
|
||||
``data/frame_cache/`` directory, ignores its contents, and re-includes
|
||||
``data/frame_cache/.gitkeep`` so cache payloads under
|
||||
``data/frame_cache/{frame_id}/{signature_hash}.json`` remain ignored.
|
||||
|
||||
If somebody removes the ``.gitkeep`` marker, drops the negation lines from
|
||||
``.gitignore``, or commits a real cache payload, this test fails. The cache
|
||||
module surface (cache.py) is exercised by ``test_cache.py`` /
|
||||
``test_cache_invalidation.py`` and is intentionally *not* re-asserted here —
|
||||
this file is the layout-only lock that Stage 2 u6 declared.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[2]
|
||||
GITIGNORE_PATH = REPO_ROOT / ".gitignore"
|
||||
CACHE_ROOT = REPO_ROOT / "data" / "frame_cache"
|
||||
GITKEEP_PATH = CACHE_ROOT / ".gitkeep"
|
||||
|
||||
|
||||
def _gitignore_lines() -> list[str]:
|
||||
assert GITIGNORE_PATH.is_file(), f".gitignore missing at {GITIGNORE_PATH}"
|
||||
text = GITIGNORE_PATH.read_text(encoding="utf-8")
|
||||
return [line.strip() for line in text.splitlines()]
|
||||
|
||||
|
||||
def test_frame_cache_root_directory_exists() -> None:
|
||||
"""``data/frame_cache/`` must exist on disk as the cache root."""
|
||||
assert CACHE_ROOT.is_dir(), (
|
||||
f"frame cache root missing: {CACHE_ROOT}. The directory must exist "
|
||||
"for save_proposal to write JSON payloads without first conjuring a "
|
||||
"parent on demand from outside the cache module."
|
||||
)
|
||||
|
||||
|
||||
def test_gitkeep_marker_is_tracked_file() -> None:
|
||||
"""``data/frame_cache/.gitkeep`` is the marker that keeps the dir tracked."""
|
||||
assert GITKEEP_PATH.is_file(), (
|
||||
f".gitkeep marker missing: {GITKEEP_PATH}. Without it the cache root "
|
||||
"would disappear on a fresh clone (everything under data/ is "
|
||||
"ignored by default)."
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"rule",
|
||||
[
|
||||
# Broad ignore for everything under data/ (cache payloads, runs/, etc.).
|
||||
"data/*",
|
||||
# Re-include the frame_cache directory itself so child negations work.
|
||||
"!data/frame_cache/",
|
||||
# Ignore everything inside frame_cache/ (cached JSON payloads).
|
||||
"data/frame_cache/*",
|
||||
# Re-include the .gitkeep marker only.
|
||||
"!data/frame_cache/.gitkeep",
|
||||
],
|
||||
)
|
||||
def test_gitignore_contains_frame_cache_exception(rule: str) -> None:
|
||||
"""The four ignore rules together pin the 'track marker only' contract."""
|
||||
lines = _gitignore_lines()
|
||||
assert rule in lines, (
|
||||
f".gitignore missing IMP-46 u6 rule: {rule!r}. The four-line block "
|
||||
"(data/*, !data/frame_cache/, data/frame_cache/*, "
|
||||
"!data/frame_cache/.gitkeep) together ensure the cache root is "
|
||||
"tracked while cached payloads remain ignored."
|
||||
)
|
||||
|
||||
|
||||
def test_gitignore_rule_order_keeps_payloads_ignored() -> None:
|
||||
"""Rule order matters: the ``data/frame_cache/*`` re-ignore must come
|
||||
AFTER the ``!data/frame_cache/`` directory re-include, otherwise the
|
||||
re-include would shadow it and cached JSON payloads would be tracked."""
|
||||
lines = _gitignore_lines()
|
||||
reinclude_dir = lines.index("!data/frame_cache/")
|
||||
reignore_contents = lines.index("data/frame_cache/*")
|
||||
reinclude_marker = lines.index("!data/frame_cache/.gitkeep")
|
||||
assert reinclude_dir < reignore_contents < reinclude_marker, (
|
||||
"gitignore IMP-46 u6 block out of order: expected "
|
||||
"'!data/frame_cache/' < 'data/frame_cache/*' < "
|
||||
"'!data/frame_cache/.gitkeep' so cached payloads stay ignored while "
|
||||
"only the marker is tracked."
|
||||
)
|
||||
@@ -0,0 +1,151 @@
|
||||
"""IMP-33 u4 — fallback client mock tests.
|
||||
|
||||
Scope (Stage 2 plan, u4):
|
||||
- Success path returns a validated ``AiFallbackProposal`` (u2 schema).
|
||||
- Transient errors (timeout / connection / 429 / 5xx) are retried.
|
||||
- Retries exhausted → last transient error propagates + consec-fail bumps.
|
||||
- Non-transient errors are NOT retried.
|
||||
- Per-run budget exhaustion raises ``AiFallbackBudgetExceeded``.
|
||||
- Circuit breaker opens after consecutive-failure threshold reached.
|
||||
- Policy values are sourced from ``settings`` (no inline literals).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import time
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
import anthropic
|
||||
import httpx
|
||||
import pytest
|
||||
|
||||
from src.config import settings
|
||||
from src.phase_z2_ai_fallback.client import (
|
||||
AiFallbackBudgetExceeded,
|
||||
AiFallbackCircuitOpen,
|
||||
AiFallbackClient,
|
||||
)
|
||||
|
||||
|
||||
class _NonTransient(Exception):
|
||||
"""Stand-in for any anthropic error not in the transient whitelist."""
|
||||
|
||||
|
||||
def _ok_response() -> SimpleNamespace:
|
||||
block = SimpleNamespace(
|
||||
text=json.dumps(
|
||||
{
|
||||
"proposal_kind": "builder_options_patch",
|
||||
"payload": {"k": 1},
|
||||
"rationale": "ok",
|
||||
}
|
||||
)
|
||||
)
|
||||
return SimpleNamespace(content=[block])
|
||||
|
||||
|
||||
def _timeout_err() -> anthropic.APITimeoutError:
|
||||
return anthropic.APITimeoutError(request=httpx.Request("POST", "https://x"))
|
||||
|
||||
|
||||
def _connection_err() -> anthropic.APIConnectionError:
|
||||
return anthropic.APIConnectionError(request=httpx.Request("POST", "https://x"))
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _no_real_sleep(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setattr(time, "sleep", lambda _s: None)
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _restore_settings():
|
||||
snapshot = settings.model_dump()
|
||||
yield
|
||||
for key, value in snapshot.items():
|
||||
setattr(settings, key, value)
|
||||
|
||||
|
||||
def _client_with(side_effect=None, return_value=None) -> AiFallbackClient:
|
||||
fake = MagicMock()
|
||||
if side_effect is not None:
|
||||
fake.messages.create.side_effect = side_effect
|
||||
else:
|
||||
fake.messages.create.return_value = return_value or _ok_response()
|
||||
return AiFallbackClient(client=fake)
|
||||
|
||||
|
||||
def test_success_returns_validated_proposal() -> None:
|
||||
out = _client_with().request_proposal({"system": "s", "user": "u"})
|
||||
assert out.proposal_kind.value == "builder_options_patch"
|
||||
assert out.payload == {"k": 1}
|
||||
|
||||
|
||||
def test_call_uses_settings_model() -> None:
|
||||
fake = MagicMock()
|
||||
fake.messages.create.return_value = _ok_response()
|
||||
AiFallbackClient(client=fake).request_proposal({"system": "s", "user": "u"})
|
||||
kwargs = fake.messages.create.call_args.kwargs
|
||||
assert kwargs["model"] == settings.ai_fallback_model
|
||||
|
||||
|
||||
def test_transient_retries_then_succeeds() -> None:
|
||||
fake = MagicMock()
|
||||
fake.messages.create.side_effect = [_timeout_err(), _connection_err(), _ok_response()]
|
||||
AiFallbackClient(client=fake).request_proposal({"system": "s", "user": "u"})
|
||||
assert fake.messages.create.call_count == 3
|
||||
|
||||
|
||||
def test_retries_exhausted_raises_last_transient(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setattr(settings, "ai_fallback_max_retries", 1)
|
||||
fake = MagicMock()
|
||||
fake.messages.create.side_effect = [_timeout_err(), _timeout_err()]
|
||||
c = AiFallbackClient(client=fake)
|
||||
with pytest.raises(anthropic.APITimeoutError):
|
||||
c.request_proposal({"system": "s", "user": "u"})
|
||||
assert fake.messages.create.call_count == 2
|
||||
assert c._consecutive_failures == 1
|
||||
|
||||
|
||||
def test_non_transient_not_retried() -> None:
|
||||
fake = MagicMock()
|
||||
fake.messages.create.side_effect = _NonTransient("boom")
|
||||
c = AiFallbackClient(client=fake)
|
||||
with pytest.raises(_NonTransient):
|
||||
c.request_proposal({"system": "s", "user": "u"})
|
||||
assert fake.messages.create.call_count == 1
|
||||
|
||||
|
||||
def test_budget_exceeded(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setattr(settings, "ai_fallback_budget_per_run", 1)
|
||||
c = _client_with()
|
||||
c.request_proposal({"system": "s", "user": "u"})
|
||||
with pytest.raises(AiFallbackBudgetExceeded):
|
||||
c.request_proposal({"system": "s", "user": "u"})
|
||||
|
||||
|
||||
def test_circuit_breaker_opens(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setattr(settings, "ai_fallback_circuit_breaker_threshold", 1)
|
||||
monkeypatch.setattr(settings, "ai_fallback_max_retries", 0)
|
||||
fake = MagicMock()
|
||||
fake.messages.create.side_effect = _timeout_err()
|
||||
c = AiFallbackClient(client=fake)
|
||||
with pytest.raises(anthropic.APITimeoutError):
|
||||
c.request_proposal({"system": "s", "user": "u"})
|
||||
with pytest.raises(AiFallbackCircuitOpen):
|
||||
c.request_proposal({"system": "s", "user": "u"})
|
||||
|
||||
|
||||
def test_backoff_uses_settings(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
"""Sleep delay must be derived from settings (no inline literals)."""
|
||||
monkeypatch.setattr(settings, "ai_fallback_max_retries", 1)
|
||||
monkeypatch.setattr(settings, "ai_fallback_backoff_base_s", 0.25)
|
||||
monkeypatch.setattr(settings, "ai_fallback_backoff_cap_s", 0.5)
|
||||
monkeypatch.setattr(settings, "ai_fallback_backoff_jitter", 0.0)
|
||||
sleeps: list[float] = []
|
||||
monkeypatch.setattr(time, "sleep", lambda s: sleeps.append(s))
|
||||
fake = MagicMock()
|
||||
fake.messages.create.side_effect = [_timeout_err(), _ok_response()]
|
||||
AiFallbackClient(client=fake).request_proposal({"system": "s", "user": "u"})
|
||||
# attempt 0 transient → sleep(min(cap, base * 2**0) + jitter==0) = 0.25
|
||||
assert sleeps == [0.25]
|
||||
@@ -0,0 +1,61 @@
|
||||
"""IMP-33 u11 — docs sync verification.
|
||||
|
||||
Verifies that the binding architecture docs reference the IMP-33 runtime
|
||||
module surface introduced by u1~u10. Scope is intentionally narrow per the
|
||||
Stage 2 plan: module path, Step 12 entry, Step 17 entry, cascade order, and
|
||||
the IMP-46 cache gate. Failure here means the docs and the code have
|
||||
drifted — fix the docs (or the code) before merging.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
DOCS_ROOT = Path(__file__).resolve().parents[2] / "docs" / "architecture"
|
||||
CARVE_OUT_DOC = DOCS_ROOT / "IMP-17-CARVE-OUT.md"
|
||||
GATE_AUDIT_DOC = DOCS_ROOT / "IMP-31-GATE-AUDIT.md"
|
||||
|
||||
|
||||
def _read(doc: Path) -> str:
|
||||
assert doc.is_file(), f"binding doc missing: {doc}"
|
||||
return doc.read_text(encoding="utf-8")
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"needle",
|
||||
[
|
||||
# Module path lock.
|
||||
"src/phase_z2_ai_fallback/",
|
||||
# Step 12 entry.
|
||||
"gather_step12_ai_repair_proposals",
|
||||
# Step 17 entry + blocked-reason sentinel.
|
||||
"gather_step17_ai_repair_proposals",
|
||||
"step17_ai_blocked_imp_34_35_prerequisites_missing",
|
||||
# Cascade order single source of truth.
|
||||
"OVERFLOW_CASCADE_ORDER",
|
||||
"(DETERMINISTIC, POPUP, AI_REPAIR, USER_OVERRIDE)",
|
||||
# IMP-46 cache gate.
|
||||
"visual_check_passed",
|
||||
"user_approved",
|
||||
"AiFallbackCacheGateError",
|
||||
# PZ-1 normal-path AI=0 invariant.
|
||||
"ai_fallback_enabled",
|
||||
],
|
||||
)
|
||||
def test_carve_out_doc_references_runtime_surface(needle: str) -> None:
|
||||
assert needle in _read(CARVE_OUT_DOC), (
|
||||
f"IMP-17-CARVE-OUT.md missing binding reference: {needle!r}"
|
||||
)
|
||||
|
||||
|
||||
def test_gate_audit_reflects_scaffolded_module() -> None:
|
||||
body = _read(GATE_AUDIT_DOC)
|
||||
assert "scaffolded under IMP-33" in body, (
|
||||
"IMP-31-GATE-AUDIT.md must record that the fallback module path is "
|
||||
"scaffolded (not 'not created this cycle')."
|
||||
)
|
||||
assert "ai_fallback_enabled" in body, (
|
||||
"IMP-31-GATE-AUDIT.md must record the flag default that keeps PZ-1 "
|
||||
"(normal-path AI=0) intact while the 3-condition gate is open."
|
||||
)
|
||||
@@ -0,0 +1,100 @@
|
||||
"""IMP-33 u3 — fallback prompt builder tests.
|
||||
|
||||
Scope (Stage 2 plan, u3):
|
||||
- Prompt is built only when V4 route == 'ai_adaptation_required'.
|
||||
- System prompt declares MDX READ-ONLY and pins the u2 whitelist.
|
||||
- System prompt forbids the u2 forbidden kinds + frame_id swap.
|
||||
- User payload carries all 6 declared inputs and labels MDX READ_ONLY.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
from src.phase_z2_ai_fallback.prompts import (
|
||||
SYSTEM_PROMPT,
|
||||
V4_ROUTE_AI_ADAPTATION,
|
||||
build_ai_fallback_prompt,
|
||||
)
|
||||
from src.phase_z2_ai_fallback.schema import FORBIDDEN_KINDS, ProposalKind
|
||||
|
||||
|
||||
def _v4(route: str = V4_ROUTE_AI_ADAPTATION) -> dict:
|
||||
return {
|
||||
"route": route,
|
||||
"cardinality": {"strict": 3},
|
||||
"label": "restructure",
|
||||
"frame_id": 1171281190,
|
||||
"rank": 1,
|
||||
}
|
||||
|
||||
|
||||
def _inputs(route: str = V4_ROUTE_AI_ADAPTATION) -> dict:
|
||||
return {
|
||||
"v4_result": _v4(route),
|
||||
"frame_contract": {"template_id": "three_parallel_requirements"},
|
||||
"frame_visual_html": "<section class='f13b'/>",
|
||||
"figma_partial_json": {"nodes": []},
|
||||
"internal_region": {"id": "region_top", "bbox": [0, 0, 1200, 320]},
|
||||
"mdx_text": "# 대목차\n- 항목 1\n- 항목 2\n- 항목 3",
|
||||
}
|
||||
|
||||
|
||||
def test_system_prompt_declares_mdx_read_only() -> None:
|
||||
assert "READ-ONLY" in SYSTEM_PROMPT
|
||||
|
||||
|
||||
def test_system_prompt_lists_all_whitelisted_kinds() -> None:
|
||||
for kind in ProposalKind:
|
||||
assert kind.value in SYSTEM_PROMPT
|
||||
|
||||
|
||||
def test_system_prompt_forbids_all_forbidden_kinds() -> None:
|
||||
for forbidden in FORBIDDEN_KINDS:
|
||||
assert forbidden in SYSTEM_PROMPT
|
||||
|
||||
|
||||
def test_system_prompt_locks_frame_id_swap() -> None:
|
||||
assert "frame_id" in SYSTEM_PROMPT
|
||||
|
||||
|
||||
def test_build_prompt_returns_system_and_user() -> None:
|
||||
prompt = build_ai_fallback_prompt(**_inputs())
|
||||
assert set(prompt.keys()) == {"system", "user"}
|
||||
assert prompt["system"] == SYSTEM_PROMPT
|
||||
|
||||
|
||||
def test_user_payload_carries_all_inputs_and_marks_mdx_read_only() -> None:
|
||||
prompt = build_ai_fallback_prompt(**_inputs())
|
||||
payload = json.loads(prompt["user"])
|
||||
assert payload["v4"]["route"] == V4_ROUTE_AI_ADAPTATION
|
||||
assert payload["v4"]["cardinality"] == {"strict": 3}
|
||||
assert payload["v4"]["frame_id"] == 1171281190
|
||||
assert payload["frame_contract"]["template_id"] == "three_parallel_requirements"
|
||||
assert payload["frame_visual_html"] == "<section class='f13b'/>"
|
||||
assert payload["figma_partial_json"] == {"nodes": []}
|
||||
assert payload["internal_region"]["id"] == "region_top"
|
||||
assert "mdx_text_READ_ONLY" in payload
|
||||
assert payload["mdx_text_READ_ONLY"].startswith("# 대목차")
|
||||
assert "mdx_text" not in payload # only the READ_ONLY key, not a writable alias
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"route", ["direct_render", "deterministic_minor_adjustment", "design_reference_only", None]
|
||||
)
|
||||
def test_non_ai_route_rejected(route) -> None:
|
||||
inputs = _inputs(route=route) if route is not None else _inputs()
|
||||
if route is None:
|
||||
inputs["v4_result"].pop("route")
|
||||
with pytest.raises(ValueError, match=V4_ROUTE_AI_ADAPTATION):
|
||||
build_ai_fallback_prompt(**inputs)
|
||||
|
||||
|
||||
def test_cardinality_signature_alias_accepted() -> None:
|
||||
"""Some V4 callers expose ``cardinality_signature``; both keys must resolve."""
|
||||
inputs = _inputs()
|
||||
inputs["v4_result"].pop("cardinality")
|
||||
inputs["v4_result"]["cardinality_signature"] = {"strict": 4}
|
||||
payload = json.loads(build_ai_fallback_prompt(**inputs)["user"])
|
||||
assert payload["v4"]["cardinality"] == {"strict": 4}
|
||||
@@ -0,0 +1,156 @@
|
||||
"""IMP-33 u7 — AI fallback router tests.
|
||||
|
||||
Scope (Stage 2 plan, u7):
|
||||
- flag-off gate returns None and does NOT touch the client / prompt
|
||||
- route-mismatch gate returns None and does NOT touch the client / prompt
|
||||
- cache-hit short-circuits the client and still re-validates against the
|
||||
current frame contract (defence-in-depth)
|
||||
- cache-miss calls the client and validates the returned proposal
|
||||
- validation errors propagate
|
||||
- budget / circuit exceptions from u4 propagate
|
||||
- router never imports ``save_proposal`` (cache save is caller-driven
|
||||
after visual_check + user_approved per u6 IMP-46 gate)
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
import pytest
|
||||
|
||||
from src.phase_z2_ai_fallback import AiFallbackProposal, ProposalKind
|
||||
from src.phase_z2_ai_fallback import router as router_mod
|
||||
from src.phase_z2_ai_fallback.client import (
|
||||
AiFallbackBudgetExceeded,
|
||||
AiFallbackCircuitOpen,
|
||||
AiFallbackClient,
|
||||
)
|
||||
from src.phase_z2_ai_fallback.router import route_ai_fallback
|
||||
from src.phase_z2_ai_fallback.validate import AiFallbackValidationError
|
||||
|
||||
|
||||
_FRAME_CONTRACT = {
|
||||
"frame_id": 1171281190,
|
||||
"sub_zones": [{"id": "pillar_1", "accepts": ["text_block"]}],
|
||||
"payload": {"builder_options": {"item_parser": "pillar_item"}},
|
||||
}
|
||||
_REGION = {"id": "zone_top.region_a"}
|
||||
_V4_AI = {
|
||||
"route": "ai_adaptation_required",
|
||||
"cardinality": "many",
|
||||
"frame_id": 1171281190,
|
||||
"rank": 1,
|
||||
}
|
||||
_V4_NOT_AI = {"route": "light_edit", "cardinality": "many"}
|
||||
|
||||
|
||||
def _make_proposal(
|
||||
kind: ProposalKind = ProposalKind.PARTIAL_OVERRIDES,
|
||||
payload: dict | None = None,
|
||||
) -> AiFallbackProposal:
|
||||
return AiFallbackProposal(
|
||||
proposal_kind=kind,
|
||||
payload=payload if payload is not None else {"slots": {"pillar_1": "a"}},
|
||||
)
|
||||
|
||||
|
||||
def _call_kwargs() -> dict:
|
||||
return dict(
|
||||
cache_key="frame:1171281190:cardinality:many",
|
||||
v4_result=_V4_AI,
|
||||
frame_contract=_FRAME_CONTRACT,
|
||||
frame_visual_html="<div></div>",
|
||||
figma_partial_json={},
|
||||
internal_region=_REGION,
|
||||
mdx_text="# example\n- a\n- b",
|
||||
)
|
||||
|
||||
|
||||
def test_router_returns_none_when_flag_off(monkeypatch):
|
||||
monkeypatch.setattr(router_mod.settings, "ai_fallback_enabled", False)
|
||||
client = MagicMock(spec=AiFallbackClient)
|
||||
result = route_ai_fallback(**_call_kwargs(), client=client)
|
||||
assert result is None
|
||||
client.request_proposal.assert_not_called()
|
||||
|
||||
|
||||
def test_router_returns_none_when_route_not_ai_adaptation(monkeypatch):
|
||||
monkeypatch.setattr(router_mod.settings, "ai_fallback_enabled", True)
|
||||
client = MagicMock(spec=AiFallbackClient)
|
||||
kwargs = _call_kwargs()
|
||||
kwargs["v4_result"] = _V4_NOT_AI
|
||||
result = route_ai_fallback(**kwargs, client=client)
|
||||
assert result is None
|
||||
client.request_proposal.assert_not_called()
|
||||
|
||||
|
||||
def test_router_returns_cached_when_cache_hit(monkeypatch):
|
||||
monkeypatch.setattr(router_mod.settings, "ai_fallback_enabled", True)
|
||||
cached = _make_proposal()
|
||||
monkeypatch.setattr(router_mod, "read_proposal", lambda key: cached)
|
||||
client = MagicMock(spec=AiFallbackClient)
|
||||
result = route_ai_fallback(**_call_kwargs(), client=client)
|
||||
assert result is cached
|
||||
client.request_proposal.assert_not_called()
|
||||
|
||||
|
||||
def test_router_validates_cached_proposal(monkeypatch):
|
||||
monkeypatch.setattr(router_mod.settings, "ai_fallback_enabled", True)
|
||||
bad_cached = AiFallbackProposal(
|
||||
proposal_kind=ProposalKind.BUILDER_OPTIONS_PATCH,
|
||||
payload={"unknown_key": "x"},
|
||||
)
|
||||
monkeypatch.setattr(router_mod, "read_proposal", lambda key: bad_cached)
|
||||
client = MagicMock(spec=AiFallbackClient)
|
||||
with pytest.raises(AiFallbackValidationError):
|
||||
route_ai_fallback(**_call_kwargs(), client=client)
|
||||
client.request_proposal.assert_not_called()
|
||||
|
||||
|
||||
def test_router_calls_client_and_returns_validated_proposal(monkeypatch):
|
||||
monkeypatch.setattr(router_mod.settings, "ai_fallback_enabled", True)
|
||||
monkeypatch.setattr(router_mod, "read_proposal", lambda key: None)
|
||||
proposal = _make_proposal()
|
||||
client = MagicMock(spec=AiFallbackClient)
|
||||
client.request_proposal.return_value = proposal
|
||||
result = route_ai_fallback(**_call_kwargs(), client=client)
|
||||
assert result is proposal
|
||||
client.request_proposal.assert_called_once()
|
||||
sent_prompt = client.request_proposal.call_args.args[0]
|
||||
assert set(sent_prompt.keys()) == {"system", "user"}
|
||||
|
||||
|
||||
def test_router_propagates_validation_error(monkeypatch):
|
||||
monkeypatch.setattr(router_mod.settings, "ai_fallback_enabled", True)
|
||||
monkeypatch.setattr(router_mod, "read_proposal", lambda key: None)
|
||||
bad = AiFallbackProposal(
|
||||
proposal_kind=ProposalKind.BUILDER_OPTIONS_PATCH,
|
||||
payload={"unknown_key": "x"},
|
||||
)
|
||||
client = MagicMock(spec=AiFallbackClient)
|
||||
client.request_proposal.return_value = bad
|
||||
with pytest.raises(AiFallbackValidationError):
|
||||
route_ai_fallback(**_call_kwargs(), client=client)
|
||||
|
||||
|
||||
def test_router_propagates_budget_exceeded(monkeypatch):
|
||||
monkeypatch.setattr(router_mod.settings, "ai_fallback_enabled", True)
|
||||
monkeypatch.setattr(router_mod, "read_proposal", lambda key: None)
|
||||
client = MagicMock(spec=AiFallbackClient)
|
||||
client.request_proposal.side_effect = AiFallbackBudgetExceeded("over")
|
||||
with pytest.raises(AiFallbackBudgetExceeded):
|
||||
route_ai_fallback(**_call_kwargs(), client=client)
|
||||
|
||||
|
||||
def test_router_propagates_circuit_open(monkeypatch):
|
||||
monkeypatch.setattr(router_mod.settings, "ai_fallback_enabled", True)
|
||||
monkeypatch.setattr(router_mod, "read_proposal", lambda key: None)
|
||||
client = MagicMock(spec=AiFallbackClient)
|
||||
client.request_proposal.side_effect = AiFallbackCircuitOpen("tripped")
|
||||
with pytest.raises(AiFallbackCircuitOpen):
|
||||
route_ai_fallback(**_call_kwargs(), client=client)
|
||||
|
||||
|
||||
def test_router_does_not_import_save_proposal():
|
||||
"""Cache save is caller-driven AFTER visual_check + user_approved (u6 IMP-46
|
||||
gate); structurally guaranteed by NOT importing save_proposal in the router."""
|
||||
assert not hasattr(router_mod, "save_proposal")
|
||||
@@ -0,0 +1,46 @@
|
||||
"""IMP-33 u2 — AiFallbackProposal schema tests.
|
||||
|
||||
Scope (Stage 2 plan, u2):
|
||||
- Whitelisted proposal_kind values are accepted.
|
||||
- Forbidden output forms are rejected: mdx_text / frame_id_change / raw_html / raw_css.
|
||||
- extra fields outside the declared schema are rejected (MDX read-only signal).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
from pydantic import ValidationError
|
||||
|
||||
from src.phase_z2_ai_fallback import AiFallbackProposal, ProposalKind
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"kind_value",
|
||||
[
|
||||
"builder_options_patch",
|
||||
"partial_overrides",
|
||||
"slot_mapping_proposal",
|
||||
],
|
||||
)
|
||||
def test_whitelisted_proposal_kinds_accepted(kind_value: str) -> None:
|
||||
proposal = AiFallbackProposal(proposal_kind=kind_value)
|
||||
assert proposal.proposal_kind == ProposalKind(kind_value)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"forbidden",
|
||||
["mdx_text", "frame_id_change", "raw_html", "raw_css"],
|
||||
)
|
||||
def test_forbidden_proposal_kinds_rejected(forbidden: str) -> None:
|
||||
with pytest.raises(ValidationError):
|
||||
AiFallbackProposal(proposal_kind=forbidden)
|
||||
|
||||
|
||||
def test_unknown_proposal_kind_rejected() -> None:
|
||||
with pytest.raises(ValidationError):
|
||||
AiFallbackProposal(proposal_kind="something_else")
|
||||
|
||||
|
||||
def test_extra_fields_rejected() -> None:
|
||||
"""`extra=forbid` keeps the AI from smuggling raw_html/mdx_text alongside a valid kind."""
|
||||
with pytest.raises(ValidationError):
|
||||
AiFallbackProposal(proposal_kind="partial_overrides", raw_html="<div/>")
|
||||
@@ -0,0 +1,184 @@
|
||||
"""IMP-46 u1 — Frame cache signature builder tests.
|
||||
|
||||
Verifies:
|
||||
* Determinism — identical inputs yield the same SHA256 digest.
|
||||
* Axis-change sensitivity — every one of the 8 declared axes mutates the
|
||||
digest when changed in isolation.
|
||||
* Public surface — only the 8 declared axes are accepted (no
|
||||
sample/section identifier leakage).
|
||||
* char_count bucket boundaries (0-50, 51-150, 151-400, 401-1000, 1001+).
|
||||
* source_shape enum equivalence (string and SourceShape inputs match).
|
||||
* schema_version is part of the hashed payload (digest stable for fixture).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import inspect
|
||||
|
||||
import pytest
|
||||
|
||||
from src.phase_z2_ai_fallback.signature import (
|
||||
CHAR_COUNT_BUCKET_LABELS,
|
||||
SCHEMA_VERSION,
|
||||
SourceShape,
|
||||
bucket_char_count,
|
||||
build_signature,
|
||||
)
|
||||
|
||||
|
||||
def _base_kwargs() -> dict:
|
||||
return dict(
|
||||
frame_id="frame_03",
|
||||
v4_label="light_edit",
|
||||
cardinality=3,
|
||||
source_shape=SourceShape.BULLET,
|
||||
h3_count=2,
|
||||
char_count_bucket="51-150",
|
||||
layout_preset="sidebar-right",
|
||||
zone_position="top",
|
||||
)
|
||||
|
||||
|
||||
def test_schema_version_is_one() -> None:
|
||||
assert SCHEMA_VERSION == 1
|
||||
|
||||
|
||||
def test_bucket_labels_match_spec() -> None:
|
||||
assert CHAR_COUNT_BUCKET_LABELS == (
|
||||
"0-50",
|
||||
"51-150",
|
||||
"151-400",
|
||||
"401-1000",
|
||||
"1001+",
|
||||
)
|
||||
|
||||
|
||||
def test_signature_is_deterministic() -> None:
|
||||
a = build_signature(**_base_kwargs())
|
||||
b = build_signature(**_base_kwargs())
|
||||
assert a == b
|
||||
assert len(a) == 64
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"axis, new_value",
|
||||
[
|
||||
("frame_id", "frame_04"),
|
||||
("v4_label", "restructure"),
|
||||
("cardinality", 5),
|
||||
("source_shape", SourceShape.PARAGRAPH),
|
||||
("h3_count", 3),
|
||||
("char_count_bucket", "151-400"),
|
||||
("layout_preset", "two-column"),
|
||||
("zone_position", "bottom_l"),
|
||||
],
|
||||
)
|
||||
def test_signature_changes_for_each_axis(axis: str, new_value: object) -> None:
|
||||
base = build_signature(**_base_kwargs())
|
||||
kwargs = _base_kwargs()
|
||||
kwargs[axis] = new_value
|
||||
assert build_signature(**kwargs) != base
|
||||
|
||||
|
||||
def test_signature_accepts_string_source_shape() -> None:
|
||||
enum_sig = build_signature(**_base_kwargs())
|
||||
kwargs = _base_kwargs()
|
||||
kwargs["source_shape"] = "bullet"
|
||||
assert build_signature(**kwargs) == enum_sig
|
||||
|
||||
|
||||
def test_signature_rejects_unknown_source_shape() -> None:
|
||||
kwargs = _base_kwargs()
|
||||
kwargs["source_shape"] = "nonsense"
|
||||
with pytest.raises(ValueError):
|
||||
build_signature(**kwargs)
|
||||
|
||||
|
||||
def test_signature_rejects_unknown_char_count_bucket() -> None:
|
||||
kwargs = _base_kwargs()
|
||||
kwargs["char_count_bucket"] = "999-1234"
|
||||
with pytest.raises(ValueError):
|
||||
build_signature(**kwargs)
|
||||
|
||||
|
||||
def test_signature_handles_none_cardinality() -> None:
|
||||
kwargs = _base_kwargs()
|
||||
kwargs["cardinality"] = None
|
||||
sig = build_signature(**kwargs)
|
||||
assert len(sig) == 64
|
||||
kwargs2 = _base_kwargs()
|
||||
kwargs2["cardinality"] = 0
|
||||
assert build_signature(**kwargs2) != sig
|
||||
|
||||
|
||||
def test_signature_surface_only_8_declared_axes() -> None:
|
||||
params = set(inspect.signature(build_signature).parameters)
|
||||
expected = {
|
||||
"frame_id",
|
||||
"v4_label",
|
||||
"cardinality",
|
||||
"source_shape",
|
||||
"h3_count",
|
||||
"char_count_bucket",
|
||||
"layout_preset",
|
||||
"zone_position",
|
||||
}
|
||||
assert params == expected
|
||||
|
||||
|
||||
def test_bucket_boundaries() -> None:
|
||||
assert bucket_char_count(0) == "0-50"
|
||||
assert bucket_char_count(50) == "0-50"
|
||||
assert bucket_char_count(51) == "51-150"
|
||||
assert bucket_char_count(150) == "51-150"
|
||||
assert bucket_char_count(151) == "151-400"
|
||||
assert bucket_char_count(400) == "151-400"
|
||||
assert bucket_char_count(401) == "401-1000"
|
||||
assert bucket_char_count(1000) == "401-1000"
|
||||
assert bucket_char_count(1001) == "1001+"
|
||||
assert bucket_char_count(10_000) == "1001+"
|
||||
|
||||
|
||||
def test_bucket_rejects_negative() -> None:
|
||||
with pytest.raises(ValueError):
|
||||
bucket_char_count(-1)
|
||||
|
||||
|
||||
def test_bucket_rejects_non_int() -> None:
|
||||
with pytest.raises(TypeError):
|
||||
bucket_char_count(3.14) # type: ignore[arg-type]
|
||||
with pytest.raises(TypeError):
|
||||
bucket_char_count(True) # type: ignore[arg-type]
|
||||
|
||||
|
||||
def test_signature_stable_known_fixture() -> None:
|
||||
"""Lock the digest for a known fixture so a silent payload-shape change
|
||||
(e.g. a new axis sneaks in, or schema_version drifts) breaks this test.
|
||||
"""
|
||||
sig = build_signature(
|
||||
frame_id="frame_03",
|
||||
v4_label="light_edit",
|
||||
cardinality=3,
|
||||
source_shape=SourceShape.BULLET,
|
||||
h3_count=2,
|
||||
char_count_bucket="51-150",
|
||||
layout_preset="sidebar-right",
|
||||
zone_position="top",
|
||||
)
|
||||
import hashlib
|
||||
import json
|
||||
|
||||
expected_payload = {
|
||||
"schema_version": 1,
|
||||
"frame_id": "frame_03",
|
||||
"v4_label": "light_edit",
|
||||
"cardinality": 3,
|
||||
"source_shape": "bullet",
|
||||
"h3_count": 2,
|
||||
"char_count_bucket": "51-150",
|
||||
"layout_preset": "sidebar-right",
|
||||
"zone_position": "top",
|
||||
}
|
||||
expected = hashlib.sha256(
|
||||
json.dumps(expected_payload, sort_keys=True, ensure_ascii=False).encode("utf-8")
|
||||
).hexdigest()
|
||||
assert sig == expected
|
||||
@@ -0,0 +1,502 @@
|
||||
"""IMP-33 u8 + IMP-46 u4 + IMP-47B u2 — Step 12 AI repair wiring tests.
|
||||
|
||||
Covers the structural gates layered on top of the u7 router:
|
||||
* IMP-30 provisional gate (only provisional units may invoke AI repair)
|
||||
* Catch-all ``route_not_ai_adaptation:<hint>`` skip — every route_hint
|
||||
other than ``ai_adaptation_required`` (including the legacy
|
||||
``design_reference_only`` hint) falls through to a single uniform skip
|
||||
after the IMP-47B u2 removal of the bespoke reject gate.
|
||||
Plus the record-shape contract returned for downstream Step 12 artifacts
|
||||
and the IMP-46 u4 structural cache key + fingerprints contract.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from src.phase_z2_ai_fallback import step12 as step12_mod
|
||||
from src.phase_z2_ai_fallback.schema import AiFallbackProposal, ProposalKind
|
||||
|
||||
|
||||
@dataclass
|
||||
class FakeUnit:
|
||||
label: str | None
|
||||
provisional: bool
|
||||
frame_template_id: str = "tmpl"
|
||||
frame_id: str = "fid"
|
||||
source_section_ids: list[str] = field(default_factory=lambda: ["s1"])
|
||||
raw_content: str = "raw"
|
||||
v4_rank: int | None = 1
|
||||
cardinality: int | None = None
|
||||
layout_preset: str = ""
|
||||
zone_position: str = ""
|
||||
source_shape: str = "paragraph"
|
||||
h3_count: int = 0
|
||||
char_count: int = 0
|
||||
|
||||
|
||||
_ROUTE_HINTS: dict[str | None, str | None] = {
|
||||
"use_as_is": "direct_render",
|
||||
"light_edit": "deterministic_minor_adjustment",
|
||||
"restructure": "ai_adaptation_required",
|
||||
"reject": "design_reference_only",
|
||||
None: None,
|
||||
}
|
||||
|
||||
|
||||
def _route_for_label(label: str | None) -> str | None:
|
||||
return _ROUTE_HINTS.get(label)
|
||||
|
||||
|
||||
def _get_contract(_tid: str) -> dict[str, Any]:
|
||||
return {"frame_id": "fid", "payload": {"builder_options": {}}, "sub_zones": []}
|
||||
|
||||
|
||||
def _frame_visual(_tid: str) -> str:
|
||||
return "<html></html>"
|
||||
|
||||
|
||||
def _call(
|
||||
units: list[FakeUnit],
|
||||
*,
|
||||
route_ai_fallback: Any | None = None,
|
||||
**overrides: Any,
|
||||
) -> list[dict]:
|
||||
if route_ai_fallback is not None:
|
||||
step12_mod.route_ai_fallback = route_ai_fallback # type: ignore[assignment]
|
||||
kwargs: dict[str, Any] = dict(
|
||||
route_for_label=_route_for_label,
|
||||
get_contract_fn=_get_contract,
|
||||
frame_visual_loader=_frame_visual,
|
||||
)
|
||||
kwargs.update(overrides)
|
||||
return step12_mod.gather_step12_ai_repair_proposals(units, **kwargs)
|
||||
|
||||
|
||||
def _ai_unit(**overrides: Any) -> FakeUnit:
|
||||
"""Construct an AI-eligible FakeUnit (provisional + restructure) with sane defaults."""
|
||||
base: dict[str, Any] = dict(
|
||||
label="restructure",
|
||||
provisional=True,
|
||||
frame_template_id="tmpl_x",
|
||||
frame_id="fid_123",
|
||||
source_section_ids=["02-1"],
|
||||
layout_preset="single_column",
|
||||
zone_position="zone_a",
|
||||
source_shape="bullet",
|
||||
h3_count=3,
|
||||
char_count=200,
|
||||
cardinality=5,
|
||||
)
|
||||
base.update(overrides)
|
||||
return FakeUnit(**base)
|
||||
|
||||
|
||||
def test_non_provisional_unit_is_skipped_without_ai_call(monkeypatch):
|
||||
router = MagicMock()
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", router)
|
||||
units = [FakeUnit(label="restructure", provisional=False)]
|
||||
records = _call(units)
|
||||
assert records[0]["ai_called"] is False
|
||||
assert records[0]["skip_reason"] == "not_provisional"
|
||||
assert records[0]["provisional"] is False
|
||||
router.assert_not_called()
|
||||
|
||||
|
||||
def test_design_reference_route_falls_through_to_route_not_ai_adaptation(monkeypatch):
|
||||
"""IMP-47B u2 — the bespoke 'design_reference_only_no_ai' skip is gone.
|
||||
|
||||
Any non-AI-adaptation route_hint (including the legacy
|
||||
``design_reference_only`` hint exercised here via the local test mapping
|
||||
of ``reject``) now flows into the single ``route_not_ai_adaptation:<hint>``
|
||||
catch-all. Production reject routing is exercised by u9.
|
||||
"""
|
||||
router = MagicMock()
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", router)
|
||||
units = [FakeUnit(label="reject", provisional=True)]
|
||||
records = _call(units)
|
||||
assert records[0]["ai_called"] is False
|
||||
assert records[0]["skip_reason"] == "route_not_ai_adaptation:design_reference_only"
|
||||
assert records[0]["route_hint"] == "design_reference_only"
|
||||
router.assert_not_called()
|
||||
|
||||
|
||||
def test_non_ai_route_is_skipped_with_reason(monkeypatch):
|
||||
router = MagicMock()
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", router)
|
||||
units = [FakeUnit(label="light_edit", provisional=True)]
|
||||
records = _call(units)
|
||||
assert records[0]["ai_called"] is False
|
||||
assert records[0]["skip_reason"] == (
|
||||
"route_not_ai_adaptation:deterministic_minor_adjustment"
|
||||
)
|
||||
router.assert_not_called()
|
||||
|
||||
|
||||
def test_router_short_circuit_returns_none_skip_reason(monkeypatch):
|
||||
router = MagicMock(return_value=None)
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", router)
|
||||
units = [FakeUnit(label="restructure", provisional=True)]
|
||||
records = _call(units)
|
||||
assert records[0]["ai_called"] is False
|
||||
assert records[0]["skip_reason"] == "router_short_circuit"
|
||||
assert records[0]["proposal"] is None
|
||||
router.assert_called_once()
|
||||
|
||||
|
||||
def test_ai_adaptation_call_records_proposal(monkeypatch):
|
||||
proposal = AiFallbackProposal(
|
||||
proposal_kind=ProposalKind.PARTIAL_OVERRIDES,
|
||||
payload={"slots": {"s_text": "x"}},
|
||||
rationale="r",
|
||||
)
|
||||
router = MagicMock(return_value=proposal)
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", router)
|
||||
units = [FakeUnit(label="restructure", provisional=True)]
|
||||
records = _call(units)
|
||||
rec = records[0]
|
||||
assert rec["ai_called"] is True
|
||||
assert rec["skip_reason"] is None
|
||||
assert rec["proposal"]["proposal_kind"] == "partial_overrides"
|
||||
router.assert_called_once()
|
||||
kwargs = router.call_args.kwargs
|
||||
assert kwargs["v4_result"]["route"] == "ai_adaptation_required"
|
||||
assert kwargs["v4_result"]["label"] == "restructure"
|
||||
|
||||
|
||||
def test_router_exception_is_captured_per_record(monkeypatch):
|
||||
router = MagicMock(side_effect=RuntimeError("transient_boom"))
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", router)
|
||||
units = [FakeUnit(label="restructure", provisional=True)]
|
||||
records = _call(units)
|
||||
rec = records[0]
|
||||
assert rec["ai_called"] is True
|
||||
assert rec["proposal"] is None
|
||||
assert rec["error"] == "RuntimeError: transient_boom"
|
||||
router.assert_called_once()
|
||||
|
||||
|
||||
def test_mixed_units_each_independently_classified(monkeypatch):
|
||||
router = MagicMock(return_value=None)
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", router)
|
||||
units = [
|
||||
FakeUnit(label="use_as_is", provisional=False),
|
||||
FakeUnit(label="reject", provisional=True),
|
||||
FakeUnit(label="restructure", provisional=True),
|
||||
FakeUnit(label="restructure", provisional=False),
|
||||
]
|
||||
records = _call(units)
|
||||
assert [r["skip_reason"] for r in records] == [
|
||||
"not_provisional",
|
||||
"route_not_ai_adaptation:design_reference_only",
|
||||
"router_short_circuit",
|
||||
"not_provisional",
|
||||
]
|
||||
assert router.call_count == 1
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# IMP-46 u4 — structural cache key + fingerprints
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_cache_key_format_is_frame_id_plus_sha256(monkeypatch):
|
||||
"""cache_key is '{frame_id}::{64-hex-sha256}', NOT template_id + section_ids."""
|
||||
router = MagicMock(return_value=None)
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", router)
|
||||
_call([_ai_unit()])
|
||||
cache_key = router.call_args.kwargs["cache_key"]
|
||||
assert "::" in cache_key
|
||||
frame_part, _, signature_part = cache_key.partition("::")
|
||||
assert frame_part == "fid_123"
|
||||
assert len(signature_part) == 64
|
||||
assert all(c in "0123456789abcdef" for c in signature_part)
|
||||
# The legacy "template_id::sorted(section_ids)" form is gone.
|
||||
assert "tmpl_x" not in cache_key
|
||||
assert "02-1" not in cache_key
|
||||
|
||||
|
||||
def test_cache_key_invariant_to_section_id_changes(monkeypatch):
|
||||
"""Same structural axes → same cache_key regardless of source_section_ids."""
|
||||
router = MagicMock(return_value=None)
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", router)
|
||||
_call([_ai_unit(source_section_ids=["02-1"])])
|
||||
key_a = router.call_args.kwargs["cache_key"]
|
||||
router.reset_mock()
|
||||
_call([_ai_unit(source_section_ids=["05-2", "07-3"])])
|
||||
key_b = router.call_args.kwargs["cache_key"]
|
||||
assert key_a == key_b
|
||||
|
||||
|
||||
def test_cache_key_invariant_to_template_id_changes(monkeypatch):
|
||||
"""frame_template_id is NOT part of the structural signature (frame_id is)."""
|
||||
router = MagicMock(return_value=None)
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", router)
|
||||
_call([_ai_unit(frame_template_id="tmpl_x")])
|
||||
key_a = router.call_args.kwargs["cache_key"]
|
||||
router.reset_mock()
|
||||
_call([_ai_unit(frame_template_id="tmpl_OTHER")])
|
||||
key_b = router.call_args.kwargs["cache_key"]
|
||||
assert key_a == key_b
|
||||
|
||||
|
||||
def test_cache_key_changes_when_any_signature_axis_changes(monkeypatch):
|
||||
"""Flipping any of the 7 unit-derived signature axes mutates cache_key."""
|
||||
router = MagicMock(return_value=None)
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", router)
|
||||
_call([_ai_unit()])
|
||||
base_key = router.call_args.kwargs["cache_key"]
|
||||
perturbations: dict[str, Any] = {
|
||||
"frame_id": "fid_OTHER",
|
||||
"label": "use_as_is", # v4_label axis change; still routed to AI via _ROUTE_HINTS? No.
|
||||
# ↑ "use_as_is" → "direct_render" → would skip. Use another ai-adaptation-mapped label.
|
||||
# Replace with frame_id-only diff to keep route stable. Drop this entry below.
|
||||
}
|
||||
# Rebuild perturbations restricted to axes that don't change routing.
|
||||
perturbations = {
|
||||
"frame_id": "fid_OTHER",
|
||||
"layout_preset": "two_column",
|
||||
"zone_position": "zone_b",
|
||||
"source_shape": "paragraph",
|
||||
"h3_count": 7,
|
||||
"char_count": 500, # bucket boundary crossing (151-400 → 401-1000)
|
||||
"cardinality": 4,
|
||||
}
|
||||
for axis, value in perturbations.items():
|
||||
router.reset_mock()
|
||||
_call([_ai_unit(**{axis: value})])
|
||||
new_key = router.call_args.kwargs["cache_key"]
|
||||
assert new_key != base_key, f"signature axis {axis!r} did not mutate cache_key"
|
||||
|
||||
|
||||
def test_char_count_bucket_collapses_within_bucket(monkeypatch):
|
||||
"""Different char_counts in the SAME bucket → identical cache_key."""
|
||||
router = MagicMock(return_value=None)
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", router)
|
||||
_call([_ai_unit(char_count=160)])
|
||||
key_low = router.call_args.kwargs["cache_key"]
|
||||
router.reset_mock()
|
||||
_call([_ai_unit(char_count=399)])
|
||||
key_high = router.call_args.kwargs["cache_key"]
|
||||
assert key_low == key_high # both fall in "151-400"
|
||||
router.reset_mock()
|
||||
_call([_ai_unit(char_count=401)])
|
||||
key_overflow = router.call_args.kwargs["cache_key"]
|
||||
assert key_overflow != key_low # crossed into "401-1000"
|
||||
|
||||
|
||||
def test_fingerprints_attached_to_ai_record(monkeypatch):
|
||||
"""AI-called records expose contract_sha + partial_sha + catalog_sha."""
|
||||
router = MagicMock(return_value=None)
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", router)
|
||||
contract = {"frame_id": "fid", "payload": {"x": 1}, "sub_zones": []}
|
||||
partial = {"some": "partial", "deeper": [1, 2, 3]}
|
||||
catalog_value = "deadbeef" * 8
|
||||
recs = _call(
|
||||
[_ai_unit()],
|
||||
get_contract_fn=lambda _t: contract,
|
||||
figma_partial_loader=lambda _t: partial,
|
||||
catalog_sha_loader=lambda: catalog_value,
|
||||
)
|
||||
fps = recs[0]["fingerprints"]
|
||||
assert isinstance(fps, dict)
|
||||
assert set(fps.keys()) == {"contract_sha", "partial_sha", "catalog_sha"}
|
||||
assert all(isinstance(v, str) for v in fps.values())
|
||||
assert fps["catalog_sha"] == catalog_value
|
||||
# contract_sha and partial_sha must be deterministic SHA256 over JSON-sorted payloads.
|
||||
expected_contract = hashlib.sha256(
|
||||
json.dumps(contract, sort_keys=True, ensure_ascii=False).encode("utf-8")
|
||||
).hexdigest()
|
||||
expected_partial = hashlib.sha256(
|
||||
json.dumps(partial, sort_keys=True, ensure_ascii=False).encode("utf-8")
|
||||
).hexdigest()
|
||||
assert fps["contract_sha"] == expected_contract
|
||||
assert fps["partial_sha"] == expected_partial
|
||||
|
||||
|
||||
def test_fingerprints_default_catalog_sha_is_empty_string(monkeypatch):
|
||||
"""No catalog_sha_loader → catalog_sha defaults to '' (sentinel, not missing key)."""
|
||||
router = MagicMock(return_value=None)
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", router)
|
||||
recs = _call([_ai_unit()])
|
||||
fps = recs[0]["fingerprints"]
|
||||
assert fps["catalog_sha"] == ""
|
||||
# contract_sha + partial_sha keys still present (always 3 keys).
|
||||
assert set(fps.keys()) == {"contract_sha", "partial_sha", "catalog_sha"}
|
||||
|
||||
|
||||
def test_fingerprints_change_when_contract_changes(monkeypatch):
|
||||
"""Different frame_contract → different contract_sha, partial_sha unchanged."""
|
||||
router = MagicMock(return_value=None)
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", router)
|
||||
fps_a = _call([_ai_unit()], get_contract_fn=lambda _t: {"a": 1})[0]["fingerprints"]
|
||||
fps_b = _call([_ai_unit()], get_contract_fn=lambda _t: {"a": 2})[0]["fingerprints"]
|
||||
assert fps_a["contract_sha"] != fps_b["contract_sha"]
|
||||
assert fps_a["partial_sha"] == fps_b["partial_sha"]
|
||||
|
||||
|
||||
def test_fingerprints_change_when_partial_changes(monkeypatch):
|
||||
"""Different figma_partial_json → different partial_sha, contract_sha unchanged."""
|
||||
router = MagicMock(return_value=None)
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", router)
|
||||
fps_a = _call(
|
||||
[_ai_unit()], figma_partial_loader=lambda _t: {"p": 1}
|
||||
)[0]["fingerprints"]
|
||||
fps_b = _call(
|
||||
[_ai_unit()], figma_partial_loader=lambda _t: {"p": 2}
|
||||
)[0]["fingerprints"]
|
||||
assert fps_a["partial_sha"] != fps_b["partial_sha"]
|
||||
assert fps_a["contract_sha"] == fps_b["contract_sha"]
|
||||
|
||||
|
||||
def test_v4_result_cardinality_uses_unit_value(monkeypatch):
|
||||
"""v4_result['cardinality'] mirrors the unit's cardinality (no longer hardcoded None)."""
|
||||
router = MagicMock(return_value=None)
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", router)
|
||||
_call([_ai_unit(cardinality=7)])
|
||||
assert router.call_args.kwargs["v4_result"]["cardinality"] == 7
|
||||
router.reset_mock()
|
||||
_call([_ai_unit(cardinality=None)])
|
||||
assert router.call_args.kwargs["v4_result"]["cardinality"] is None
|
||||
|
||||
|
||||
def test_skipped_records_have_no_cache_key_or_fingerprints(monkeypatch):
|
||||
"""Non-AI-eligible records keep cache_key and fingerprints as None."""
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", MagicMock(return_value=None))
|
||||
units = [
|
||||
FakeUnit(label="restructure", provisional=False),
|
||||
FakeUnit(label="reject", provisional=True),
|
||||
FakeUnit(label="light_edit", provisional=True),
|
||||
]
|
||||
recs = _call(units)
|
||||
for rec in recs:
|
||||
assert rec["cache_key"] is None
|
||||
assert rec["fingerprints"] is None
|
||||
|
||||
|
||||
def test_catalog_sha_loader_called_once_per_gather(monkeypatch):
|
||||
"""catalog_sha is computed once per gather call, not per unit."""
|
||||
router = MagicMock(return_value=None)
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", router)
|
||||
loader = MagicMock(return_value="cafefeed" * 8)
|
||||
_call(
|
||||
[_ai_unit(), _ai_unit(frame_id="fid_other"), _ai_unit(frame_id="fid_third")],
|
||||
catalog_sha_loader=loader,
|
||||
)
|
||||
loader.assert_called_once()
|
||||
|
||||
|
||||
def test_record_shape_contract_is_stable_with_u4_fields(monkeypatch):
|
||||
"""Record schema includes the IMP-46 u4 cache_key + fingerprints fields."""
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", MagicMock(return_value=None))
|
||||
units = [FakeUnit(label="reject", provisional=True)]
|
||||
rec = _call(units)[0]
|
||||
assert set(rec.keys()) == {
|
||||
"unit_index",
|
||||
"source_section_ids",
|
||||
"frame_template_id",
|
||||
"label",
|
||||
"route_hint",
|
||||
"provisional",
|
||||
"ai_called",
|
||||
"skip_reason",
|
||||
"proposal",
|
||||
"error",
|
||||
"cache_key",
|
||||
"fingerprints",
|
||||
}
|
||||
|
||||
|
||||
def test_cache_key_is_compatible_with_cache_parse_key(monkeypatch):
|
||||
"""cache_key produced here must round-trip through cache.py's _parse_key."""
|
||||
from src.phase_z2_ai_fallback.cache import KEY_DELIMITER, _parse_key
|
||||
|
||||
router = MagicMock(return_value=None)
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", router)
|
||||
_call([_ai_unit()])
|
||||
cache_key = router.call_args.kwargs["cache_key"]
|
||||
parsed = _parse_key(cache_key)
|
||||
assert parsed is not None
|
||||
frame_id, signature_hash = parsed
|
||||
assert frame_id == "fid_123"
|
||||
assert len(signature_hash) == 64
|
||||
assert KEY_DELIMITER not in signature_hash
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# IMP-47B u9 — Step 12 reject eligibility + normal-path AI=0 regression
|
||||
# ---------------------------------------------------------------------------
|
||||
# Locks the end-to-end Step 12 contract against the production route helper
|
||||
# `_imp05_route_hint`. The local `_ROUTE_HINTS` mapping above intentionally
|
||||
# preserves the legacy ``reject -> design_reference_only`` form to exercise
|
||||
# the catch-all fall-through branch; u9 instead drives gather with the real
|
||||
# production map (post-u1 flip) so reject provisional units reach the router
|
||||
# and normal-path labels stay AI=0.
|
||||
|
||||
|
||||
def test_production_reject_route_reaches_router_when_provisional(monkeypatch):
|
||||
"""Post-u1, provisional reject units must reach ``route_ai_fallback``."""
|
||||
from src.phase_z2_pipeline import _imp05_route_hint
|
||||
|
||||
router = MagicMock(return_value=None)
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", router)
|
||||
records = step12_mod.gather_step12_ai_repair_proposals(
|
||||
[FakeUnit(label="reject", provisional=True)],
|
||||
route_for_label=_imp05_route_hint,
|
||||
get_contract_fn=_get_contract,
|
||||
frame_visual_loader=_frame_visual,
|
||||
)
|
||||
assert records[0]["route_hint"] == "ai_adaptation_required"
|
||||
assert records[0]["skip_reason"] == "router_short_circuit"
|
||||
assert records[0]["ai_called"] is False
|
||||
router.assert_called_once()
|
||||
|
||||
|
||||
def test_production_normal_route_labels_never_reach_router(monkeypatch):
|
||||
"""Normal-path labels stay AI=0 even when the unit is provisional."""
|
||||
from src.phase_z2_pipeline import _imp05_route_hint
|
||||
|
||||
router = MagicMock(return_value=None)
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", router)
|
||||
units = [
|
||||
FakeUnit(label="use_as_is", provisional=True),
|
||||
FakeUnit(label="light_edit", provisional=True),
|
||||
FakeUnit(label=None, provisional=True),
|
||||
]
|
||||
records = step12_mod.gather_step12_ai_repair_proposals(
|
||||
units,
|
||||
route_for_label=_imp05_route_hint,
|
||||
get_contract_fn=_get_contract,
|
||||
frame_visual_loader=_frame_visual,
|
||||
)
|
||||
assert records[0]["skip_reason"] == "route_not_ai_adaptation:direct_render"
|
||||
assert records[1]["skip_reason"] == (
|
||||
"route_not_ai_adaptation:deterministic_minor_adjustment"
|
||||
)
|
||||
assert records[2]["skip_reason"] == "route_not_ai_adaptation:None"
|
||||
router.assert_not_called()
|
||||
|
||||
|
||||
def test_production_non_provisional_reject_skipped_before_route_gate(monkeypatch):
|
||||
"""The provisional gate fires before the route gate (production routing).
|
||||
|
||||
Even with reject routed to ``ai_adaptation_required`` (post-u1), a
|
||||
non-provisional reject unit must short-circuit at ``not_provisional``
|
||||
without ever consulting ``route_for_label`` for an AI dispatch.
|
||||
"""
|
||||
from src.phase_z2_pipeline import _imp05_route_hint
|
||||
|
||||
router = MagicMock(return_value=None)
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", router)
|
||||
records = step12_mod.gather_step12_ai_repair_proposals(
|
||||
[FakeUnit(label="reject", provisional=False)],
|
||||
route_for_label=_imp05_route_hint,
|
||||
get_contract_fn=_get_contract,
|
||||
frame_visual_loader=_frame_visual,
|
||||
)
|
||||
assert records[0]["skip_reason"] == "not_provisional"
|
||||
assert records[0]["ai_called"] is False
|
||||
router.assert_not_called()
|
||||
@@ -0,0 +1,208 @@
|
||||
"""IMP-33 u9 — Step 17 AI repair wiring tests (BLOCKED until IMP-34 + IMP-35).
|
||||
|
||||
Covers:
|
||||
* :data:`OVERFLOW_CASCADE_ORDER` canonical order (4 stages).
|
||||
* :class:`OverflowCascadeStage` member values.
|
||||
* :data:`STEP17_AI_REPAIR_BLOCKED_REASON` constant value.
|
||||
* :func:`gather_step17_ai_repair_proposals` BLOCKED contract — every unit
|
||||
returns ``ai_called=False`` + ``skip_reason=STEP17_AI_REPAIR_BLOCKED_REASON``
|
||||
+ ``proposal=None`` regardless of provisional / label / route_hint.
|
||||
* Structural guarantee — the u9 module does NOT import
|
||||
:func:`src.phase_z2_ai_fallback.router.route_ai_fallback` or the
|
||||
``anthropic`` SDK. Step 17 AI repair stays structurally blocked.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import ast
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
from src.phase_z2_ai_fallback import step17 as step17_mod
|
||||
from src.phase_z2_ai_fallback.step17 import (
|
||||
OVERFLOW_CASCADE_ORDER,
|
||||
STEP17_AI_REPAIR_BLOCKED_REASON,
|
||||
OverflowCascadeStage,
|
||||
gather_step17_ai_repair_proposals,
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class FakeUnit:
|
||||
label: str | None
|
||||
provisional: bool
|
||||
frame_template_id: str = "tmpl"
|
||||
frame_id: str = "fid"
|
||||
source_section_ids: list[str] = field(default_factory=lambda: ["s1"])
|
||||
raw_content: str = "raw"
|
||||
v4_rank: int | None = 1
|
||||
|
||||
|
||||
_ROUTE_HINTS: dict[str | None, str | None] = {
|
||||
"use_as_is": "direct_render",
|
||||
"light_edit": "deterministic_minor_adjustment",
|
||||
"restructure": "ai_adaptation_required",
|
||||
"reject": "design_reference_only",
|
||||
None: None,
|
||||
}
|
||||
|
||||
|
||||
def _route_for_label(label: str | None) -> str | None:
|
||||
return _ROUTE_HINTS.get(label)
|
||||
|
||||
|
||||
# ─── Stage / order constants ─────────────────────────────────────────
|
||||
|
||||
|
||||
def test_overflow_cascade_order_is_canonical():
|
||||
assert OVERFLOW_CASCADE_ORDER == (
|
||||
OverflowCascadeStage.DETERMINISTIC,
|
||||
OverflowCascadeStage.POPUP,
|
||||
OverflowCascadeStage.AI_REPAIR,
|
||||
OverflowCascadeStage.USER_OVERRIDE,
|
||||
)
|
||||
|
||||
|
||||
def test_overflow_cascade_stage_string_values():
|
||||
assert OverflowCascadeStage.DETERMINISTIC.value == "deterministic"
|
||||
assert OverflowCascadeStage.POPUP.value == "popup"
|
||||
assert OverflowCascadeStage.AI_REPAIR.value == "ai_repair"
|
||||
assert OverflowCascadeStage.USER_OVERRIDE.value == "user_override"
|
||||
|
||||
|
||||
def test_step17_blocked_reason_constant_value():
|
||||
assert (
|
||||
STEP17_AI_REPAIR_BLOCKED_REASON
|
||||
== "step17_ai_blocked_imp_34_35_prerequisites_missing"
|
||||
)
|
||||
|
||||
|
||||
# ─── BLOCKED contract: every unit returns blocked record ─────────────
|
||||
|
||||
|
||||
def test_gather_returns_one_record_per_unit():
|
||||
units = [
|
||||
FakeUnit(label="restructure", provisional=True),
|
||||
FakeUnit(label="reject", provisional=False),
|
||||
FakeUnit(label="use_as_is", provisional=True),
|
||||
]
|
||||
records = gather_step17_ai_repair_proposals(units, route_for_label=_route_for_label)
|
||||
assert len(records) == 3
|
||||
|
||||
|
||||
def test_gather_records_blocked_skip_reason():
|
||||
"""Every record must carry the IMP-34/IMP-35 prerequisite block reason."""
|
||||
units = [FakeUnit(label="restructure", provisional=True)]
|
||||
records = gather_step17_ai_repair_proposals(units, route_for_label=_route_for_label)
|
||||
assert records[0]["skip_reason"] == STEP17_AI_REPAIR_BLOCKED_REASON
|
||||
|
||||
|
||||
def test_gather_blocks_even_when_route_is_ai_adaptation_required():
|
||||
"""Provisional + ai_adaptation_required must NOT bypass the u9 block.
|
||||
|
||||
Stage 2 contract: AI repair at Step 17 is blocked behind IMP-34 + IMP-35
|
||||
regardless of V4 route hint. Only u8 (Step 12) is allowed to invoke AI today.
|
||||
"""
|
||||
units = [FakeUnit(label="restructure", provisional=True)]
|
||||
record = gather_step17_ai_repair_proposals(
|
||||
units, route_for_label=_route_for_label
|
||||
)[0]
|
||||
assert record["route_hint"] == "ai_adaptation_required"
|
||||
assert record["ai_called"] is False
|
||||
assert record["proposal"] is None
|
||||
assert record["skip_reason"] == STEP17_AI_REPAIR_BLOCKED_REASON
|
||||
|
||||
|
||||
def test_gather_blocks_reject_units_too():
|
||||
"""Reject units (design_reference_only) are also blocked at u9 — same reason."""
|
||||
units = [FakeUnit(label="reject", provisional=False)]
|
||||
record = gather_step17_ai_repair_proposals(
|
||||
units, route_for_label=_route_for_label
|
||||
)[0]
|
||||
assert record["ai_called"] is False
|
||||
assert record["skip_reason"] == STEP17_AI_REPAIR_BLOCKED_REASON
|
||||
|
||||
|
||||
def test_gather_records_proposal_none_and_no_error():
|
||||
units = [FakeUnit(label="restructure", provisional=True)]
|
||||
record = gather_step17_ai_repair_proposals(
|
||||
units, route_for_label=_route_for_label
|
||||
)[0]
|
||||
assert record["proposal"] is None
|
||||
assert record["error"] is None
|
||||
|
||||
|
||||
def test_gather_records_cascade_stage_is_ai_repair():
|
||||
units = [FakeUnit(label="restructure", provisional=True)]
|
||||
record = gather_step17_ai_repair_proposals(
|
||||
units, route_for_label=_route_for_label
|
||||
)[0]
|
||||
assert record["cascade_stage"] == OverflowCascadeStage.AI_REPAIR.value
|
||||
|
||||
|
||||
def test_gather_preserves_unit_metadata():
|
||||
units = [
|
||||
FakeUnit(
|
||||
label="restructure",
|
||||
provisional=True,
|
||||
frame_template_id="frame_05_overview",
|
||||
source_section_ids=["s1", "s2"],
|
||||
)
|
||||
]
|
||||
record = gather_step17_ai_repair_proposals(
|
||||
units, route_for_label=_route_for_label
|
||||
)[0]
|
||||
assert record["unit_index"] == 0
|
||||
assert record["frame_template_id"] == "frame_05_overview"
|
||||
assert record["source_section_ids"] == ["s1", "s2"]
|
||||
assert record["label"] == "restructure"
|
||||
assert record["provisional"] is True
|
||||
|
||||
|
||||
def test_gather_with_empty_units_returns_empty_list():
|
||||
records = gather_step17_ai_repair_proposals([], route_for_label=_route_for_label)
|
||||
assert records == []
|
||||
|
||||
|
||||
# ─── Structural guarantee: u9 must NOT import route_ai_fallback / anthropic ─
|
||||
|
||||
|
||||
def _u9_imports() -> list[str]:
|
||||
src_path = Path(step17_mod.__file__)
|
||||
tree = ast.parse(src_path.read_text(encoding="utf-8"))
|
||||
imports: list[str] = []
|
||||
for node in ast.walk(tree):
|
||||
if isinstance(node, ast.Import):
|
||||
imports.extend(alias.name for alias in node.names)
|
||||
elif isinstance(node, ast.ImportFrom):
|
||||
module = node.module or ""
|
||||
for alias in node.names:
|
||||
imports.append(f"{module}.{alias.name}")
|
||||
return imports
|
||||
|
||||
|
||||
def test_step17_module_does_not_import_route_ai_fallback():
|
||||
"""u9 must not be able to reach the u7 router — structural block."""
|
||||
imports = _u9_imports()
|
||||
forbidden = {
|
||||
"src.phase_z2_ai_fallback.router.route_ai_fallback",
|
||||
"src.phase_z2_ai_fallback.router",
|
||||
}
|
||||
assert not any(imp in forbidden for imp in imports), imports
|
||||
assert not hasattr(step17_mod, "route_ai_fallback")
|
||||
|
||||
|
||||
def test_step17_module_does_not_import_anthropic():
|
||||
"""u9 must not reach the Anthropic SDK directly — AI=0 in this layer."""
|
||||
imports = _u9_imports()
|
||||
leaked = [imp for imp in imports if imp.split(".", 1)[0] == "anthropic"]
|
||||
assert leaked == [], leaked
|
||||
|
||||
|
||||
def test_step17_module_does_not_import_ai_fallback_client():
|
||||
"""u9 must not instantiate the u4 client either."""
|
||||
imports = _u9_imports()
|
||||
forbidden_prefixes = ("src.phase_z2_ai_fallback.client",)
|
||||
leaked = [
|
||||
imp for imp in imports if imp.startswith(forbidden_prefixes)
|
||||
]
|
||||
assert leaked == [], leaked
|
||||
@@ -0,0 +1,144 @@
|
||||
"""IMP-33 u5 — AI fallback validator tests.
|
||||
|
||||
Scope (Stage 2 plan, u5):
|
||||
- schema re-validation (defence-in-depth)
|
||||
- builder whitelist (BUILDER_OPTIONS_PATCH)
|
||||
- dropped-slot guard (PARTIAL_OVERRIDES / SLOT_MAPPING_PROPOSAL must keep
|
||||
every declared sub_zone slot present)
|
||||
- frame-swap guard (no payload.frame_id mutation; V4 rank-1 protected)
|
||||
- Internal Region containment (payload.region_id must match declared id)
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from src.phase_z2_ai_fallback import AiFallbackProposal, ProposalKind
|
||||
from src.phase_z2_ai_fallback.validate import (
|
||||
AiFallbackValidationError,
|
||||
validate_proposal,
|
||||
)
|
||||
|
||||
|
||||
_FRAME_CONTRACT = {
|
||||
"frame_id": 1171281190,
|
||||
"sub_zones": [
|
||||
{"id": "pillar_1", "accepts": ["text_block"]},
|
||||
{"id": "pillar_2", "accepts": ["text_block"]},
|
||||
{"id": "pillar_3", "accepts": ["text_block"]},
|
||||
],
|
||||
"payload": {
|
||||
"builder_options": {
|
||||
"item_parser": "pillar_item",
|
||||
"array_root": "pillars",
|
||||
"role_field": "color_class",
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
_REGION = {"id": "zone_top.region_a"}
|
||||
|
||||
|
||||
def _make(kind: ProposalKind, payload: dict) -> AiFallbackProposal:
|
||||
return AiFallbackProposal(proposal_kind=kind, payload=payload)
|
||||
|
||||
|
||||
def test_builder_options_patch_accepts_whitelisted_keys() -> None:
|
||||
proposal = _make(
|
||||
ProposalKind.BUILDER_OPTIONS_PATCH,
|
||||
{"item_parser": "alt_pillar_item"},
|
||||
)
|
||||
validate_proposal(proposal, frame_contract=_FRAME_CONTRACT)
|
||||
|
||||
|
||||
def test_builder_options_patch_rejects_unknown_key() -> None:
|
||||
proposal = _make(
|
||||
ProposalKind.BUILDER_OPTIONS_PATCH,
|
||||
{"item_parser": "x", "padding_px": 10},
|
||||
)
|
||||
with pytest.raises(AiFallbackValidationError, match="builder whitelist"):
|
||||
validate_proposal(proposal, frame_contract=_FRAME_CONTRACT)
|
||||
|
||||
|
||||
def test_partial_overrides_requires_all_declared_slots() -> None:
|
||||
proposal = _make(
|
||||
ProposalKind.PARTIAL_OVERRIDES,
|
||||
{"slots": {"pillar_1": "a", "pillar_2": "b"}},
|
||||
)
|
||||
with pytest.raises(AiFallbackValidationError, match="dropped-slot guard"):
|
||||
validate_proposal(proposal, frame_contract=_FRAME_CONTRACT)
|
||||
|
||||
|
||||
def test_partial_overrides_with_all_slots_passes() -> None:
|
||||
proposal = _make(
|
||||
ProposalKind.PARTIAL_OVERRIDES,
|
||||
{"slots": {"pillar_1": "a", "pillar_2": "b", "pillar_3": "c"}},
|
||||
)
|
||||
validate_proposal(proposal, frame_contract=_FRAME_CONTRACT)
|
||||
|
||||
|
||||
def test_slot_mapping_proposal_requires_slots_dict() -> None:
|
||||
proposal = _make(ProposalKind.SLOT_MAPPING_PROPOSAL, {"slots": []})
|
||||
with pytest.raises(AiFallbackValidationError, match="dropped-slot guard"):
|
||||
validate_proposal(proposal, frame_contract=_FRAME_CONTRACT)
|
||||
|
||||
|
||||
def test_frame_swap_guard_rejects_mismatched_frame_id() -> None:
|
||||
proposal = _make(
|
||||
ProposalKind.BUILDER_OPTIONS_PATCH,
|
||||
{"frame_id": 9999, "item_parser": "x"},
|
||||
)
|
||||
with pytest.raises(AiFallbackValidationError, match="frame-swap guard"):
|
||||
validate_proposal(proposal, frame_contract=_FRAME_CONTRACT)
|
||||
|
||||
|
||||
def test_frame_swap_guard_accepts_matching_frame_id() -> None:
|
||||
proposal = _make(
|
||||
ProposalKind.PARTIAL_OVERRIDES,
|
||||
{
|
||||
"frame_id": 1171281190,
|
||||
"slots": {"pillar_1": "a", "pillar_2": "b", "pillar_3": "c"},
|
||||
},
|
||||
)
|
||||
validate_proposal(proposal, frame_contract=_FRAME_CONTRACT)
|
||||
|
||||
|
||||
def test_internal_region_containment_rejects_mismatch() -> None:
|
||||
proposal = _make(
|
||||
ProposalKind.PARTIAL_OVERRIDES,
|
||||
{
|
||||
"slots": {"pillar_1": "a", "pillar_2": "b", "pillar_3": "c"},
|
||||
"region_id": "zone_bottom.region_x",
|
||||
},
|
||||
)
|
||||
with pytest.raises(AiFallbackValidationError, match="Internal Region"):
|
||||
validate_proposal(
|
||||
proposal,
|
||||
frame_contract=_FRAME_CONTRACT,
|
||||
internal_region=_REGION,
|
||||
)
|
||||
|
||||
|
||||
def test_internal_region_containment_accepts_match() -> None:
|
||||
proposal = _make(
|
||||
ProposalKind.PARTIAL_OVERRIDES,
|
||||
{
|
||||
"slots": {"pillar_1": "a", "pillar_2": "b", "pillar_3": "c"},
|
||||
"region_id": "zone_top.region_a",
|
||||
},
|
||||
)
|
||||
validate_proposal(
|
||||
proposal,
|
||||
frame_contract=_FRAME_CONTRACT,
|
||||
internal_region=_REGION,
|
||||
)
|
||||
|
||||
|
||||
def test_internal_region_check_skipped_when_no_region_supplied() -> None:
|
||||
proposal = _make(
|
||||
ProposalKind.PARTIAL_OVERRIDES,
|
||||
{
|
||||
"slots": {"pillar_1": "a", "pillar_2": "b", "pillar_3": "c"},
|
||||
"region_id": "zone_top.region_a",
|
||||
},
|
||||
)
|
||||
validate_proposal(proposal, frame_contract=_FRAME_CONTRACT)
|
||||
@@ -0,0 +1,421 @@
|
||||
"""IMP-27: Shared catalog loader tests (u1).
|
||||
|
||||
Validates that ``src.catalog`` provides a single file-read + mtime cache and
|
||||
that its four public functions honor the documented contracts.
|
||||
|
||||
Note: ``templates/catalog.yaml`` was deleted in cc2f434 (legacy block library
|
||||
cleanup). The shared loader still preserves the loader contract for any
|
||||
remaining Phase Q call sites; tests use a fixture catalog via monkeypatch so
|
||||
they are independent of the deleted production file.
|
||||
|
||||
Delegation tests for block_reference / block_selector / renderer wrappers are
|
||||
added in u2 / u3 / u4.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
import yaml
|
||||
|
||||
|
||||
FIXTURE_CATALOG = {
|
||||
"blocks": [
|
||||
{
|
||||
"id": "fixture-block-a",
|
||||
"category": "emphasis",
|
||||
"template": "blocks/emphasis/fixture-block-a.html",
|
||||
},
|
||||
{
|
||||
"id": "fixture-block-b",
|
||||
"category": "cards",
|
||||
"template": "blocks/cards/fixture-block-b.html",
|
||||
"variants": [
|
||||
{"id": "default", "template": "blocks/cards/fixture-block-b.html"},
|
||||
{"id": "compact", "template": "blocks/cards/fixture-block-b--compact.html"},
|
||||
],
|
||||
},
|
||||
]
|
||||
}
|
||||
|
||||
|
||||
def _reset_catalog_module():
|
||||
import src.catalog as catalog_mod
|
||||
catalog_mod._catalog_cache = None
|
||||
catalog_mod._catalog_mtime = 0.0
|
||||
|
||||
|
||||
def _reset_renderer_projection_cache():
|
||||
"""IMP-27 u4: clear renderer-local projection caches between tests."""
|
||||
import src.renderer as renderer_mod
|
||||
renderer_mod._CATALOG_MAP = None
|
||||
renderer_mod._CATALOG_MAP_MTIME = 0.0
|
||||
renderer_mod._CATALOG_VARIANT_MAP = None
|
||||
renderer_mod._CATALOG_VARIANT_MAP_MTIME = 0.0
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def fixture_catalog_path(tmp_path, monkeypatch):
|
||||
"""Point src.catalog at a tmp catalog.yaml fixture and reset its cache."""
|
||||
import src.catalog as catalog_mod
|
||||
|
||||
fixture_path = tmp_path / "catalog.yaml"
|
||||
fixture_path.write_text(yaml.safe_dump(FIXTURE_CATALOG), encoding="utf-8")
|
||||
monkeypatch.setattr(catalog_mod, "CATALOG_PATH", fixture_path)
|
||||
_reset_catalog_module()
|
||||
_reset_renderer_projection_cache()
|
||||
return fixture_path
|
||||
|
||||
|
||||
def test_load_root_catalog_returns_root_dict(fixture_catalog_path):
|
||||
from src import catalog
|
||||
|
||||
root = catalog.load_root_catalog()
|
||||
assert isinstance(root, dict)
|
||||
assert "blocks" in root
|
||||
assert len(root["blocks"]) == 2
|
||||
|
||||
|
||||
def test_load_blocks_returns_list_of_block_dicts(fixture_catalog_path):
|
||||
from src import catalog
|
||||
|
||||
blocks = catalog.load_blocks()
|
||||
assert isinstance(blocks, list)
|
||||
assert len(blocks) == 2
|
||||
assert all(isinstance(b, dict) for b in blocks)
|
||||
assert all("id" in b for b in blocks)
|
||||
|
||||
|
||||
def test_get_block_by_id_without_catalog_arg(fixture_catalog_path):
|
||||
from src import catalog
|
||||
|
||||
found = catalog.get_block_by_id("fixture-block-a")
|
||||
assert found is not None
|
||||
assert found["id"] == "fixture-block-a"
|
||||
assert found["category"] == "emphasis"
|
||||
|
||||
|
||||
def test_get_block_by_id_with_catalog_arg_preserves_block_selector_contract(fixture_catalog_path):
|
||||
from src import catalog
|
||||
|
||||
root = catalog.load_root_catalog()
|
||||
found = catalog.get_block_by_id("fixture-block-b", catalog=root)
|
||||
assert found is not None
|
||||
assert found["id"] == "fixture-block-b"
|
||||
|
||||
|
||||
def test_get_block_by_id_unknown_returns_none(fixture_catalog_path):
|
||||
from src import catalog
|
||||
|
||||
assert catalog.get_block_by_id("__nonexistent_block_id__") is None
|
||||
|
||||
|
||||
def test_mtime_cache_does_single_file_read(fixture_catalog_path, monkeypatch):
|
||||
import src.catalog as catalog_mod
|
||||
|
||||
read_count = {"n": 0}
|
||||
real_safe_load = catalog_mod.yaml.safe_load
|
||||
|
||||
def counting_safe_load(stream):
|
||||
read_count["n"] += 1
|
||||
return real_safe_load(stream)
|
||||
|
||||
monkeypatch.setattr(catalog_mod.yaml, "safe_load", counting_safe_load)
|
||||
|
||||
catalog_mod.load_root_catalog()
|
||||
catalog_mod.load_root_catalog()
|
||||
catalog_mod.load_blocks()
|
||||
catalog_mod.get_block_by_id("fixture-block-a")
|
||||
|
||||
assert read_count["n"] == 1, (
|
||||
f"Expected exactly one yaml.safe_load call on cold cache, "
|
||||
f"got {read_count['n']}"
|
||||
)
|
||||
|
||||
|
||||
def test_mtime_change_triggers_reload(fixture_catalog_path, monkeypatch):
|
||||
import os
|
||||
import src.catalog as catalog_mod
|
||||
|
||||
read_count = {"n": 0}
|
||||
real_safe_load = catalog_mod.yaml.safe_load
|
||||
|
||||
def counting_safe_load(stream):
|
||||
read_count["n"] += 1
|
||||
return real_safe_load(stream)
|
||||
|
||||
monkeypatch.setattr(catalog_mod.yaml, "safe_load", counting_safe_load)
|
||||
|
||||
catalog_mod.load_root_catalog()
|
||||
assert read_count["n"] == 1
|
||||
|
||||
# Forcibly advance file mtime and verify cache invalidates.
|
||||
original_mtime = fixture_catalog_path.stat().st_mtime
|
||||
os.utime(fixture_catalog_path, (original_mtime + 10, original_mtime + 10))
|
||||
|
||||
catalog_mod.load_root_catalog()
|
||||
assert read_count["n"] == 2
|
||||
|
||||
|
||||
def test_get_catalog_mtime_matches_file_after_load(fixture_catalog_path):
|
||||
from src import catalog
|
||||
|
||||
catalog.load_root_catalog()
|
||||
actual = fixture_catalog_path.stat().st_mtime
|
||||
assert catalog.get_catalog_mtime() == actual
|
||||
|
||||
|
||||
def test_get_catalog_mtime_is_zero_before_first_load(monkeypatch, tmp_path):
|
||||
import src.catalog as catalog_mod
|
||||
monkeypatch.setattr(catalog_mod, "CATALOG_PATH", tmp_path / "unused.yaml")
|
||||
_reset_catalog_module()
|
||||
|
||||
assert catalog_mod.get_catalog_mtime() == 0.0
|
||||
|
||||
|
||||
def test_load_root_catalog_missing_file_returns_empty_blocks(monkeypatch, tmp_path):
|
||||
import src.catalog as catalog_mod
|
||||
|
||||
missing_path = tmp_path / "no_such_catalog.yaml"
|
||||
monkeypatch.setattr(catalog_mod, "CATALOG_PATH", missing_path)
|
||||
_reset_catalog_module()
|
||||
|
||||
root = catalog_mod.load_root_catalog()
|
||||
assert root == {"blocks": []}
|
||||
assert catalog_mod.load_blocks() == []
|
||||
|
||||
|
||||
# ──────────────────────────────────────────────────────────
|
||||
# u2: block_reference delegation tests
|
||||
# ──────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_block_reference_load_catalog_returns_list_of_blocks(fixture_catalog_path):
|
||||
"""block_reference._load_catalog preserves list[dict] contract via delegation."""
|
||||
from src import block_reference
|
||||
|
||||
blocks = block_reference._load_catalog()
|
||||
assert isinstance(blocks, list)
|
||||
assert len(blocks) == 2
|
||||
assert all(isinstance(b, dict) for b in blocks)
|
||||
assert {b["id"] for b in blocks} == {"fixture-block-a", "fixture-block-b"}
|
||||
|
||||
|
||||
def test_block_reference_get_block_by_id_no_arg_signature(fixture_catalog_path):
|
||||
"""block_reference._get_block_by_id preserves no-catalog-argument contract."""
|
||||
from src import block_reference
|
||||
|
||||
found = block_reference._get_block_by_id("fixture-block-a")
|
||||
assert found is not None
|
||||
assert found["id"] == "fixture-block-a"
|
||||
assert found["category"] == "emphasis"
|
||||
|
||||
assert block_reference._get_block_by_id("__nonexistent__") is None
|
||||
|
||||
|
||||
def test_block_reference_shares_cache_with_shared_loader(fixture_catalog_path, monkeypatch):
|
||||
"""block_reference wrappers must hit the shared mtime cache, not a private copy."""
|
||||
import src.catalog as catalog_mod
|
||||
from src import block_reference
|
||||
|
||||
read_count = {"n": 0}
|
||||
real_safe_load = catalog_mod.yaml.safe_load
|
||||
|
||||
def counting_safe_load(stream):
|
||||
read_count["n"] += 1
|
||||
return real_safe_load(stream)
|
||||
|
||||
monkeypatch.setattr(catalog_mod.yaml, "safe_load", counting_safe_load)
|
||||
|
||||
block_reference._load_catalog()
|
||||
block_reference._get_block_by_id("fixture-block-a")
|
||||
catalog_mod.load_root_catalog()
|
||||
|
||||
assert read_count["n"] == 1, (
|
||||
f"block_reference wrappers should share the catalog cache; "
|
||||
f"got {read_count['n']} reads"
|
||||
)
|
||||
|
||||
|
||||
def test_block_reference_has_no_private_catalog_cache(fixture_catalog_path):
|
||||
"""IMP-27 u2 guard: block_reference must not retain a module-level _catalog_cache."""
|
||||
from src import block_reference
|
||||
|
||||
assert not hasattr(block_reference, "_catalog_cache"), (
|
||||
"block_reference._catalog_cache must be removed by u2 — "
|
||||
"loader is delegated to src.catalog"
|
||||
)
|
||||
|
||||
|
||||
# ──────────────────────────────────────────────────────────
|
||||
# u3: block_selector delegation tests
|
||||
# ──────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_block_selector_load_catalog_returns_root_dict(fixture_catalog_path):
|
||||
"""block_selector.load_catalog preserves root-dict contract via delegation."""
|
||||
from src import block_selector
|
||||
|
||||
root = block_selector.load_catalog()
|
||||
assert isinstance(root, dict)
|
||||
assert "blocks" in root
|
||||
assert isinstance(root["blocks"], list)
|
||||
assert {b["id"] for b in root["blocks"]} == {"fixture-block-a", "fixture-block-b"}
|
||||
|
||||
|
||||
def test_block_selector_get_block_by_id_catalog_injected_signature(fixture_catalog_path):
|
||||
"""block_selector._get_block_by_id preserves catalog-injected signature."""
|
||||
from src import block_selector
|
||||
|
||||
catalog_dict = block_selector.load_catalog()
|
||||
found = block_selector._get_block_by_id("fixture-block-b", catalog_dict)
|
||||
assert found is not None
|
||||
assert found["id"] == "fixture-block-b"
|
||||
assert found["category"] == "cards"
|
||||
|
||||
assert block_selector._get_block_by_id("__nonexistent__", catalog_dict) is None
|
||||
|
||||
|
||||
def test_block_selector_shares_cache_with_shared_loader(fixture_catalog_path, monkeypatch):
|
||||
"""block_selector wrappers must hit the shared mtime cache, not a private copy."""
|
||||
import src.catalog as catalog_mod
|
||||
from src import block_selector
|
||||
|
||||
read_count = {"n": 0}
|
||||
real_safe_load = catalog_mod.yaml.safe_load
|
||||
|
||||
def counting_safe_load(stream):
|
||||
read_count["n"] += 1
|
||||
return real_safe_load(stream)
|
||||
|
||||
monkeypatch.setattr(catalog_mod.yaml, "safe_load", counting_safe_load)
|
||||
|
||||
block_selector.load_catalog()
|
||||
block_selector._get_block_by_id("fixture-block-a", block_selector.load_catalog())
|
||||
catalog_mod.load_root_catalog()
|
||||
|
||||
assert read_count["n"] == 1, (
|
||||
f"block_selector wrappers should share the catalog cache; "
|
||||
f"got {read_count['n']} reads"
|
||||
)
|
||||
|
||||
|
||||
def test_block_selector_has_no_private_catalog_cache(fixture_catalog_path):
|
||||
"""IMP-27 u3 guard: block_selector must not retain module-level cache/mtime/CATALOG_PATH."""
|
||||
from src import block_selector
|
||||
|
||||
assert not hasattr(block_selector, "_catalog_cache"), (
|
||||
"block_selector._catalog_cache must be removed by u3 — "
|
||||
"loader is delegated to src.catalog"
|
||||
)
|
||||
assert not hasattr(block_selector, "_catalog_mtime"), (
|
||||
"block_selector._catalog_mtime must be removed by u3"
|
||||
)
|
||||
assert not hasattr(block_selector, "CATALOG_PATH"), (
|
||||
"block_selector.CATALOG_PATH must be removed by u3 — "
|
||||
"path lives in src.catalog only"
|
||||
)
|
||||
|
||||
|
||||
# ──────────────────────────────────────────────────────────
|
||||
# u4: renderer delegation tests
|
||||
# ──────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_renderer_load_catalog_map_returns_id_to_template_dict(fixture_catalog_path):
|
||||
"""renderer._load_catalog_map preserves id → template projection contract."""
|
||||
from src import renderer
|
||||
|
||||
mapping = renderer._load_catalog_map()
|
||||
assert isinstance(mapping, dict)
|
||||
assert mapping["fixture-block-a"] == "blocks/emphasis/fixture-block-a.html"
|
||||
assert mapping["fixture-block-b"] == "blocks/cards/fixture-block-b.html"
|
||||
|
||||
|
||||
def test_renderer_load_catalog_map_with_variants_returns_compound_key_dict(fixture_catalog_path):
|
||||
"""renderer._load_catalog_map_with_variants preserves 'id--variant' → template projection."""
|
||||
from src import renderer
|
||||
|
||||
vmap = renderer._load_catalog_map_with_variants()
|
||||
assert isinstance(vmap, dict)
|
||||
# 'default' variants must be excluded (preserves pre-IMP-27 behavior).
|
||||
assert "fixture-block-b--default" not in vmap
|
||||
# Non-default variants must be present with the compound key.
|
||||
assert vmap["fixture-block-b--compact"] == "blocks/cards/fixture-block-b--compact.html"
|
||||
|
||||
|
||||
def test_renderer_shares_cache_with_shared_loader(fixture_catalog_path, monkeypatch):
|
||||
"""renderer projections must read through src.catalog, never opening the file directly."""
|
||||
import src.catalog as catalog_mod
|
||||
from src import renderer
|
||||
|
||||
read_count = {"n": 0}
|
||||
real_safe_load = catalog_mod.yaml.safe_load
|
||||
|
||||
def counting_safe_load(stream):
|
||||
read_count["n"] += 1
|
||||
return real_safe_load(stream)
|
||||
|
||||
monkeypatch.setattr(catalog_mod.yaml, "safe_load", counting_safe_load)
|
||||
|
||||
renderer._load_catalog_map()
|
||||
renderer._load_catalog_map_with_variants()
|
||||
catalog_mod.load_root_catalog()
|
||||
|
||||
assert read_count["n"] == 1, (
|
||||
f"renderer projections should share the catalog cache; "
|
||||
f"got {read_count['n']} reads"
|
||||
)
|
||||
|
||||
|
||||
def test_renderer_projection_invalidates_when_shared_mtime_changes(
|
||||
fixture_catalog_path, monkeypatch
|
||||
):
|
||||
"""IMP-27 u4 contract: renderer projection cache keyed off src.catalog.get_catalog_mtime."""
|
||||
import os
|
||||
|
||||
import src.catalog as catalog_mod
|
||||
from src import renderer
|
||||
|
||||
first_map = renderer._load_catalog_map()
|
||||
first_id_to_path = dict(first_map)
|
||||
|
||||
# Rewrite the fixture file with a different block id and bump mtime so the
|
||||
# shared cache (and therefore the renderer projection) must rebuild.
|
||||
new_catalog = {
|
||||
"blocks": [
|
||||
{
|
||||
"id": "fixture-block-c",
|
||||
"category": "headers",
|
||||
"template": "blocks/headers/fixture-block-c.html",
|
||||
},
|
||||
]
|
||||
}
|
||||
fixture_catalog_path.write_text(yaml.safe_dump(new_catalog), encoding="utf-8")
|
||||
original_mtime = fixture_catalog_path.stat().st_mtime
|
||||
os.utime(fixture_catalog_path, (original_mtime + 10, original_mtime + 10))
|
||||
# Force src.catalog to drop its in-memory cache so the next call re-reads.
|
||||
catalog_mod._catalog_cache = None
|
||||
catalog_mod._catalog_mtime = 0.0
|
||||
|
||||
second_map = renderer._load_catalog_map()
|
||||
assert "fixture-block-c" in second_map
|
||||
assert "fixture-block-a" not in second_map
|
||||
assert first_id_to_path != second_map
|
||||
|
||||
|
||||
def test_renderer_has_no_private_catalog_path_or_yaml(fixture_catalog_path):
|
||||
"""IMP-27 u4 guard: renderer must not retain its own file-read state."""
|
||||
from src import renderer
|
||||
|
||||
assert not hasattr(renderer, "CATALOG_PATH"), (
|
||||
"renderer.CATALOG_PATH must be removed by u4 — path lives in src.catalog only"
|
||||
)
|
||||
assert not hasattr(renderer, "yaml"), (
|
||||
"renderer.yaml import must be removed by u4 — file-read is delegated to src.catalog"
|
||||
)
|
||||
assert not hasattr(renderer, "_CATALOG_MTIME"), (
|
||||
"renderer._CATALOG_MTIME (legacy single mtime) must be removed by u4 — "
|
||||
"projection caches now key off src.catalog.get_catalog_mtime() via "
|
||||
"_CATALOG_MAP_MTIME / _CATALOG_VARIANT_MAP_MTIME"
|
||||
)
|
||||
@@ -0,0 +1,159 @@
|
||||
"""IMP-38 U2 — dynamic effective max_rank + trace 8-field + 3-tier usable predicate.
|
||||
|
||||
Verify:
|
||||
- max_rank=None (default) → policy applied (usable_count + effective_max_rank 결정)
|
||||
- max_rank=int (caller override) → that value used as-is (backward compat)
|
||||
- trace contains 8 IMP-38 fields + legacy "max_rank" alias
|
||||
- usable_count >= threshold → default_max_rank (mdx03 정상 case)
|
||||
- usable_count < threshold → effective_extended_ceiling (mdx05-2 확장 case)
|
||||
- effective_extended_ceiling = min(configured, len(judgments_full32)) (Codex #2)
|
||||
- IMP-30 allow_provisional byte-identical (chain_exhausted 후 provisional 합성)
|
||||
|
||||
4 round 합의 (#67):
|
||||
- Codex #1: 별 yaml (catalog 오염 방지)
|
||||
- Codex #2: min(configured, len(judgments)) 정정
|
||||
- Codex #3: load_frame_contracts() shape 무변
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _reset_policy_cache():
|
||||
"""Reset module-level _V4_FALLBACK_POLICY_CACHE for test isolation."""
|
||||
import src.phase_z2_mapper as mapper
|
||||
mapper._V4_FALLBACK_POLICY_CACHE = None
|
||||
yield
|
||||
mapper._V4_FALLBACK_POLICY_CACHE = None
|
||||
|
||||
|
||||
def _make_v4_section(judgments: list[dict]) -> dict:
|
||||
"""Helper — V4 fixture with mdx_sections[section_id].judgments_full32."""
|
||||
return {
|
||||
"mdx_sections": {
|
||||
"sec-1": {
|
||||
"judgments_full32": judgments,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
def _judgment(template_id: str, label: str, confidence: float = 0.5, frame_id: int = 0) -> dict:
|
||||
"""Helper — V4 judgment entry shape."""
|
||||
return {
|
||||
"template_id": template_id,
|
||||
"frame_id": frame_id or hash(template_id) % 10000,
|
||||
"frame_number": 0,
|
||||
"confidence": confidence,
|
||||
"label": label,
|
||||
}
|
||||
|
||||
|
||||
# ─── U2 Test: caller override (backward compat) ────────────────────
|
||||
|
||||
|
||||
def test_caller_override_uses_explicit_max_rank():
|
||||
"""max_rank=3 explicit → effective_max_rank=3, policy_applied=caller_override."""
|
||||
from src.phase_z2_pipeline import lookup_v4_match_with_fallback
|
||||
judgments = [_judgment(f"t{i}", "reject") for i in range(5)]
|
||||
v4 = _make_v4_section(judgments)
|
||||
_match, trace = lookup_v4_match_with_fallback(v4, "sec-1", max_rank=3)
|
||||
assert trace["policy_applied"] == "caller_override"
|
||||
assert trace["effective_max_rank"] == 3
|
||||
assert trace["max_rank"] == 3 # legacy alias
|
||||
|
||||
|
||||
def test_caller_override_max_rank_5_used_directly():
|
||||
"""max_rank=5 explicit → effective_max_rank=5 (policy 무시)."""
|
||||
from src.phase_z2_pipeline import lookup_v4_match_with_fallback
|
||||
judgments = [_judgment(f"t{i}", "reject") for i in range(10)]
|
||||
v4 = _make_v4_section(judgments)
|
||||
_match, trace = lookup_v4_match_with_fallback(v4, "sec-1", max_rank=5)
|
||||
assert trace["policy_applied"] == "caller_override"
|
||||
assert trace["effective_max_rank"] == 5
|
||||
|
||||
|
||||
# ─── U2 Test: 8 trace fields presence ──────────────────────────────
|
||||
|
||||
|
||||
def test_trace_contains_8_imp38_fields():
|
||||
"""trace dict must contain all 8 IMP-38 fields + legacy max_rank alias."""
|
||||
from src.phase_z2_pipeline import lookup_v4_match_with_fallback
|
||||
judgments = [_judgment(f"t{i}", "reject") for i in range(3)]
|
||||
v4 = _make_v4_section(judgments)
|
||||
_match, trace = lookup_v4_match_with_fallback(v4, "sec-1")
|
||||
expected = {
|
||||
"requested_max_rank",
|
||||
"default_max_rank",
|
||||
"configured_extended_max_rank",
|
||||
"judgments_count",
|
||||
"effective_extended_ceiling",
|
||||
"effective_max_rank",
|
||||
"usable_count",
|
||||
"policy_applied",
|
||||
"max_rank", # legacy alias
|
||||
}
|
||||
missing = expected - set(trace.keys())
|
||||
assert not missing, f"missing IMP-38 trace fields: {missing}"
|
||||
|
||||
|
||||
# ─── U2 Test: Codex #2 정정 — min(configured, len(judgments_full32)) ──
|
||||
|
||||
|
||||
def test_effective_extended_ceiling_is_min_of_configured_and_judgments_count():
|
||||
"""Codex #2 LOCK — judgments_count < configured 일 때 ceiling = judgments_count."""
|
||||
from src.phase_z2_pipeline import lookup_v4_match_with_fallback
|
||||
# 5 judgments only — configured extended (32) 보다 작음
|
||||
judgments = [_judgment(f"t{i}", "reject") for i in range(5)]
|
||||
v4 = _make_v4_section(judgments)
|
||||
_match, trace = lookup_v4_match_with_fallback(v4, "sec-1")
|
||||
assert trace["judgments_count"] == 5
|
||||
assert trace["effective_extended_ceiling"] == 5 # min(32, 5) = 5
|
||||
|
||||
|
||||
# ─── U2 Test: no_judgments path ──────────────────────────────────
|
||||
|
||||
|
||||
def test_no_judgments_path():
|
||||
"""judgments_count=0 → policy_applied=no_judgments, effective_max_rank=default."""
|
||||
from src.phase_z2_pipeline import lookup_v4_match_with_fallback
|
||||
v4 = _make_v4_section([])
|
||||
_match, trace = lookup_v4_match_with_fallback(v4, "sec-1")
|
||||
assert trace["policy_applied"] == "no_judgments"
|
||||
assert trace["judgments_count"] == 0
|
||||
assert trace["effective_max_rank"] == trace["default_max_rank"]
|
||||
assert trace["fallback_reason"] == "empty_v4_judgments"
|
||||
|
||||
|
||||
# ─── U2 Test: no_v4_section ─────────────────────────────────────
|
||||
|
||||
|
||||
def test_no_v4_section_path():
|
||||
"""unknown section_id → fallback_reason=no_v4_section + trace still has 8 IMP-38 fields."""
|
||||
from src.phase_z2_pipeline import lookup_v4_match_with_fallback
|
||||
v4 = {"mdx_sections": {}}
|
||||
_match, trace = lookup_v4_match_with_fallback(v4, "unknown-sec")
|
||||
assert trace["fallback_reason"] == "no_v4_section"
|
||||
# 8 fields still present even when no section found
|
||||
assert "policy_applied" in trace
|
||||
assert "effective_max_rank" in trace
|
||||
|
||||
|
||||
# ─── U2 Test: chain_exhausted message reflects effective_max_rank ──
|
||||
|
||||
|
||||
def test_chain_exhausted_message_includes_effective_max_rank():
|
||||
"""fallback_reason 메시지가 동적 effective_max_rank 반영 (hardcoded "1_to_3" X)."""
|
||||
from src.phase_z2_pipeline import lookup_v4_match_with_fallback
|
||||
# 3 judgments all reject (catalog 등록 X 가정 — t1/t2/t3 는 catalog 에 없음)
|
||||
judgments = [_judgment(f"unregistered_t{i}", "reject") for i in range(3)]
|
||||
v4 = _make_v4_section(judgments)
|
||||
_match, trace = lookup_v4_match_with_fallback(v4, "sec-1", max_rank=3)
|
||||
# chain exhausted — 메시지 가 effective_max_rank=3 반영
|
||||
if trace["selection_path"] == "chain_exhausted":
|
||||
# first_skip_reason 가 있으면 그게 우선, 없으면 default 메시지
|
||||
assert (
|
||||
trace["fallback_reason"] is not None
|
||||
and ("no_auto_renderable" in trace["fallback_reason"] or "phase_z_status" in trace["fallback_reason"] or "no_contract" in trace["fallback_reason"])
|
||||
)
|
||||
@@ -0,0 +1,165 @@
|
||||
"""Phase Z family-template ↔ frame_contracts.yaml baseline invariant.
|
||||
|
||||
#52 F-2 option (c) lock (2026-05-19) — locks active contracted family count
|
||||
at 11/11 with a WIP allowlist for the 2 untracked WIP family templates
|
||||
documented in `templates/phase_z2/families/_WIP_FILES.md`. Any drift after
|
||||
this point (new family file on disk without a contract entry, or a contract
|
||||
entry pointing to a missing file) fails fast in CI.
|
||||
|
||||
IMP-04b (#42) extends the rule with a second exemption axis: catalog entries
|
||||
flagged `visual_pending: true` are contracted but have no family partial on
|
||||
disk yet (Track A/B VP frames). The invariant becomes:
|
||||
|
||||
contracts == (disk - wip) ∪ vp
|
||||
|
||||
where `vp ∩ disk == ∅` (VP entry with disk file = promotion overdue).
|
||||
|
||||
References:
|
||||
- docs/architecture/INTEGRATION-AUDIT-01-REPORT.md §10.2 F-2
|
||||
- docs/architecture/IMP-18-SVG-GAP-REPORT.md L28/L30/L51
|
||||
- templates/phase_z2/families/_WIP_FILES.md (WIP allowlist source)
|
||||
- Gitea #52 (this reconciliation), #42 (promote/remove gate + VP axis)
|
||||
|
||||
Pattern mirrors `tests/test_catalog_invariant.py` — fail fast with explicit
|
||||
diff message if family ↔ contract surfaces drift.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
import yaml
|
||||
|
||||
PROJECT_ROOT = Path(__file__).parent.parent
|
||||
FAMILIES_DIR = PROJECT_ROOT / "templates" / "phase_z2" / "families"
|
||||
CATALOG_PATH = PROJECT_ROOT / "templates" / "phase_z2" / "catalog" / "frame_contracts.yaml"
|
||||
WIP_DOC_PATH = FAMILIES_DIR / "_WIP_FILES.md"
|
||||
V4_EVIDENCE_PATH = PROJECT_ROOT / "tests" / "matching" / "v4_full32_result.yaml"
|
||||
|
||||
|
||||
def _load_contract_keys() -> set[str]:
|
||||
with CATALOG_PATH.open(encoding="utf-8") as f:
|
||||
catalog = yaml.safe_load(f)
|
||||
return {k for k, v in catalog.items() if isinstance(v, dict)}
|
||||
|
||||
|
||||
def _load_disk_family_stems() -> set[str]:
|
||||
return {p.stem for p in FAMILIES_DIR.glob("*.html")}
|
||||
|
||||
|
||||
def _load_wip_allowlist() -> set[str]:
|
||||
text = WIP_DOC_PATH.read_text(encoding="utf-8")
|
||||
return {m.group(1) for m in re.finditer(r"`([A-Za-z0-9_\-]+)\.html`", text)}
|
||||
|
||||
|
||||
def _load_v4_evidence_template_ids() -> set[str]:
|
||||
"""Unique `template_id` values across all V4 full32 judgments.
|
||||
|
||||
IMP-04b (#42) closure gate derives its 32-frame target from V4 evidence
|
||||
rather than a hardcoded count, so future Track C additions extend both
|
||||
surfaces in lockstep.
|
||||
"""
|
||||
with V4_EVIDENCE_PATH.open(encoding="utf-8") as f:
|
||||
v4 = yaml.safe_load(f)
|
||||
return {
|
||||
j["template_id"]
|
||||
for sec in v4["mdx_sections"].values()
|
||||
for j in sec.get("judgments_full32", [])
|
||||
if "template_id" in j
|
||||
}
|
||||
|
||||
|
||||
def _load_vp_exempt_keys() -> set[str]:
|
||||
"""Catalog entries flagged `visual_pending: true` (IMP-04b / #42).
|
||||
|
||||
VP = contracted but no family partial on disk yet (Track A/B VP frames).
|
||||
Exempt from the disk-family existence check until the partial is authored.
|
||||
"""
|
||||
with CATALOG_PATH.open(encoding="utf-8") as f:
|
||||
catalog = yaml.safe_load(f)
|
||||
return {
|
||||
k for k, v in catalog.items()
|
||||
if isinstance(v, dict) and v.get("visual_pending") is True
|
||||
}
|
||||
|
||||
|
||||
def test_contracts_set_equals_disk_families_minus_wip():
|
||||
"""`frame_contracts.yaml` keys ↔ (disk family stems − WIP) ∪ VP-exempt."""
|
||||
contracts = _load_contract_keys()
|
||||
disk = _load_disk_family_stems()
|
||||
wip = _load_wip_allowlist()
|
||||
vp = _load_vp_exempt_keys()
|
||||
expected = disk - wip
|
||||
missing = expected - contracts
|
||||
extra = (contracts - vp) - expected
|
||||
assert not missing, (
|
||||
f"Family files on disk without frame_contracts.yaml entry "
|
||||
f"(and not in _WIP_FILES.md): {sorted(missing)}. "
|
||||
"Add a contract entry, or list the file in "
|
||||
"templates/phase_z2/families/_WIP_FILES.md as WIP."
|
||||
)
|
||||
assert not extra, (
|
||||
f"frame_contracts.yaml has entries with no matching family file "
|
||||
f"(and not flagged `visual_pending: true`): {sorted(extra)}."
|
||||
)
|
||||
|
||||
|
||||
def test_wip_allowlist_is_disk_only_and_uncontracted():
|
||||
"""WIP allowlist names must exist on disk AND have no contract entry."""
|
||||
contracts = _load_contract_keys()
|
||||
disk = _load_disk_family_stems()
|
||||
wip = _load_wip_allowlist()
|
||||
missing_on_disk = wip - disk
|
||||
leaked_into_contracts = wip & contracts
|
||||
assert not missing_on_disk, (
|
||||
f"_WIP_FILES.md names files not on disk: {sorted(missing_on_disk)}."
|
||||
)
|
||||
assert not leaked_into_contracts, (
|
||||
f"_WIP_FILES.md names files that already have a contract entry: "
|
||||
f"{sorted(leaked_into_contracts)}. Promote via #42 instead — "
|
||||
"WIP allowlist must be disk-only / uncontracted."
|
||||
)
|
||||
|
||||
|
||||
def test_vp_exempt_keys_are_contracted_and_disk_absent():
|
||||
"""VP-exempt keys: must be in catalog (by construction) AND must not have
|
||||
a family partial on disk (VP = pending partial authoring per #42)."""
|
||||
contracts = _load_contract_keys()
|
||||
disk = _load_disk_family_stems()
|
||||
vp = _load_vp_exempt_keys()
|
||||
leaked_outside_contracts = vp - contracts
|
||||
has_disk_partial = vp & disk
|
||||
assert not leaked_outside_contracts, (
|
||||
f"VP-exempt keys not in catalog: {sorted(leaked_outside_contracts)}."
|
||||
)
|
||||
assert not has_disk_partial, (
|
||||
f"VP-exempt entries have a family partial on disk: "
|
||||
f"{sorted(has_disk_partial)}. Drop `visual_pending: true` from the "
|
||||
"catalog entry (authoring complete) instead of carrying the flag."
|
||||
)
|
||||
|
||||
|
||||
def test_imp04b_closure_gate_v4_coverage_and_wip_empty():
|
||||
"""IMP-04b (#42) u24 closure gate: catalog set-equals V4 evidence + WIP==0.
|
||||
|
||||
Locks the 32/32 frame coverage by comparing catalog top-level keys to
|
||||
`tests/matching/v4_full32_result.yaml` unique `template_id` values, and
|
||||
asserts the WIP allowlist is empty (both partials absorbed in u3/u4).
|
||||
"""
|
||||
contracts = _load_contract_keys()
|
||||
wip = _load_wip_allowlist()
|
||||
v4_template_ids = _load_v4_evidence_template_ids()
|
||||
missing = v4_template_ids - contracts
|
||||
extra = contracts - v4_template_ids
|
||||
assert not missing, (
|
||||
f"IMP-04b closure gate: V4 template_ids not in frame_contracts.yaml: "
|
||||
f"{sorted(missing)}."
|
||||
)
|
||||
assert not extra, (
|
||||
f"IMP-04b closure gate: frame_contracts.yaml has entries outside "
|
||||
f"V4 evidence: {sorted(extra)}. Extend V4 evidence first or scope "
|
||||
"the addition under a follow-up IMP."
|
||||
)
|
||||
assert not wip, (
|
||||
f"IMP-04b closure gate: WIP allowlist must be empty: {sorted(wip)}."
|
||||
)
|
||||
@@ -0,0 +1,213 @@
|
||||
"""IMP-47B u13 — Persist validated proposals through ``save_proposal`` after gates.
|
||||
|
||||
Scope (this slice):
|
||||
Verify the new ``_persist_ai_repair_proposals_to_cache`` helper in
|
||||
``src/phase_z2_pipeline.py`` honours the IMP-46 dual-gate truth table
|
||||
on the post-Step-14 cache-save seam. The helper is exercised in
|
||||
isolation (no Selenium, no full pipeline) with synthetic AI repair
|
||||
records that mirror the gather → apply → coverage chain shape
|
||||
produced by IMP-47B u4 / u5 / u7.
|
||||
|
||||
Guardrails proven by this test (IMP-46 + IMP-47B policy bullets):
|
||||
* ``visual_check_passed=False`` always blocks — never bypassable, even
|
||||
when ``auto_cache=True`` (IMP-46 u5 truth table cell).
|
||||
* ``user_approved=False`` AND ``auto_cache=False`` → gate blocked
|
||||
(default pipeline path has no UX approval gate; ``--auto-cache`` is
|
||||
the documented bypass).
|
||||
* ``visual_check_passed=True`` AND ``auto_cache=True`` → proposal
|
||||
persisted on disk under ``data/frame_cache/{frame_id}/{hash}.json``
|
||||
via ``cache.save_proposal``.
|
||||
* Non-applied records (no_proposal / no_zone_match / unsupported /
|
||||
error) → ``cache_save_status='not_applied'`` and NEVER reach
|
||||
``save_proposal`` (no filesystem touch).
|
||||
* Settings axis — ``settings.ai_fallback_auto_cache`` sourced through
|
||||
the helper kwargs, never inlined (hardcoding ban).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pathlib
|
||||
|
||||
import pytest
|
||||
|
||||
from src.phase_z2_ai_fallback import cache as cache_mod
|
||||
from src.phase_z2_ai_fallback.cache import AiFallbackCacheGateError
|
||||
from src.phase_z2_ai_fallback.schema import AiFallbackProposal, ProposalKind
|
||||
from src.phase_z2_pipeline import _persist_ai_repair_proposals_to_cache
|
||||
|
||||
|
||||
def _applied_record(
|
||||
*,
|
||||
cache_key: str = "MOCK_FRAME::deadbeef" + "0" * 56,
|
||||
fingerprints: dict | None = None,
|
||||
slots: dict | None = None,
|
||||
) -> dict:
|
||||
"""Build an IMP-47B u4/u5 shaped record marked ``applied:partial_overrides``."""
|
||||
if fingerprints is None:
|
||||
fingerprints = {"contract_sha": "c1", "partial_sha": "p1", "catalog_sha": "k1"}
|
||||
if slots is None:
|
||||
slots = {"title": "AI repaired", "bullets": ["b1", "b2"]}
|
||||
proposal = AiFallbackProposal(
|
||||
proposal_kind=ProposalKind.PARTIAL_OVERRIDES,
|
||||
payload={"slots": slots},
|
||||
rationale="cache save gate test",
|
||||
)
|
||||
return {
|
||||
"unit_index": 0,
|
||||
"source_section_ids": ["MOCK_S1"],
|
||||
"frame_template_id": "MOCK_FRAME",
|
||||
"label": "reject",
|
||||
"route_hint": "ai_adaptation_required",
|
||||
"provisional": True,
|
||||
"ai_called": True,
|
||||
"skip_reason": None,
|
||||
"proposal": proposal.model_dump(),
|
||||
"error": None,
|
||||
"cache_key": cache_key,
|
||||
"fingerprints": fingerprints,
|
||||
"apply_status": "applied:partial_overrides",
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _isolate_cache_root(tmp_path: pathlib.Path, monkeypatch: pytest.MonkeyPatch):
|
||||
"""Redirect ``cache.CACHE_ROOT`` to a per-test tmp dir so save_proposal
|
||||
writes never touch the real ``data/frame_cache/`` tree."""
|
||||
monkeypatch.setattr(cache_mod, "CACHE_ROOT", tmp_path / "frame_cache")
|
||||
yield tmp_path / "frame_cache"
|
||||
|
||||
|
||||
def test_visual_check_failed_blocks_save_even_with_auto_cache(_isolate_cache_root):
|
||||
"""visual_check_passed=False is never bypassable — auto_cache cannot override."""
|
||||
record = _applied_record()
|
||||
records = [record]
|
||||
_persist_ai_repair_proposals_to_cache(
|
||||
records,
|
||||
visual_check_passed=False,
|
||||
user_approved=True,
|
||||
auto_cache=True,
|
||||
)
|
||||
assert record["cache_save_status"].startswith("gate_blocked:")
|
||||
assert "visual_check_passed=False" in record["cache_save_status"]
|
||||
# No filesystem write occurred.
|
||||
assert not _isolate_cache_root.exists() or not any(_isolate_cache_root.rglob("*.json"))
|
||||
|
||||
|
||||
def test_user_not_approved_and_no_auto_cache_blocks_save(_isolate_cache_root):
|
||||
"""Default pipeline path (user_approved=False, auto_cache=False) → gate blocked."""
|
||||
record = _applied_record()
|
||||
records = [record]
|
||||
_persist_ai_repair_proposals_to_cache(
|
||||
records,
|
||||
visual_check_passed=True,
|
||||
user_approved=False,
|
||||
auto_cache=False,
|
||||
)
|
||||
assert record["cache_save_status"].startswith("gate_blocked:")
|
||||
assert "user_approved=False" in record["cache_save_status"]
|
||||
assert not _isolate_cache_root.exists() or not any(_isolate_cache_root.rglob("*.json"))
|
||||
|
||||
|
||||
def test_visual_passed_and_auto_cache_persists_proposal(_isolate_cache_root):
|
||||
"""Happy path — visual_check_passed=True + auto_cache=True persists JSON."""
|
||||
record = _applied_record()
|
||||
records = [record]
|
||||
_persist_ai_repair_proposals_to_cache(
|
||||
records,
|
||||
visual_check_passed=True,
|
||||
user_approved=False,
|
||||
auto_cache=True,
|
||||
)
|
||||
assert record["cache_save_status"] == "saved"
|
||||
written = list(_isolate_cache_root.rglob("*.json"))
|
||||
assert len(written) == 1
|
||||
# Layout = {CACHE_ROOT}/{frame_id}/{signature_hash}.json.
|
||||
written_path = written[0]
|
||||
assert written_path.parent.name == "MOCK_FRAME"
|
||||
|
||||
|
||||
def test_non_applied_records_are_skipped_without_filesystem_touch(_isolate_cache_root):
|
||||
"""no_proposal / no_zone_match / unsupported_kind / error → never reach save_proposal."""
|
||||
no_proposal_record = {
|
||||
"unit_index": 0,
|
||||
"apply_status": "no_proposal",
|
||||
"proposal": None,
|
||||
"cache_key": None,
|
||||
"fingerprints": None,
|
||||
}
|
||||
no_zone_record = {
|
||||
"unit_index": 1,
|
||||
"apply_status": "no_zone_match",
|
||||
"proposal": {"proposal_kind": "partial_overrides", "payload": {"slots": {}}, "rationale": ""},
|
||||
"cache_key": "MOCK::abc",
|
||||
"fingerprints": {"contract_sha": "c", "partial_sha": "p", "catalog_sha": "k"},
|
||||
}
|
||||
unsupported_record = {
|
||||
"unit_index": 2,
|
||||
"apply_status": "unsupported_kind_for_reject_route:builder_options_patch",
|
||||
"proposal": {"proposal_kind": "builder_options_patch", "payload": {}, "rationale": ""},
|
||||
"cache_key": "MOCK::def",
|
||||
"fingerprints": {"contract_sha": "c", "partial_sha": "p", "catalog_sha": "k"},
|
||||
}
|
||||
error_record = {
|
||||
"unit_index": 3,
|
||||
"apply_status": None,
|
||||
"proposal": None,
|
||||
"cache_key": "MOCK::ghi",
|
||||
"fingerprints": {"contract_sha": "c", "partial_sha": "p", "catalog_sha": "k"},
|
||||
"error": "RuntimeError: boom",
|
||||
}
|
||||
records = [no_proposal_record, no_zone_record, unsupported_record, error_record]
|
||||
_persist_ai_repair_proposals_to_cache(
|
||||
records,
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
auto_cache=True,
|
||||
)
|
||||
for r in records:
|
||||
assert r["cache_save_status"] == "not_applied"
|
||||
# Zero JSON files written because none of the records were applied.
|
||||
assert not _isolate_cache_root.exists() or not any(_isolate_cache_root.rglob("*.json"))
|
||||
|
||||
|
||||
def test_mixed_records_only_persist_applied_ones(_isolate_cache_root):
|
||||
"""Mixed batch — only the ``applied:`` record is persisted."""
|
||||
applied = _applied_record(cache_key="MOCK_FRAME::aaaaaaaa" + "0" * 56)
|
||||
not_applied = {
|
||||
"unit_index": 1,
|
||||
"apply_status": "no_proposal",
|
||||
"proposal": None,
|
||||
"cache_key": None,
|
||||
"fingerprints": None,
|
||||
}
|
||||
records = [applied, not_applied]
|
||||
_persist_ai_repair_proposals_to_cache(
|
||||
records,
|
||||
visual_check_passed=True,
|
||||
user_approved=False,
|
||||
auto_cache=True,
|
||||
)
|
||||
assert applied["cache_save_status"] == "saved"
|
||||
assert not_applied["cache_save_status"] == "not_applied"
|
||||
written = list(_isolate_cache_root.rglob("*.json"))
|
||||
assert len(written) == 1
|
||||
|
||||
|
||||
def test_invalid_proposal_payload_surfaces_without_raising(_isolate_cache_root):
|
||||
"""Malformed ``proposal`` dict → ``cache_save_status='invalid_proposal:...'``,
|
||||
no filesystem write, no exception bubbling into the pipeline runtime."""
|
||||
bad_record = {
|
||||
"unit_index": 0,
|
||||
"apply_status": "applied:partial_overrides",
|
||||
"proposal": {"proposal_kind": "not_a_valid_enum_value", "payload": {}, "rationale": ""},
|
||||
"cache_key": "MOCK::bad",
|
||||
"fingerprints": {"contract_sha": "c", "partial_sha": "p", "catalog_sha": "k"},
|
||||
}
|
||||
records = [bad_record]
|
||||
_persist_ai_repair_proposals_to_cache(
|
||||
records,
|
||||
visual_check_passed=True,
|
||||
user_approved=True,
|
||||
auto_cache=True,
|
||||
)
|
||||
assert bad_record["cache_save_status"].startswith("invalid_proposal:")
|
||||
assert not _isolate_cache_root.exists() or not any(_isolate_cache_root.rglob("*.json"))
|
||||
@@ -0,0 +1,95 @@
|
||||
"""IMP-47B u7 — Post-AI source_section_ids coverage invariant tests.
|
||||
|
||||
Scope (this slice):
|
||||
* Helper ``_check_post_ai_coverage_invariant(units, ai_repair_records)``
|
||||
(src/phase_z2_pipeline.py) compares the pre-AI superset (unit
|
||||
``source_section_ids``) to the post-apply superset present on
|
||||
gather records. Per the AI isolation contract + dropped 절대 룰
|
||||
(``feedback_ai_isolation_contract``), AI repair must not silently
|
||||
drop a section.
|
||||
* The helper returns a structured dict (``pre_ai_section_ids``,
|
||||
``post_ai_section_ids``, ``dropped_section_ids``, ``status``) so u8
|
||||
can surface ``status`` through ``slide_status.ai_repair_status``.
|
||||
|
||||
u8 slide_status surfacing and u10 E2E no-text-loss assertion are out
|
||||
of scope for this unit. The helper is pure (no AI call, no IO) so a
|
||||
synthetic stub-unit / stub-record fixture exercises it directly.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
from src.phase_z2_pipeline import _check_post_ai_coverage_invariant
|
||||
|
||||
|
||||
@dataclass
|
||||
class _StubUnit:
|
||||
source_section_ids: list[str] = field(default_factory=list)
|
||||
|
||||
|
||||
def _record(source_section_ids: list[str]) -> dict:
|
||||
"""Minimal gather-record stub — only the field u7 reads."""
|
||||
return {"source_section_ids": list(source_section_ids)}
|
||||
|
||||
|
||||
# ─── Case 1 : matched coverage → status='ok' ────────────────────────
|
||||
|
||||
|
||||
def test_coverage_invariant_ok_when_records_match_units():
|
||||
"""Records carry every unit's source_section_ids → no drop, status='ok'."""
|
||||
units = [_StubUnit(["MOCK_S1", "MOCK_S2"]), _StubUnit(["MOCK_S3"])]
|
||||
records = [_record(["MOCK_S1", "MOCK_S2"]), _record(["MOCK_S3"])]
|
||||
result = _check_post_ai_coverage_invariant(units, records)
|
||||
assert result["status"] == "ok"
|
||||
assert result["dropped_section_ids"] == []
|
||||
assert result["pre_ai_section_ids"] == ["MOCK_S1", "MOCK_S2", "MOCK_S3"]
|
||||
assert result["post_ai_section_ids"] == ["MOCK_S1", "MOCK_S2", "MOCK_S3"]
|
||||
|
||||
|
||||
# ─── Case 2 : record drops a section → status='violated' ────────────
|
||||
|
||||
|
||||
def test_coverage_invariant_violated_when_record_drops_section():
|
||||
"""If a record loses a unit's section_id (e.g., apply mutation bug),
|
||||
the invariant reports status='violated' + dropped list (dropped 절대 룰).
|
||||
"""
|
||||
units = [_StubUnit(["MOCK_S1", "MOCK_S2"]), _StubUnit(["MOCK_S3"])]
|
||||
records = [_record(["MOCK_S1"]), _record(["MOCK_S3"])] # MOCK_S2 dropped
|
||||
result = _check_post_ai_coverage_invariant(units, records)
|
||||
assert result["status"] == "violated"
|
||||
assert result["dropped_section_ids"] == ["MOCK_S2"]
|
||||
assert "MOCK_S2" in result["pre_ai_section_ids"]
|
||||
assert "MOCK_S2" not in result["post_ai_section_ids"]
|
||||
|
||||
|
||||
# ─── Case 3 : empty inputs → status='ok' (no false positive) ────────
|
||||
|
||||
|
||||
def test_coverage_invariant_ok_on_empty_units_and_records():
|
||||
"""Empty pipeline (no units / no records) is a vacuous pass —
|
||||
avoids false-positive 'violated' on edge-case shapes (no AI work).
|
||||
"""
|
||||
result = _check_post_ai_coverage_invariant([], [])
|
||||
assert result["status"] == "ok"
|
||||
assert result["dropped_section_ids"] == []
|
||||
assert result["pre_ai_section_ids"] == []
|
||||
assert result["post_ai_section_ids"] == []
|
||||
|
||||
|
||||
# ─── Case 4 : multiple drops + dedup ────────────────────────────────
|
||||
|
||||
|
||||
def test_coverage_invariant_lists_all_dropped_sections_sorted_and_deduped():
|
||||
"""Multiple missing sections → dropped_section_ids is sorted + deduped.
|
||||
Duplicate ids across units / records collapse to a set comparison.
|
||||
"""
|
||||
units = [
|
||||
_StubUnit(["MOCK_S3", "MOCK_S1"]),
|
||||
_StubUnit(["MOCK_S2", "MOCK_S1"]), # MOCK_S1 duplicate
|
||||
]
|
||||
records: list[dict] = [] # full drop — every unit section missing
|
||||
result = _check_post_ai_coverage_invariant(units, records)
|
||||
assert result["status"] == "violated"
|
||||
assert result["dropped_section_ids"] == ["MOCK_S1", "MOCK_S2", "MOCK_S3"]
|
||||
assert result["pre_ai_section_ids"] == ["MOCK_S1", "MOCK_S2", "MOCK_S3"]
|
||||
assert result["post_ai_section_ids"] == []
|
||||
@@ -0,0 +1,269 @@
|
||||
"""IMP-47B u10 — End-to-end reject smoke (mocked client + full chain + render).
|
||||
|
||||
Scope (this slice):
|
||||
E2E chain proving the IMP-47B reject route activates, preserves
|
||||
full coverage, and propagates the AI-repaired ``slot_payload``
|
||||
into the rendered ``final.html`` artifact when the AI fallback
|
||||
client returns a deterministic PARTIAL_OVERRIDES proposal. Wires
|
||||
together the four pipeline helpers introduced by u4 / u5 / u7 / u8
|
||||
plus the Step 13 render step:
|
||||
|
||||
gather → apply → coverage_invariant → ai_repair_status surfacing
|
||||
→ render_slide → final.html
|
||||
|
||||
The chain mirrors the ``run_phase_z2_mvp1`` call sequence between
|
||||
the Step 12 slot_payload write and the Step 20 ``slide_status``
|
||||
attach (src/phase_z2_pipeline.py — u4 call site, u5 apply, u6
|
||||
artifact, u7 invariant, u8 surface). The Step 13 render path
|
||||
(``render_slide`` at src/phase_z2_pipeline.py:2319, called from the
|
||||
production write site at src/phase_z2_pipeline.py:5107-5111)
|
||||
consumes ``zones_data[i]["slot_payload"]`` verbatim, so this test
|
||||
drives that exact production seam: it calls ``render_slide`` on
|
||||
the post-apply ``zones_data`` and writes the resulting HTML to a
|
||||
``final.html`` file inside ``tmp_path``, then asserts the AI
|
||||
proposal text appears in the on-disk artifact. A heavy
|
||||
``run_phase_z2_mvp1`` integration variant with Selenium overflow
|
||||
check remains deferred — this smoke test stops at the rendered
|
||||
HTML.
|
||||
|
||||
Guardrails proven by this test (IMP-47B policy bullets):
|
||||
* AI 호출 = fallback path only → master flag default OFF preserved
|
||||
(test enables for itself only, restores after).
|
||||
* MDX 원문 100% 보존 → coverage_invariant.status == "ok",
|
||||
source_section_ids identical before/after AI.
|
||||
* 자동 frame swap 금지 → frame_template_id unchanged.
|
||||
* frame visual 임의 변경 금지 → frame_contract / partial untouched
|
||||
(apply only merges proposal.payload.slots into slot_payload).
|
||||
* dropped 절대 룰 → slot_payload AI keys merged on top
|
||||
of deterministic keys; pre-existing meta keys survive.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
from src.phase_z2_ai_fallback.schema import AiFallbackProposal, ProposalKind
|
||||
from src.phase_z2_pipeline import (
|
||||
_apply_ai_repair_proposals_to_zones,
|
||||
_check_post_ai_coverage_invariant,
|
||||
_run_step12_ai_repair,
|
||||
_summarize_ai_repair_status,
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class _StubUnit:
|
||||
"""Synthetic CompositionUnit stand-in (subset of fields gather reads)."""
|
||||
label: str | None = "reject"
|
||||
provisional: bool = True
|
||||
frame_template_id: str = "MOCK_T_reject"
|
||||
frame_id: str = "MOCK_F_reject"
|
||||
source_section_ids: list[str] = field(default_factory=lambda: ["MOCK_S1"])
|
||||
raw_content: str = "MOCK MDX paragraph that must survive AI repair."
|
||||
v4_rank: int | None = 1
|
||||
cardinality: int | None = None
|
||||
layout_preset: str = "two_zone_vertical"
|
||||
zone_position: str = "top"
|
||||
source_shape: str = "paragraph"
|
||||
h3_count: int = 0
|
||||
char_count: int = 48
|
||||
|
||||
|
||||
def _patched_route_ai_fallback(**kwargs):
|
||||
"""Deterministic stand-in for ``route_ai_fallback`` — returns a
|
||||
PARTIAL_OVERRIDES proposal that mirrors the declared frame slots.
|
||||
The validator (src/phase_z2_ai_fallback/validate.py:61-74) is not
|
||||
re-invoked here because this helper bypasses the router; the
|
||||
structural slot completeness is asserted by the apply step + the
|
||||
coverage invariant downstream.
|
||||
"""
|
||||
return AiFallbackProposal(
|
||||
proposal_kind=ProposalKind.PARTIAL_OVERRIDES,
|
||||
payload={
|
||||
"slots": {
|
||||
"title": "AI repaired title",
|
||||
"bullets": ["AI repaired bullet 1", "AI repaired bullet 2"],
|
||||
}
|
||||
},
|
||||
rationale="E2E smoke proposal — deterministic.",
|
||||
)
|
||||
|
||||
|
||||
def test_e2e_reject_chain_applies_proposal_and_preserves_coverage(monkeypatch):
|
||||
"""End-to-end reject smoke (synthetic chain, mocked client).
|
||||
|
||||
Drives the four IMP-47B u4/u5/u7/u8 helpers in pipeline order with
|
||||
a single reject+provisional unit. Asserts every guardrail listed
|
||||
in the module docstring + the four E2E invariants
|
||||
(final.html-bound slot_payload / full coverage / no text loss /
|
||||
human_review NOT required on the success path).
|
||||
"""
|
||||
# IMP-47B u4 wiring — patch the router seam in src/phase_z2_ai_fallback/step12.py
|
||||
# so the gather call returns a deterministic PARTIAL_OVERRIDES proposal
|
||||
# without touching the master flag / network / cache layers.
|
||||
import src.phase_z2_ai_fallback.step12 as step12_mod
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", _patched_route_ai_fallback)
|
||||
|
||||
unit = _StubUnit()
|
||||
units = [unit]
|
||||
|
||||
# Step 12 gather (u4) — eligible reject reaches the patched router.
|
||||
records = _run_step12_ai_repair(units)
|
||||
assert len(records) == 1
|
||||
assert records[0]["route_hint"] == "ai_adaptation_required"
|
||||
assert records[0]["ai_called"] is True
|
||||
assert records[0]["skip_reason"] is None
|
||||
assert records[0]["proposal"]["proposal_kind"] == "partial_overrides"
|
||||
assert records[0]["source_section_ids"] == ["MOCK_S1"]
|
||||
|
||||
# Step 12 apply (u5) — PARTIAL_OVERRIDES merged into the matching zone.
|
||||
# zones_data[0]["slot_payload"] is exactly what render_slide consumes
|
||||
# to emit final.html (src/phase_z2_pipeline.py:5107) — asserting it
|
||||
# here proves the reject route now flows into the rendered HTML.
|
||||
zones = [{
|
||||
"position": "top",
|
||||
"template_id": "MOCK_T_reject",
|
||||
"slot_payload": {
|
||||
"title": "deterministic title",
|
||||
"bullets": ["deterministic bullet"],
|
||||
"_truncated_count": 0,
|
||||
},
|
||||
}]
|
||||
_apply_ai_repair_proposals_to_zones(records, ["top"], zones)
|
||||
assert records[0]["apply_status"] == "applied:partial_overrides"
|
||||
# final.html-bound slot_payload carries AI proposal values
|
||||
assert zones[0]["slot_payload"]["title"] == "AI repaired title"
|
||||
assert zones[0]["slot_payload"]["bullets"] == [
|
||||
"AI repaired bullet 1",
|
||||
"AI repaired bullet 2",
|
||||
]
|
||||
# frame visual / pre-existing meta keys survive (no silent shrink).
|
||||
assert zones[0]["template_id"] == "MOCK_T_reject"
|
||||
assert zones[0]["slot_payload"]["_truncated_count"] == 0
|
||||
# frame_template_id on the unit is byte-identical (no auto frame swap).
|
||||
assert unit.frame_template_id == "MOCK_T_reject"
|
||||
|
||||
# Step 12 coverage invariant (u7) — full coverage, no text loss.
|
||||
coverage = _check_post_ai_coverage_invariant(units, records)
|
||||
assert coverage["status"] == "ok"
|
||||
assert coverage["pre_ai_section_ids"] == ["MOCK_S1"]
|
||||
assert coverage["post_ai_section_ids"] == ["MOCK_S1"]
|
||||
assert coverage["dropped_section_ids"] == []
|
||||
|
||||
# Step 20 ai_repair_status surfacing (u8) — applied without human review.
|
||||
status = _summarize_ai_repair_status(records, coverage)
|
||||
assert status["status"] == "applied"
|
||||
assert status["counts"]["applied"] == 1
|
||||
assert status["counts"]["error"] == 0
|
||||
assert status["counts"]["unsupported_kind"] == 0
|
||||
assert status["coverage_status"] == "ok"
|
||||
assert status.get("human_review_required") is not True
|
||||
|
||||
|
||||
def test_e2e_reject_chain_writes_final_html_with_ai_repaired_slot(monkeypatch, tmp_path):
|
||||
"""End-to-end reject smoke (real render path → final.html on disk).
|
||||
|
||||
Drives the full Stage-2 u10 chain INCLUDING ``render_slide``: the
|
||||
AI-repaired ``slot_payload`` is fed through the same Jinja2
|
||||
rendering seam the production pipeline uses
|
||||
(src/phase_z2_pipeline.py:5107-5111), the resulting HTML is
|
||||
written to ``tmp_path / "final.html"``, and the on-disk artifact
|
||||
is then asserted to carry the AI proposal value. Uses
|
||||
``bim_dx_comparison_table`` — a real registered frame partial
|
||||
(templates/phase_z2/families/bim_dx_comparison_table.html) whose
|
||||
template emits ``{{ slot_payload.title }}`` verbatim, so a
|
||||
proposal-overridden title surfaces literally in the HTML output.
|
||||
"""
|
||||
import src.phase_z2_ai_fallback.step12 as step12_mod
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", _patched_route_ai_fallback)
|
||||
from src.phase_z2_pipeline import build_layout_css, render_slide
|
||||
|
||||
unit = _StubUnit(
|
||||
frame_template_id="bim_dx_comparison_table",
|
||||
zone_position="primary",
|
||||
layout_preset="single",
|
||||
)
|
||||
|
||||
# Step 12 gather + apply. Deterministic non-overridden slots
|
||||
# (col_a_label, col_b_label, rows[*]) are seeded BEFORE apply so the
|
||||
# post-render assertions below can prove u5 merge semantics
|
||||
# (dict.update — not dict-replace) survive the render seam. The
|
||||
# router proposal only carries ``{title, bullets}`` — every other
|
||||
# slot must reach final.html untouched.
|
||||
records = _run_step12_ai_repair([unit])
|
||||
zones = [{
|
||||
"position": "primary",
|
||||
"template_id": "bim_dx_comparison_table",
|
||||
"slot_payload": {
|
||||
"title": "deterministic frame title",
|
||||
"col_a_label": "DETERMINISTIC_COL_A_LABEL",
|
||||
"col_b_label": "DETERMINISTIC_COL_B_LABEL",
|
||||
"rows": [
|
||||
{"label": "DET_ROW_LABEL", "col_a": "DET_ROW_A", "col_b": "DET_ROW_B"},
|
||||
],
|
||||
},
|
||||
}]
|
||||
_apply_ai_repair_proposals_to_zones(records, ["primary"], zones)
|
||||
assert records[0]["apply_status"] == "applied:partial_overrides"
|
||||
|
||||
# Step 13 render — production seam (src/phase_z2_pipeline.py:5107-5111).
|
||||
layout_css = build_layout_css("single", zones)
|
||||
html = render_slide("IMP-47B E2E reject smoke", None, zones, "single", layout_css)
|
||||
final_html_path = tmp_path / "final.html"
|
||||
final_html_path.write_text(html, encoding="utf-8")
|
||||
|
||||
# final.html artifact exists on disk and is non-empty.
|
||||
assert final_html_path.is_file()
|
||||
assert final_html_path.stat().st_size > 0
|
||||
rendered = final_html_path.read_text(encoding="utf-8")
|
||||
|
||||
# AI-repaired slot content appears in the rendered HTML.
|
||||
assert "AI repaired title" in rendered
|
||||
# Deterministic pre-apply title was overridden in the HTML output
|
||||
# (no silent merge that leaves both values visible).
|
||||
assert "deterministic frame title" not in rendered
|
||||
# Non-overridden deterministic slots survive merge → render (u5
|
||||
# dict.update semantics, not dict-replace; dropped 절대 룰 honoured
|
||||
# at the render seam, not just in slot_payload memory).
|
||||
assert "DETERMINISTIC_COL_A_LABEL" in rendered
|
||||
assert "DETERMINISTIC_COL_B_LABEL" in rendered
|
||||
assert "DET_ROW_LABEL" in rendered
|
||||
assert "DET_ROW_A" in rendered
|
||||
assert "DET_ROW_B" in rendered
|
||||
# Frame template id is preserved end-to-end (no auto frame swap).
|
||||
assert 'data-template-id="bim_dx_comparison_table"' in rendered
|
||||
assert unit.frame_template_id == "bim_dx_comparison_table"
|
||||
|
||||
# MDX 원문 100% 보존 — coverage invariant + status surfacing.
|
||||
coverage = _check_post_ai_coverage_invariant([unit], records)
|
||||
assert coverage["status"] == "ok"
|
||||
assert coverage["dropped_section_ids"] == []
|
||||
status = _summarize_ai_repair_status(records, coverage)
|
||||
assert status["status"] == "applied"
|
||||
assert status.get("human_review_required") is not True
|
||||
|
||||
|
||||
def test_e2e_reject_chain_no_text_loss_on_multi_section_unit(monkeypatch):
|
||||
"""Multi-section reject unit — every section id flows through gather,
|
||||
apply, coverage invariant, and ai_repair_status surfacing without a
|
||||
drop. Locks the 'MDX 원문 100% 보존' guardrail at unit-multiplicity
|
||||
granularity (gather copies the list via ``list(...)`` at
|
||||
src/phase_z2_ai_fallback/step12.py:124 so apply mutations cannot
|
||||
silently drop it)."""
|
||||
import src.phase_z2_ai_fallback.step12 as step12_mod
|
||||
monkeypatch.setattr(step12_mod, "route_ai_fallback", _patched_route_ai_fallback)
|
||||
|
||||
unit = _StubUnit(source_section_ids=["MOCK_S1", "MOCK_S2", "MOCK_S3"])
|
||||
records = _run_step12_ai_repair([unit])
|
||||
zones = [{
|
||||
"position": "top",
|
||||
"template_id": "MOCK_T_reject",
|
||||
"slot_payload": {"title": "det", "bullets": ["det"]},
|
||||
}]
|
||||
_apply_ai_repair_proposals_to_zones(records, ["top"], zones)
|
||||
coverage = _check_post_ai_coverage_invariant([unit], records)
|
||||
assert coverage["pre_ai_section_ids"] == ["MOCK_S1", "MOCK_S2", "MOCK_S3"]
|
||||
assert coverage["post_ai_section_ids"] == ["MOCK_S1", "MOCK_S2", "MOCK_S3"]
|
||||
assert coverage["dropped_section_ids"] == []
|
||||
status = _summarize_ai_repair_status(records, coverage)
|
||||
assert status["status"] == "applied"
|
||||
assert status.get("human_review_required") is not True
|
||||
@@ -0,0 +1,174 @@
|
||||
"""IMP-47B u8 — slide_status.ai_repair_status surfacing tests.
|
||||
|
||||
Scope (this slice):
|
||||
Helper ``_summarize_ai_repair_status(ai_repair_records, coverage_invariant)``
|
||||
(src/phase_z2_pipeline.py) composes u4 gather ``error`` + u5
|
||||
``apply_status`` + u7 ``coverage_invariant`` into a single
|
||||
``ai_repair_status`` axis attached to ``slide_status``. Failure-axis
|
||||
priority (highest → lowest): ``error`` > ``coverage_violated`` >
|
||||
``unsupported_kind`` > ``applied`` > ``ok``. ``human_review_required``
|
||||
flips True on the three failure axes for u11 frontend surfacing.
|
||||
|
||||
The frontend reads ``slide_status.ai_repair_status`` to render a
|
||||
notification per the IMP-47B policy ("AI 호출 실패 / proposal validation
|
||||
실패 / coverage 미달 → frontend notification"). u9~u13 are out of scope.
|
||||
The helper is pure (no IO, no AI call) so synthetic record / invariant
|
||||
dicts exercise every branch directly.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from src.phase_z2_pipeline import _summarize_ai_repair_status
|
||||
|
||||
|
||||
def _record(
|
||||
*,
|
||||
unit_index: int = 0,
|
||||
apply_status: str | None = None,
|
||||
error: str | None = None,
|
||||
source_section_ids: list[str] | None = None,
|
||||
) -> dict:
|
||||
"""Minimal Step 12 AI repair record stub — fields u8 reads."""
|
||||
return {
|
||||
"unit_index": unit_index,
|
||||
"source_section_ids": source_section_ids or [f"MOCK_S{unit_index}"],
|
||||
"apply_status": apply_status,
|
||||
"error": error,
|
||||
}
|
||||
|
||||
|
||||
_OK_COVERAGE = {"status": "ok", "dropped_section_ids": []}
|
||||
_VIOLATED_COVERAGE = {"status": "violated", "dropped_section_ids": ["MOCK_S2"]}
|
||||
|
||||
|
||||
# ─── Case 1 : empty pipeline → status='ok' ──────────────────────────
|
||||
|
||||
|
||||
def test_empty_records_returns_ok_no_human_review():
|
||||
"""No AI work executed → status='ok', human_review_required=False.
|
||||
The flag-off default (no provisional units) lands here."""
|
||||
result = _summarize_ai_repair_status([], _OK_COVERAGE)
|
||||
assert result["status"] == "ok"
|
||||
assert result["human_review_required"] is False
|
||||
assert result["counts"]["total"] == 0
|
||||
assert result["unsupported_kind_records"] == []
|
||||
assert result["error_records"] == []
|
||||
assert result["dropped_section_ids"] == []
|
||||
|
||||
|
||||
# ─── Case 2 : applied → status='applied', no human_review ───────────
|
||||
|
||||
|
||||
def test_applied_partial_overrides_marks_applied_no_human_review():
|
||||
"""Successful AI repair (PARTIAL_OVERRIDES applied) is the happy
|
||||
path. status='applied', no human_review surfacing."""
|
||||
records = [_record(apply_status="applied:partial_overrides")]
|
||||
result = _summarize_ai_repair_status(records, _OK_COVERAGE)
|
||||
assert result["status"] == "applied"
|
||||
assert result["human_review_required"] is False
|
||||
assert result["counts"]["applied"] == 1
|
||||
assert result["counts"]["error"] == 0
|
||||
|
||||
|
||||
# ─── Case 3 : unsupported kind → status='unsupported_kind' ──────────
|
||||
|
||||
|
||||
def test_unsupported_kind_marks_human_review_required():
|
||||
"""u5 surfaces ``unsupported_kind_for_reject_route:<kind>`` for
|
||||
builder_options_patch / slot_mapping_proposal. u8 must classify as
|
||||
human_review_required so the frontend renders a notification."""
|
||||
records = [
|
||||
_record(
|
||||
unit_index=1,
|
||||
apply_status="unsupported_kind_for_reject_route:builder_options_patch",
|
||||
source_section_ids=["MOCK_S1"],
|
||||
),
|
||||
]
|
||||
result = _summarize_ai_repair_status(records, _OK_COVERAGE)
|
||||
assert result["status"] == "unsupported_kind"
|
||||
assert result["human_review_required"] is True
|
||||
assert result["counts"]["unsupported_kind"] == 1
|
||||
assert result["unsupported_kind_records"] == [
|
||||
{
|
||||
"unit_index": 1,
|
||||
"source_section_ids": ["MOCK_S1"],
|
||||
"apply_status": "unsupported_kind_for_reject_route:builder_options_patch",
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
# ─── Case 4 : gather error → status='error' (highest priority) ──────
|
||||
|
||||
|
||||
def test_gather_error_marks_status_error_with_records():
|
||||
"""``record['error']`` set means ``gather_step12_ai_repair_proposals``
|
||||
caught a router exception (AI call / validator). status='error'
|
||||
is the highest-priority failure axis."""
|
||||
records = [_record(
|
||||
unit_index=2,
|
||||
error="ValueError: missing slot 'title'",
|
||||
source_section_ids=["MOCK_S2"],
|
||||
)]
|
||||
result = _summarize_ai_repair_status(records, _OK_COVERAGE)
|
||||
assert result["status"] == "error"
|
||||
assert result["human_review_required"] is True
|
||||
assert result["counts"]["error"] == 1
|
||||
assert result["error_records"] == [
|
||||
{
|
||||
"unit_index": 2,
|
||||
"source_section_ids": ["MOCK_S2"],
|
||||
"error": "ValueError: missing slot 'title'",
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
# ─── Case 5 : coverage violated → status='coverage_violated' ────────
|
||||
|
||||
|
||||
def test_coverage_violation_surfaces_dropped_sections():
|
||||
"""u7 coverage_invariant 'violated' means the AI repair dropped a
|
||||
section_id from the post-AI superset. dropped 절대 룰 — surface as
|
||||
human_review_required."""
|
||||
records = [_record(apply_status="applied:partial_overrides")]
|
||||
result = _summarize_ai_repair_status(records, _VIOLATED_COVERAGE)
|
||||
assert result["status"] == "coverage_violated"
|
||||
assert result["human_review_required"] is True
|
||||
assert result["coverage_status"] == "violated"
|
||||
assert result["dropped_section_ids"] == ["MOCK_S2"]
|
||||
|
||||
|
||||
# ─── Case 6 : priority order — error > coverage > unsupported ───────
|
||||
|
||||
|
||||
def test_error_dominates_over_coverage_and_unsupported():
|
||||
"""When multiple failure axes coexist, priority order is
|
||||
error > coverage_violated > unsupported_kind > applied > ok."""
|
||||
records = [
|
||||
_record(unit_index=0, error="RuntimeError"),
|
||||
_record(unit_index=1,
|
||||
apply_status="unsupported_kind_for_reject_route:slot_mapping_proposal"),
|
||||
_record(unit_index=2, apply_status="applied:partial_overrides"),
|
||||
]
|
||||
result = _summarize_ai_repair_status(records, _VIOLATED_COVERAGE)
|
||||
assert result["status"] == "error"
|
||||
assert result["human_review_required"] is True
|
||||
assert result["counts"]["error"] == 1
|
||||
assert result["counts"]["unsupported_kind"] == 1
|
||||
assert result["counts"]["applied"] == 1
|
||||
|
||||
|
||||
# ─── Case 7 : no_proposal + no_zone_match counted, not failure ──────
|
||||
|
||||
|
||||
def test_no_proposal_and_no_zone_match_do_not_trigger_human_review():
|
||||
"""Flag-off short-circuit, not_provisional, route_not_ai_adaptation,
|
||||
and B4-mismatch (no_zone_match) are structural skips — not AI
|
||||
failures. They count but do not flip human_review_required."""
|
||||
records = [
|
||||
_record(unit_index=0, apply_status="no_proposal"),
|
||||
_record(unit_index=1, apply_status="no_zone_match"),
|
||||
]
|
||||
result = _summarize_ai_repair_status(records, _OK_COVERAGE)
|
||||
assert result["status"] == "ok"
|
||||
assert result["human_review_required"] is False
|
||||
assert result["counts"]["no_proposal"] == 1
|
||||
assert result["counts"]["no_zone_match"] == 1
|
||||
@@ -0,0 +1,304 @@
|
||||
"""IMP-47B u12 — Initial plan_composition allow_provisional_fill for mixed direct+reject.
|
||||
|
||||
Scope (this slice):
|
||||
The u12 glue inserted in ``run_phase_z2_mvp1`` (src/phase_z2_pipeline.py,
|
||||
right after the initial plan_composition + telemetry build, before the
|
||||
Step 7-A layout override block) detects the mixed direct+reject case
|
||||
(initial plan_composition returns a viable layout but some sections
|
||||
remain uncovered) and re-runs plan_composition with:
|
||||
|
||||
* a lookup_fn that passes ``allow_provisional=True`` (so chain_exhausted
|
||||
sections synthesize a provisional rank-1 V4Match), and
|
||||
* ``allow_provisional_fill=True`` (so uncovered sections receive a
|
||||
last-resort provisional candidate fill in select_composition_units).
|
||||
|
||||
This admits the mixed direct+reject case to the AI repair path
|
||||
(IMP-47B u4/u5) on first render — the reject section becomes a
|
||||
provisional unit (``provisional=True`` + ``label="reject"``) which Step
|
||||
12's reject route gather (u4) routes to AI fallback.
|
||||
|
||||
Gate predicates (mirrored from src/phase_z2_pipeline.py u12 block):
|
||||
* units non-empty (all-reject case is handled by IMP-30 u4 retry below)
|
||||
* layout_preset is not None
|
||||
* not override_section_assignments (operator override bypasses the gate)
|
||||
* at least one section_id is uncovered after initial pass
|
||||
|
||||
Guardrails proven by these tests:
|
||||
* MDX 원문 100% 보존 — every section_id covered after mixed admission
|
||||
(no silent drop).
|
||||
* 자동 frame swap 금지 — mixed admission only re-runs plan_composition
|
||||
with provisional flags; rank-1 reject judgment is preserved as the
|
||||
provisional V4Match (no template_id swap to a different rank).
|
||||
* Normal-path AI=0 — the mixed admission still emits the reject label;
|
||||
AI activation is gated separately in router (config.py:19 default OFF).
|
||||
* All-direct slides are a no-op — gate skips when no uncovered sections.
|
||||
|
||||
This test file exercises ``plan_composition`` directly with synthetic
|
||||
stub V4 matches + a stub lookup_fn that mirrors the u12 retry seam.
|
||||
Stub naming follows the IMP-30 u3 convention (MOCK_ prefix mandatory,
|
||||
no real catalog template_id / frame_id leakage).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Optional
|
||||
|
||||
from src.phase_z2_composition import plan_composition
|
||||
|
||||
|
||||
# ─── Synthetic V4Match duck-type (mirrors IMP-30 _StubV4Match) ───────────
|
||||
|
||||
@dataclass
|
||||
class _StubV4Match:
|
||||
template_id: str
|
||||
frame_id: str
|
||||
frame_number: int
|
||||
confidence: float
|
||||
label: str
|
||||
v4_rank: Optional[int] = None
|
||||
selection_path: str = "rank_1"
|
||||
fallback_reason: Optional[str] = None
|
||||
provisional: bool = False
|
||||
|
||||
|
||||
@dataclass
|
||||
class _StubSection:
|
||||
section_id: str
|
||||
title: str = ""
|
||||
raw_content: str = ""
|
||||
|
||||
|
||||
_LABEL_TO_STATUS = {
|
||||
"use_as_is": "matched_zone",
|
||||
"light_edit": "adapt_matched_zone",
|
||||
"restructure": "extract_matched_zone",
|
||||
"reject": "fallback_candidate",
|
||||
}
|
||||
|
||||
_ALLOWED_STATUSES = {"matched_zone", "adapt_matched_zone"}
|
||||
|
||||
|
||||
def _make_normal_lookup(matches_by_section: dict[str, _StubV4Match]):
|
||||
"""Lookup_fn that returns the synthetic rank-1 match (no provisional path).
|
||||
|
||||
Mirrors the pipeline initial ``lookup_fn`` at
|
||||
src/phase_z2_pipeline.py:3456-3465 (no ``allow_provisional`` kwarg).
|
||||
"""
|
||||
def _fn(section_id: str):
|
||||
return matches_by_section.get(section_id)
|
||||
return _fn
|
||||
|
||||
|
||||
def _make_provisional_lookup(matches_by_section: dict[str, _StubV4Match]):
|
||||
"""Lookup_fn that flags reject rank-1 matches provisional.
|
||||
|
||||
Mirrors the pipeline u12 retry ``_lookup_fn_mixed_admission`` at the
|
||||
inserted block — for reject judgments, returns a provisional=True
|
||||
rank-1 V4Match-shaped stub so plan_composition's last-resort fill
|
||||
pool can see it (provisional candidates are otherwise filtered out
|
||||
of the normal greedy pass).
|
||||
"""
|
||||
def _fn(section_id: str):
|
||||
m = matches_by_section.get(section_id)
|
||||
if m is not None and m.label == "reject":
|
||||
# Synthesize the provisional shape that
|
||||
# lookup_v4_match_with_fallback returns when allow_provisional
|
||||
# is True: provisional=True + selection_path="provisional_rank_1".
|
||||
return _StubV4Match(
|
||||
template_id=m.template_id,
|
||||
frame_id=m.frame_id,
|
||||
frame_number=m.frame_number,
|
||||
confidence=m.confidence,
|
||||
label=m.label,
|
||||
v4_rank=1,
|
||||
selection_path="provisional_rank_1",
|
||||
provisional=True,
|
||||
)
|
||||
return m
|
||||
return _fn
|
||||
|
||||
|
||||
def _make_candidates_lookup_empty():
|
||||
def _fn(section_id: str):
|
||||
return []
|
||||
return _fn
|
||||
|
||||
|
||||
# ─── u12 case 1 : mechanic — mixed admission via provisional lookup + fill ────
|
||||
|
||||
|
||||
def test_u12_mechanic_mixed_admission_covers_reject_section_via_provisional_fill():
|
||||
"""Positive proof. Mixed direct+reject (S1=use_as_is, S2=reject).
|
||||
|
||||
Without u12 (initial path: normal lookup + allow_provisional_fill=False),
|
||||
plan_composition returns only the S1 unit and S2 is silently dropped.
|
||||
|
||||
With u12 (retry: provisional lookup + allow_provisional_fill=True),
|
||||
plan_composition returns both units; S2 is a provisional unit with
|
||||
label="reject" — ready to be picked up by Step 12's reject route
|
||||
gather (IMP-47B u4).
|
||||
"""
|
||||
sections = [_StubSection("S1"), _StubSection("S2")]
|
||||
matches = {
|
||||
"S1": _StubV4Match(
|
||||
template_id="MOCK_template_direct_a",
|
||||
frame_id="MOCK_frame_001",
|
||||
frame_number=1,
|
||||
confidence=0.92,
|
||||
label="use_as_is",
|
||||
v4_rank=1,
|
||||
),
|
||||
"S2": _StubV4Match(
|
||||
template_id="MOCK_template_reject_a",
|
||||
frame_id="MOCK_frame_002",
|
||||
frame_number=2,
|
||||
confidence=0.30,
|
||||
label="reject",
|
||||
v4_rank=1,
|
||||
),
|
||||
}
|
||||
|
||||
# Pre-u12 baseline — normal lookup, no provisional fill.
|
||||
units_pre, preset_pre, _ = plan_composition(
|
||||
sections,
|
||||
_make_normal_lookup(matches),
|
||||
_LABEL_TO_STATUS,
|
||||
_ALLOWED_STATUSES,
|
||||
v4_candidates_lookup_fn=_make_candidates_lookup_empty(),
|
||||
)
|
||||
covered_pre = {sid for u in units_pre for sid in u.source_section_ids}
|
||||
assert "S1" in covered_pre, "S1 (use_as_is) must cover pre-u12"
|
||||
assert "S2" not in covered_pre, (
|
||||
"Pre-u12 baseline regression: reject S2 should be uncovered (no provisional fill)"
|
||||
)
|
||||
|
||||
# u12 mixed-admission retry — provisional lookup + allow_provisional_fill=True.
|
||||
units_post, preset_post, _ = plan_composition(
|
||||
sections,
|
||||
_make_provisional_lookup(matches),
|
||||
_LABEL_TO_STATUS,
|
||||
_ALLOWED_STATUSES,
|
||||
v4_candidates_lookup_fn=_make_candidates_lookup_empty(),
|
||||
allow_provisional_fill=True,
|
||||
)
|
||||
covered_post = {sid for u in units_post for sid in u.source_section_ids}
|
||||
assert covered_post == {"S1", "S2"}, (
|
||||
"u12 mixed admission must cover every section (no text loss)"
|
||||
)
|
||||
assert preset_post is not None
|
||||
# The S2 unit must be marked provisional so the reject route gather
|
||||
# (src/phase_z2_ai_fallback/step12.py:133-136) admits it.
|
||||
s2_unit = next(u for u in units_post if "S2" in u.source_section_ids)
|
||||
assert s2_unit.provisional is True, (
|
||||
"Reject S2 unit must be provisional so Step 12 reject route admits it"
|
||||
)
|
||||
assert s2_unit.label == "reject"
|
||||
# Frame template id is preserved — no auto frame swap.
|
||||
assert s2_unit.frame_template_id == "MOCK_template_reject_a"
|
||||
|
||||
|
||||
# ─── u12 case 2 : gate — all-direct slides are a no-op ──────────────────────
|
||||
|
||||
|
||||
def test_u12_gate_all_direct_yields_no_uncovered_sections():
|
||||
"""No-op proof. When every section is auto-renderable (use_as_is or
|
||||
light_edit), the initial plan_composition covers everything — the
|
||||
u12 mixed-admission gate's ``_u12_uncovered_ids`` list is empty and
|
||||
the retry is skipped.
|
||||
"""
|
||||
sections = [_StubSection("S1"), _StubSection("S2")]
|
||||
matches = {
|
||||
"S1": _StubV4Match(
|
||||
template_id="MOCK_template_direct_a",
|
||||
frame_id="MOCK_frame_001",
|
||||
frame_number=1,
|
||||
confidence=0.92,
|
||||
label="use_as_is",
|
||||
v4_rank=1,
|
||||
),
|
||||
"S2": _StubV4Match(
|
||||
template_id="MOCK_template_direct_b",
|
||||
frame_id="MOCK_frame_002",
|
||||
frame_number=2,
|
||||
confidence=0.81,
|
||||
label="light_edit",
|
||||
v4_rank=1,
|
||||
),
|
||||
}
|
||||
units, preset, _ = plan_composition(
|
||||
sections,
|
||||
_make_normal_lookup(matches),
|
||||
_LABEL_TO_STATUS,
|
||||
_ALLOWED_STATUSES,
|
||||
v4_candidates_lookup_fn=_make_candidates_lookup_empty(),
|
||||
)
|
||||
covered = {sid for u in units for sid in u.source_section_ids}
|
||||
assert covered == {"S1", "S2"}, "All-direct must cover every section pre-u12"
|
||||
# Predicate from src/phase_z2_pipeline.py u12 block:
|
||||
uncovered = [s.section_id for s in sections if s.section_id not in covered]
|
||||
assert uncovered == [], (
|
||||
"u12 gate must classify all-direct as no-op (uncovered list empty)"
|
||||
)
|
||||
assert preset is not None
|
||||
|
||||
|
||||
# ─── u12 case 3 : gate — initial empty units bypass u12 (IMP-30 retry owns it) ──
|
||||
|
||||
|
||||
def test_u12_gate_skips_when_initial_units_empty():
|
||||
"""All-reject case is owned by IMP-30 u4 retry (units=[] guard at
|
||||
src/phase_z2_pipeline.py:3646). u12 mixed-admission must NOT compete
|
||||
with that path; the gate ``units and layout_preset is not None``
|
||||
short-circuits when the initial plan_composition returns nothing.
|
||||
"""
|
||||
sections = [_StubSection("S1")]
|
||||
matches = {
|
||||
"S1": _StubV4Match(
|
||||
template_id="MOCK_template_reject_a",
|
||||
frame_id="MOCK_frame_002",
|
||||
frame_number=2,
|
||||
confidence=0.30,
|
||||
label="reject",
|
||||
v4_rank=1,
|
||||
),
|
||||
}
|
||||
units, preset, _ = plan_composition(
|
||||
sections,
|
||||
_make_normal_lookup(matches),
|
||||
_LABEL_TO_STATUS,
|
||||
_ALLOWED_STATUSES,
|
||||
v4_candidates_lookup_fn=_make_candidates_lookup_empty(),
|
||||
)
|
||||
# All-reject initial pass: no auto-renderable units, no layout preset.
|
||||
assert units == [] and preset is None
|
||||
# u12 gate predicate would short-circuit on `units` truthiness:
|
||||
gate_active = bool(units) and preset is not None
|
||||
assert gate_active is False, (
|
||||
"u12 mixed-admission gate must skip the all-reject case (IMP-30 u4 owns it)"
|
||||
)
|
||||
|
||||
|
||||
# ─── u12 case 4 : code-path anchor — pipeline source contains u12 marker ────
|
||||
|
||||
|
||||
def test_u12_pipeline_source_contains_mixed_admission_marker():
|
||||
"""Anchor test. Ensures the inserted u12 block in src/phase_z2_pipeline.py
|
||||
is reachable (not silently removed by a future refactor).
|
||||
|
||||
Asserts on the marker comment + ``imp47b_u12_mixed_admission`` debug key
|
||||
+ ``allow_provisional_fill=True`` invocation co-located in the file.
|
||||
Cheap structural guard — does not run the heavy pipeline.
|
||||
"""
|
||||
from pathlib import Path
|
||||
src_path = Path(__file__).resolve().parent.parent / "src" / "phase_z2_pipeline.py"
|
||||
text = src_path.read_text(encoding="utf-8")
|
||||
assert "IMP-47B u12 — mixed direct+reject first-render admission" in text, (
|
||||
"u12 marker comment missing from pipeline — block may have been removed"
|
||||
)
|
||||
assert "imp47b_u12_mixed_admission" in text, (
|
||||
"u12 comp_debug telemetry key missing"
|
||||
)
|
||||
# The mixed-admission retry must pass allow_provisional_fill=True.
|
||||
# Anchor against the helper function name + the kwarg co-occurrence.
|
||||
assert "_lookup_fn_mixed_admission" in text
|
||||
assert "allow_provisional_fill=True" in text
|
||||
@@ -0,0 +1,180 @@
|
||||
"""IMP-47B u3 — override-selected reject frames are admitted as provisional.
|
||||
|
||||
Scope (this slice):
|
||||
Helper `_apply_frame_override_to_unit` (src/phase_z2_pipeline.py) covers
|
||||
the three probe layers used by the `--override-frame` path:
|
||||
|
||||
1. ``v4_candidates`` exact match (non-reject; existing behaviour).
|
||||
2. Full 32 V4 judgments probe (reject inclusive) — when the user
|
||||
picks a reject frame, the unit is promoted to
|
||||
``provisional=True`` with ``label="reject"`` so Step 12
|
||||
(IMP-47B u4) admits the AI repair path.
|
||||
3. Raw fall-through (template_id only) — no provisional promotion,
|
||||
no label mutation.
|
||||
|
||||
Frame visual / contract stay untouched per the AI isolation contract
|
||||
(frame auto-swap forbidden — AI re-places content into the existing
|
||||
frame only). Sibling test confirms a non-reject override still goes
|
||||
through the v4_candidates path without provisional promotion.
|
||||
|
||||
Synthetic naming convention mirrors tests/test_phase_z2_imp30_first_render.py
|
||||
(MOCK_ prefix mandatory, no real catalog template_id / frame_id leakage).
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Optional
|
||||
|
||||
from src.phase_z2_pipeline import _apply_frame_override_to_unit
|
||||
|
||||
|
||||
@dataclass
|
||||
class _StubCandidate:
|
||||
template_id: str
|
||||
frame_id: str
|
||||
frame_number: int
|
||||
confidence: float
|
||||
label: str
|
||||
|
||||
|
||||
@dataclass
|
||||
class _StubUnit:
|
||||
source_section_ids: list[str]
|
||||
frame_template_id: Optional[str] = None
|
||||
frame_id: Optional[str] = None
|
||||
frame_number: int = 0
|
||||
confidence: float = 0.0
|
||||
label: Optional[str] = None
|
||||
provisional: bool = False
|
||||
v4_candidates: list = field(default_factory=list)
|
||||
|
||||
|
||||
def _v4_with_reject(section_id: str, target_tid: str) -> dict:
|
||||
"""Synthetic V4 dict with target_tid mapped to a reject judgment.
|
||||
|
||||
Mirrors the production V4 schema surface (``mdx_sections`` →
|
||||
``judgments_full32`` → list of judgment dicts with template_id /
|
||||
frame_id / frame_number / confidence / label). Two judgments so we
|
||||
can also assert that the helper picks the reject entry rather than
|
||||
the first non-reject one when the template_ids differ.
|
||||
"""
|
||||
return {
|
||||
"mdx_sections": {
|
||||
section_id: {
|
||||
"judgments_full32": [
|
||||
{
|
||||
"template_id": "MOCK_T_other",
|
||||
"frame_id": "F_other",
|
||||
"frame_number": 1,
|
||||
"confidence": 0.85,
|
||||
"label": "use_as_is",
|
||||
},
|
||||
{
|
||||
"template_id": target_tid,
|
||||
"frame_id": "F_reject",
|
||||
"frame_number": 32,
|
||||
"confidence": 0.40,
|
||||
"label": "reject",
|
||||
},
|
||||
],
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
# ─── Case 1 : reject override → provisional promotion ────────────
|
||||
|
||||
|
||||
def test_override_to_reject_judgment_marks_unit_provisional():
|
||||
"""User picks a reject frame → unit.label=reject, provisional=True.
|
||||
|
||||
Frame metadata is sourced from the reject judgment (frame_id /
|
||||
frame_number / confidence) so Step 9 metadata stays consistent.
|
||||
"""
|
||||
unit = _StubUnit(
|
||||
source_section_ids=["MOCK_S1"],
|
||||
frame_template_id="MOCK_T_auto",
|
||||
frame_id="F_auto",
|
||||
frame_number=5,
|
||||
confidence=0.90,
|
||||
label="use_as_is",
|
||||
provisional=False,
|
||||
)
|
||||
v4 = _v4_with_reject("MOCK_S1", "MOCK_T_reject")
|
||||
|
||||
meta = _apply_frame_override_to_unit(unit, "MOCK_T_reject", v4)
|
||||
|
||||
assert meta == "v4_reject_judgment_provisional"
|
||||
assert unit.frame_template_id == "MOCK_T_reject"
|
||||
assert unit.frame_id == "F_reject"
|
||||
assert unit.frame_number == 32
|
||||
assert unit.confidence == 0.40
|
||||
assert unit.label == "reject"
|
||||
assert unit.provisional is True
|
||||
|
||||
|
||||
# ─── Case 2 : non-reject override → existing v4_candidates path ───
|
||||
|
||||
|
||||
def test_override_to_v4_candidate_keeps_non_provisional():
|
||||
"""User picks a non-reject candidate → existing v4_candidates path.
|
||||
|
||||
Helper takes the early v4_candidates branch without consulting the
|
||||
full 32 judgments. provisional remains False (normal-path AI=0
|
||||
contract — IMP-30 / IMP-47B router gate intact for this unit).
|
||||
"""
|
||||
unit = _StubUnit(
|
||||
source_section_ids=["MOCK_S2"],
|
||||
frame_template_id="MOCK_T_auto",
|
||||
frame_id="F_auto",
|
||||
frame_number=3,
|
||||
confidence=0.95,
|
||||
label="use_as_is",
|
||||
provisional=False,
|
||||
v4_candidates=[
|
||||
_StubCandidate(
|
||||
template_id="MOCK_T_pick",
|
||||
frame_id="F_pick",
|
||||
frame_number=2,
|
||||
confidence=0.85,
|
||||
label="light_edit",
|
||||
),
|
||||
],
|
||||
)
|
||||
v4 = {"mdx_sections": {}} # full-judgment probe must NOT be reached
|
||||
|
||||
meta = _apply_frame_override_to_unit(unit, "MOCK_T_pick", v4)
|
||||
|
||||
assert meta == "v4_candidates"
|
||||
assert unit.frame_template_id == "MOCK_T_pick"
|
||||
assert unit.frame_id == "F_pick"
|
||||
assert unit.label == "light_edit"
|
||||
assert unit.provisional is False
|
||||
|
||||
|
||||
# ─── Case 3 : unknown template → raw fall-through (no provisional) ─
|
||||
|
||||
|
||||
def test_override_unknown_template_falls_through_without_provisional():
|
||||
"""Template ID absent from v4_candidates AND from judgments_full32 →
|
||||
raw_template_id_only path. No provisional flag, no label change.
|
||||
"""
|
||||
unit = _StubUnit(
|
||||
source_section_ids=["MOCK_S3"],
|
||||
frame_template_id="MOCK_T_auto",
|
||||
frame_id="F_auto",
|
||||
frame_number=4,
|
||||
confidence=0.92,
|
||||
label="use_as_is",
|
||||
provisional=False,
|
||||
)
|
||||
v4 = {"mdx_sections": {}}
|
||||
|
||||
meta = _apply_frame_override_to_unit(unit, "MOCK_T_unknown", v4)
|
||||
|
||||
assert meta == "raw_template_id_only"
|
||||
assert unit.frame_template_id == "MOCK_T_unknown"
|
||||
# frame_id / label unchanged — caller's print path warns on this case.
|
||||
assert unit.frame_id == "F_auto"
|
||||
assert unit.label == "use_as_is"
|
||||
assert unit.provisional is False
|
||||
@@ -0,0 +1,223 @@
|
||||
"""IMP-47B u5 — PARTIAL_OVERRIDES apply tests.
|
||||
|
||||
Scope (this slice):
|
||||
Helper ``_apply_ai_repair_proposals_to_zones`` (src/phase_z2_pipeline.py)
|
||||
merges ``proposal.payload.slots`` into ``zones_data[k]["slot_payload"]``
|
||||
for PARTIAL_OVERRIDES proposals only, and loud-fails out-of-scope
|
||||
proposal kinds (builder_options_patch, slot_mapping_proposal) with an
|
||||
explicit ``apply_status`` marker.
|
||||
|
||||
The IMP-33 u5 validator inside ``route_ai_fallback`` already enforces
|
||||
declared-slot completeness — the apply helper is therefore a structural
|
||||
merge over the validator's contract, not a per-slot guard re-implementation.
|
||||
|
||||
u6 (step12_ai_repair.json audit), u7 (coverage invariant), and u8
|
||||
(slide_status surfacing) are out of scope for this unit.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from src.phase_z2_pipeline import _apply_ai_repair_proposals_to_zones
|
||||
|
||||
|
||||
def _record(
|
||||
*,
|
||||
unit_index: int,
|
||||
proposal: dict | None,
|
||||
source_section_ids: list[str] | None = None,
|
||||
) -> dict:
|
||||
"""Synthetic gather_step12_ai_repair_proposals record."""
|
||||
return {
|
||||
"unit_index": unit_index,
|
||||
"source_section_ids": source_section_ids or [f"MOCK_S{unit_index}"],
|
||||
"frame_template_id": "MOCK_T",
|
||||
"label": "reject",
|
||||
"route_hint": "ai_adaptation_required",
|
||||
"provisional": True,
|
||||
"ai_called": proposal is not None,
|
||||
"skip_reason": None,
|
||||
"proposal": proposal,
|
||||
"error": None,
|
||||
"cache_key": "MOCK_F::abc" if proposal is not None else None,
|
||||
"fingerprints": {"contract_sha": "x", "partial_sha": "y", "catalog_sha": ""}
|
||||
if proposal is not None
|
||||
else None,
|
||||
}
|
||||
|
||||
|
||||
def _zone(*, position: str, slot_payload: dict | None = None) -> dict:
|
||||
"""Synthetic zones_data entry — only fields the apply helper touches."""
|
||||
return {
|
||||
"position": position,
|
||||
"template_id": "MOCK_T",
|
||||
"slot_payload": slot_payload if slot_payload is not None else {},
|
||||
}
|
||||
|
||||
|
||||
# ─── Case 1 : PARTIAL_OVERRIDES → merged + applied marker ──────────
|
||||
|
||||
|
||||
def test_partial_overrides_merges_slots_into_zone_slot_payload():
|
||||
"""The validator already guarantees declared-slot completeness, so
|
||||
apply is a structural ``dict.update``. Pre-existing meta keys
|
||||
(``_truncated_count``) survive; declared slot values are replaced
|
||||
by the AI proposal values."""
|
||||
proposal = {
|
||||
"proposal_kind": "partial_overrides",
|
||||
"payload": {
|
||||
"slots": {
|
||||
"title": "AI title",
|
||||
"bullets": ["AI bullet 1", "AI bullet 2"],
|
||||
}
|
||||
},
|
||||
"rationale": "MOCK",
|
||||
}
|
||||
records = [_record(unit_index=0, proposal=proposal)]
|
||||
zones = [
|
||||
_zone(
|
||||
position="top",
|
||||
slot_payload={
|
||||
"title": "deterministic title",
|
||||
"bullets": ["det bullet"],
|
||||
"_truncated_count": 0,
|
||||
},
|
||||
)
|
||||
]
|
||||
|
||||
_apply_ai_repair_proposals_to_zones(records, ["top"], zones)
|
||||
|
||||
assert records[0]["apply_status"] == "applied:partial_overrides"
|
||||
assert zones[0]["slot_payload"]["title"] == "AI title"
|
||||
assert zones[0]["slot_payload"]["bullets"] == ["AI bullet 1", "AI bullet 2"]
|
||||
# meta keys not in proposal must survive the merge
|
||||
assert zones[0]["slot_payload"]["_truncated_count"] == 0
|
||||
|
||||
|
||||
# ─── Case 2 : BUILDER_OPTIONS_PATCH → loud-fail unsupported_kind ───
|
||||
|
||||
|
||||
def test_builder_options_patch_is_unsupported_for_reject_route():
|
||||
"""Builder-options application is out-of-scope for IMP-47B reject
|
||||
route (see Stage 2 plan). u5 must mark, not apply — the zone
|
||||
slot_payload stays byte-identical and the record carries the
|
||||
``unsupported_kind_for_reject_route:<kind>`` marker so u8 can
|
||||
surface human_review downstream."""
|
||||
proposal = {
|
||||
"proposal_kind": "builder_options_patch",
|
||||
"payload": {"font_size_px": 14},
|
||||
"rationale": "MOCK",
|
||||
}
|
||||
records = [_record(unit_index=0, proposal=proposal)]
|
||||
original_slot_payload = {"title": "deterministic"}
|
||||
zones = [_zone(position="top", slot_payload=dict(original_slot_payload))]
|
||||
|
||||
_apply_ai_repair_proposals_to_zones(records, ["top"], zones)
|
||||
|
||||
assert (
|
||||
records[0]["apply_status"]
|
||||
== "unsupported_kind_for_reject_route:builder_options_patch"
|
||||
)
|
||||
assert zones[0]["slot_payload"] == original_slot_payload
|
||||
|
||||
|
||||
# ─── Case 3 : SLOT_MAPPING_PROPOSAL → loud-fail unsupported_kind ───
|
||||
|
||||
|
||||
def test_slot_mapping_proposal_is_unsupported_for_reject_route():
|
||||
"""Slot-mapping (restructuring) application is also out-of-scope —
|
||||
builder-options + slot-mapping share the same marker path."""
|
||||
proposal = {
|
||||
"proposal_kind": "slot_mapping_proposal",
|
||||
"payload": {"slots": {"title": "x"}},
|
||||
"rationale": "MOCK",
|
||||
}
|
||||
records = [_record(unit_index=0, proposal=proposal)]
|
||||
zones = [_zone(position="top", slot_payload={"title": "deterministic"})]
|
||||
|
||||
_apply_ai_repair_proposals_to_zones(records, ["top"], zones)
|
||||
|
||||
assert (
|
||||
records[0]["apply_status"]
|
||||
== "unsupported_kind_for_reject_route:slot_mapping_proposal"
|
||||
)
|
||||
assert zones[0]["slot_payload"] == {"title": "deterministic"}
|
||||
|
||||
|
||||
# ─── Case 4 : no proposal (router short-circuit / not_provisional) ──
|
||||
|
||||
|
||||
def test_record_without_proposal_marked_no_proposal_and_zone_untouched():
|
||||
"""Flag-off short-circuit and non-AI-route units carry
|
||||
``proposal=None``. apply_status must distinguish "no proposal to
|
||||
apply" from real apply outcomes so u8 can categorise the per-unit
|
||||
status without re-reading skip_reason."""
|
||||
records = [_record(unit_index=0, proposal=None)]
|
||||
zones = [_zone(position="top", slot_payload={"title": "deterministic"})]
|
||||
|
||||
_apply_ai_repair_proposals_to_zones(records, ["top"], zones)
|
||||
|
||||
assert records[0]["apply_status"] == "no_proposal"
|
||||
assert zones[0]["slot_payload"] == {"title": "deterministic"}
|
||||
|
||||
|
||||
# ─── Case 5 : proposal exists but no matching zone (B4 mismatch) ────
|
||||
|
||||
|
||||
def test_proposal_for_unit_without_zone_match_marked_no_zone_match():
|
||||
"""When a unit is dropped from zones_data (B4 mismatch or FitError
|
||||
in the Step 12 render loop) but still gathered an AI proposal,
|
||||
apply must surface the mismatch via ``no_zone_match`` rather than
|
||||
silently dropping the proposal or writing into a wrong zone."""
|
||||
proposal = {
|
||||
"proposal_kind": "partial_overrides",
|
||||
"payload": {"slots": {"title": "AI title"}},
|
||||
"rationale": "MOCK",
|
||||
}
|
||||
records = [_record(unit_index=0, proposal=proposal)]
|
||||
# unit_positions[0]="top" but zones_data has only the bottom zone
|
||||
# → no match for the dropped unit's position.
|
||||
zones = [_zone(position="bottom", slot_payload={"title": "other zone"})]
|
||||
|
||||
_apply_ai_repair_proposals_to_zones(records, ["top"], zones)
|
||||
|
||||
assert records[0]["apply_status"] == "no_zone_match"
|
||||
# untouched zone — apply must not bleed into a different position
|
||||
assert zones[0]["slot_payload"] == {"title": "other zone"}
|
||||
|
||||
|
||||
# ─── Case 6 : mixed records — independent per-record classification ──
|
||||
|
||||
|
||||
def test_mixed_records_classified_independently():
|
||||
"""All five apply_status branches coexist in one batch — confirms
|
||||
the helper does not short-circuit on the first non-applied record."""
|
||||
records = [
|
||||
_record(unit_index=0, proposal={
|
||||
"proposal_kind": "partial_overrides",
|
||||
"payload": {"slots": {"title": "AI"}},
|
||||
"rationale": "",
|
||||
}),
|
||||
_record(unit_index=1, proposal={
|
||||
"proposal_kind": "builder_options_patch",
|
||||
"payload": {"font_size_px": 14},
|
||||
"rationale": "",
|
||||
}),
|
||||
_record(unit_index=2, proposal=None),
|
||||
]
|
||||
zones = [
|
||||
_zone(position="top", slot_payload={"title": "det"}),
|
||||
_zone(position="middle", slot_payload={"title": "det"}),
|
||||
_zone(position="bottom", slot_payload={"title": "det"}),
|
||||
]
|
||||
|
||||
_apply_ai_repair_proposals_to_zones(
|
||||
records, ["top", "middle", "bottom"], zones,
|
||||
)
|
||||
|
||||
assert [r["apply_status"] for r in records] == [
|
||||
"applied:partial_overrides",
|
||||
"unsupported_kind_for_reject_route:builder_options_patch",
|
||||
"no_proposal",
|
||||
]
|
||||
assert zones[0]["slot_payload"]["title"] == "AI"
|
||||
assert zones[1]["slot_payload"]["title"] == "det"
|
||||
assert zones[2]["slot_payload"]["title"] == "det"
|
||||
@@ -0,0 +1,154 @@
|
||||
"""IMP-47B u4 + u6 — Step 12 AI repair wiring + audit artifact tests.
|
||||
|
||||
Scope (this slice):
|
||||
* u4 — Helper ``_run_step12_ai_repair`` (src/phase_z2_pipeline.py)
|
||||
wires the pipeline's local route-hint helper (``_imp05_route_hint``),
|
||||
the frame contract loader (``get_contract``), and a
|
||||
templates/phase_z2/families partial reader
|
||||
(``_load_frame_partial_html``) into
|
||||
``gather_step12_ai_repair_proposals``.
|
||||
* u6 — The gather records flow into ``_write_step_artifact`` under
|
||||
``step12_ai_repair.json``. The audit shape must stay
|
||||
JSON-serialisable (no Pydantic / dataclass leakage) so the artifact
|
||||
write never raises on real runs.
|
||||
|
||||
The router short-circuits when ``settings.ai_fallback_enabled`` is
|
||||
False (default), so AI=0 for non-AI-route units stays a structural
|
||||
guarantee. Synthetic naming mirrors tests/test_imp47b_override_provisional.py
|
||||
(MOCK_ prefix; no real catalog template_id / frame_id leakage).
|
||||
|
||||
u5 (PARTIAL_OVERRIDES apply), u7 (coverage invariant), and u8
|
||||
(slide_status surfacing) are out of scope for this unit.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
from src.phase_z2_pipeline import (
|
||||
_load_frame_partial_html,
|
||||
_run_step12_ai_repair,
|
||||
_write_step_artifact,
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class _StubUnit:
|
||||
label: str | None
|
||||
provisional: bool
|
||||
frame_template_id: str = "MOCK_T_x"
|
||||
frame_id: str = "MOCK_F_x"
|
||||
source_section_ids: list[str] = field(default_factory=lambda: ["MOCK_S1"])
|
||||
raw_content: str = "MOCK_raw"
|
||||
v4_rank: int | None = 1
|
||||
cardinality: int | None = None
|
||||
layout_preset: str = ""
|
||||
zone_position: str = ""
|
||||
source_shape: str = "paragraph"
|
||||
h3_count: int = 0
|
||||
char_count: int = 0
|
||||
|
||||
|
||||
# ─── Case 1 : mixed units → per-unit skip_reason classification ─────
|
||||
|
||||
|
||||
def test_mixed_units_classified_by_route_and_provisional_flag():
|
||||
"""Reject + restructure provisional both route to ai_adaptation;
|
||||
use_as_is / light_edit / non-provisional skip without router call.
|
||||
|
||||
With ai_fallback_enabled=False (default) the router returns None,
|
||||
so the two ai_adaptation provisional units record
|
||||
``skip_reason='router_short_circuit'``; the rest record their
|
||||
structural skip_reason (not_provisional / route_not_ai_adaptation).
|
||||
"""
|
||||
units = [
|
||||
_StubUnit(label="use_as_is", provisional=False),
|
||||
_StubUnit(label="light_edit", provisional=True),
|
||||
_StubUnit(label="restructure", provisional=True),
|
||||
_StubUnit(label="reject", provisional=True),
|
||||
_StubUnit(label="restructure", provisional=False),
|
||||
]
|
||||
records = _run_step12_ai_repair(units)
|
||||
assert [r["skip_reason"] for r in records] == [
|
||||
"not_provisional",
|
||||
"route_not_ai_adaptation:deterministic_minor_adjustment",
|
||||
"router_short_circuit",
|
||||
"router_short_circuit",
|
||||
"not_provisional",
|
||||
]
|
||||
assert [r["route_hint"] for r in records] == [
|
||||
"direct_render",
|
||||
"deterministic_minor_adjustment",
|
||||
"ai_adaptation_required",
|
||||
"ai_adaptation_required",
|
||||
"ai_adaptation_required",
|
||||
]
|
||||
assert all(r["ai_called"] is False for r in records)
|
||||
|
||||
|
||||
# ─── Case 2 : reject provisional unit reaches AI gate ───────────────
|
||||
|
||||
|
||||
def test_reject_provisional_unit_reaches_router_short_circuit():
|
||||
"""Reject + provisional → route_hint=ai_adaptation_required.
|
||||
|
||||
Router short-circuit (flag-off default) is the only thing keeping
|
||||
AI from firing; the wiring proves reject is no longer blocked by
|
||||
Step 12's bespoke design_reference_only skip (removed by u2).
|
||||
"""
|
||||
records = _run_step12_ai_repair([_StubUnit(label="reject", provisional=True)])
|
||||
assert records[0]["route_hint"] == "ai_adaptation_required"
|
||||
assert records[0]["skip_reason"] == "router_short_circuit"
|
||||
assert records[0]["ai_called"] is False
|
||||
# cache_key / fingerprints populated only after the route + provisional
|
||||
# gates pass — confirms gather reached the AI-eligible code path.
|
||||
assert records[0]["cache_key"] is not None
|
||||
assert records[0]["fingerprints"] is not None
|
||||
|
||||
|
||||
# ─── Case 3 : frame visual loader degrades on missing partial ──────
|
||||
|
||||
|
||||
def test_load_frame_partial_html_returns_empty_for_missing_file():
|
||||
"""__empty__ shell (IMP-30) and any unknown template_id → "".
|
||||
|
||||
Keeps gather() crash-free for the IMP-30 first-render-invariant
|
||||
path where the synthesized empty-shell unit has no families partial.
|
||||
"""
|
||||
assert _load_frame_partial_html("__empty__") == ""
|
||||
assert _load_frame_partial_html("MOCK_T_does_not_exist") == ""
|
||||
|
||||
|
||||
# ─── Case 4 (u6) : audit artifact write is JSON-serialisable ────────
|
||||
|
||||
|
||||
def test_step12_ai_repair_artifact_writes_json_serialisable_records(tmp_path):
|
||||
"""IMP-47B u6 — gather records feed ``_write_step_artifact`` as the
|
||||
``step12_ai_repair.json`` audit. Confirms the gather schema contains
|
||||
only JSON-native primitives (str / int / None / bool / list / dict)
|
||||
so the artifact write never raises on real runs and the audit
|
||||
payload preserves per-unit ``route_hint`` / ``skip_reason`` /
|
||||
``ai_called`` for reviewers.
|
||||
"""
|
||||
records = _run_step12_ai_repair([
|
||||
_StubUnit(label="reject", provisional=True),
|
||||
_StubUnit(label="use_as_is", provisional=False),
|
||||
])
|
||||
fpath = _write_step_artifact(
|
||||
tmp_path, 12, "ai_repair",
|
||||
data={"per_unit": records},
|
||||
outputs=["step12_ai_repair.json"],
|
||||
)
|
||||
assert fpath.is_file()
|
||||
assert fpath.name == "step12_ai_repair.json"
|
||||
payload = json.loads(fpath.read_text(encoding="utf-8"))
|
||||
assert payload["step_num"] == 12
|
||||
assert payload["step_name"] == "ai_repair"
|
||||
assert payload["step_status"] == "done"
|
||||
per_unit = payload["data"]["per_unit"]
|
||||
assert len(per_unit) == 2
|
||||
assert per_unit[0]["route_hint"] == "ai_adaptation_required"
|
||||
assert per_unit[0]["skip_reason"] == "router_short_circuit"
|
||||
assert per_unit[0]["ai_called"] is False
|
||||
assert per_unit[1]["route_hint"] == "direct_render"
|
||||
assert per_unit[1]["skip_reason"] == "not_provisional"
|
||||
@@ -0,0 +1,55 @@
|
||||
"""Unit tests for ``src.json_utils.parse_json``.
|
||||
|
||||
IMP-28 L4 — `_parse_json` dedup unit u2.
|
||||
|
||||
Pins the shared helper semantics that previously lived in
|
||||
content_editor.py / design_director.py / kei_client.py (fuller form) and
|
||||
pipeline.py (simple form). The fuller form is a strict superset of the
|
||||
simple form; these tests cover both axes.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from src.json_utils import parse_json
|
||||
|
||||
|
||||
def test_parse_json_fenced_json_block():
|
||||
text = 'prefix\n```json\n{"a": 1, "b": "x"}\n```\nsuffix'
|
||||
assert parse_json(text) == {"a": 1, "b": "x"}
|
||||
|
||||
|
||||
def test_parse_json_plain_fenced_block():
|
||||
text = 'prefix\n```\n{"x": 2}\n```\nsuffix'
|
||||
assert parse_json(text) == {"x": 2}
|
||||
|
||||
|
||||
def test_parse_json_bare_braces():
|
||||
text = 'noise before {"y": 3, "z": [1, 2]} noise after'
|
||||
assert parse_json(text) == {"y": 3, "z": [1, 2]}
|
||||
|
||||
|
||||
def test_parse_json_list_prefix_dash_cleanup():
|
||||
text = '- {"b": 4}'
|
||||
assert parse_json(text) == {"b": 4}
|
||||
|
||||
|
||||
def test_parse_json_list_prefix_star_cleanup():
|
||||
text = '* {"c": 5}'
|
||||
assert parse_json(text) == {"c": 5}
|
||||
|
||||
|
||||
def test_parse_json_no_json_returns_none():
|
||||
assert parse_json("no json here at all") is None
|
||||
|
||||
|
||||
def test_parse_json_malformed_returns_none():
|
||||
assert parse_json("{ invalid json") is None
|
||||
|
||||
|
||||
def test_parse_json_prefix_free_no_op():
|
||||
text = '{"d": 6}'
|
||||
assert parse_json(text) == {"d": 6}
|
||||
|
||||
|
||||
def test_parse_json_fenced_preferred_over_bare_braces():
|
||||
text = 'outer {"outer": true} ```json\n{"inner": 1}\n```'
|
||||
assert parse_json(text) == {"inner": 1}
|
||||
@@ -0,0 +1,86 @@
|
||||
"""IMP-33 u1 — AI fallback Settings defaults (locked).
|
||||
|
||||
These defaults are the binding contract from Stage 2 plan (per-unit u1):
|
||||
- ai_fallback_enabled = False (master flag OFF; fallback path only)
|
||||
- ai_fallback_model = "claude-opus-4-6-20250415"
|
||||
- ai_fallback_timeout_s = 60.0
|
||||
- ai_fallback_max_retries = 3
|
||||
- ai_fallback_backoff_base_s = 1.0
|
||||
- ai_fallback_backoff_cap_s = 8.0
|
||||
- ai_fallback_backoff_jitter = 0.3
|
||||
- ai_fallback_budget_per_run = 10
|
||||
- ai_fallback_circuit_breaker_threshold = 5
|
||||
|
||||
Downstream u4 (client) MUST source timeout/retry/backoff/budget/circuit from
|
||||
Settings; inline literals are forbidden by Stage 2 plan.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from src.config import Settings
|
||||
|
||||
|
||||
def test_ai_fallback_master_flag_default_off() -> None:
|
||||
s = Settings()
|
||||
assert s.ai_fallback_enabled is False, (
|
||||
"AI fallback master flag MUST default OFF (normal path AI=0 contract)."
|
||||
)
|
||||
|
||||
|
||||
def test_ai_fallback_model_default_locked() -> None:
|
||||
s = Settings()
|
||||
assert s.ai_fallback_model == "claude-opus-4-6-20250415"
|
||||
|
||||
|
||||
def test_ai_fallback_retry_timeout_backoff_defaults_locked() -> None:
|
||||
s = Settings()
|
||||
assert s.ai_fallback_timeout_s == 60.0
|
||||
assert s.ai_fallback_max_retries == 3
|
||||
assert s.ai_fallback_backoff_base_s == 1.0
|
||||
assert s.ai_fallback_backoff_cap_s == 8.0
|
||||
assert s.ai_fallback_backoff_jitter == 0.3
|
||||
|
||||
|
||||
def test_ai_fallback_budget_and_circuit_defaults_locked() -> None:
|
||||
s = Settings()
|
||||
assert s.ai_fallback_budget_per_run == 10
|
||||
assert s.ai_fallback_circuit_breaker_threshold == 5
|
||||
|
||||
|
||||
# IMP-46 u5 — auto-cache opt-in setting default lock.
|
||||
# The CLI flag ``--auto-cache`` in src/phase_z2_pipeline.py mutates this
|
||||
# setting at parse time. The default MUST stay OFF so the dual-gate
|
||||
# contract (visual_check_passed AND user_approved) survives without an
|
||||
# explicit operator opt-in.
|
||||
|
||||
|
||||
def test_ai_fallback_auto_cache_default_off() -> None:
|
||||
s = Settings()
|
||||
assert s.ai_fallback_auto_cache is False, (
|
||||
"IMP-46 u5 auto-cache MUST default OFF; the dual-gate contract "
|
||||
"(visual_check_passed AND user_approved) survives without an "
|
||||
"explicit --auto-cache opt-in."
|
||||
)
|
||||
|
||||
|
||||
# IMP-47B u1 — reject route hint policy correction.
|
||||
# Prior to 2026-05-21 the reject V4 label routed to ``design_reference_only``
|
||||
# (no AI). The user policy correction (issue #76) reroutes reject to
|
||||
# ``ai_adaptation_required`` so the rank-1 reject frame is kept and the AI
|
||||
# re-maps MDX content into its declared slots. Activation remains gated by
|
||||
# ``ai_fallback_enabled`` (default OFF preserves the normal-path AI=0
|
||||
# contract — see test_ai_fallback_master_flag_default_off above).
|
||||
|
||||
|
||||
def test_reject_route_hint_routes_to_ai_adaptation() -> None:
|
||||
from src.phase_z2_pipeline import _IMP05_ROUTE_HINTS, _imp05_route_hint
|
||||
|
||||
assert _IMP05_ROUTE_HINTS["reject"] == "ai_adaptation_required", (
|
||||
"IMP-47B u1: reject must route to ai_adaptation_required so the "
|
||||
"rank-1 reject frame is retained and AI re-maps MDX content into "
|
||||
"its slots (frame auto-swap forbidden)."
|
||||
)
|
||||
assert _imp05_route_hint("reject") == "ai_adaptation_required"
|
||||
# Sibling routes unchanged — guardrail against accidental drift.
|
||||
assert _imp05_route_hint("use_as_is") == "direct_render"
|
||||
assert _imp05_route_hint("light_edit") == "deterministic_minor_adjustment"
|
||||
assert _imp05_route_hint("restructure") == "ai_adaptation_required"
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,137 @@
|
||||
"""IMP-38 U3 regression — call site cleanup (max_rank=3 제거) 후 policy 활성 검증.
|
||||
|
||||
Scenarios:
|
||||
(A) normal case: rank 1~default_max_rank window 에 usable candidate 충분
|
||||
→ effective_max_rank=default_max_rank (rank-3-preserved)
|
||||
→ mdx03 식: rank 1 use_as_is 매칭 정상 case 보호 확인
|
||||
(B) extended case: rank 1~default_max_rank window 에 usable candidate 0
|
||||
→ effective_max_rank=effective_extended_ceiling (rank-extended)
|
||||
→ mdx05-2 식: rank 1~9 미등록/reject + rank 10+ 등록 frame case 처리
|
||||
|
||||
4 round 합의 (#67):
|
||||
- Codex #1: 별 yaml + loader (catalog 오염 방지)
|
||||
- Codex #2: min(configured, len(judgments)) 정정
|
||||
- Codex #6: 2 call site cleanup (HEAD 기준 — IMP-47B 가 추가한 3 번째는 별 axis)
|
||||
- Codex #7: U3 execute ready
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _reset_policy_cache():
|
||||
"""Reset module-level _V4_FALLBACK_POLICY_CACHE for test isolation."""
|
||||
import src.phase_z2_mapper as mapper
|
||||
mapper._V4_FALLBACK_POLICY_CACHE = None
|
||||
yield
|
||||
mapper._V4_FALLBACK_POLICY_CACHE = None
|
||||
|
||||
|
||||
def _make_v4_section(judgments: list[dict]) -> dict:
|
||||
return {"mdx_sections": {"sec-1": {"judgments_full32": judgments}}}
|
||||
|
||||
|
||||
def _judgment(template_id: str, label: str, confidence: float = 0.5, frame_id: int = 0) -> dict:
|
||||
return {
|
||||
"template_id": template_id,
|
||||
"frame_id": frame_id or (hash(template_id) % 10000),
|
||||
"frame_number": 0,
|
||||
"confidence": confidence,
|
||||
"label": label,
|
||||
}
|
||||
|
||||
|
||||
# ─── Scenario A — normal case (rank-3-preserved) ──────────────────
|
||||
|
||||
|
||||
def test_normal_case_with_usable_candidates_preserves_default_max_rank():
|
||||
"""rank 1~3 window 에 usable >= threshold(1) 시 effective_max_rank=default_max_rank(3)."""
|
||||
from src.phase_z2_pipeline import lookup_v4_match_with_fallback
|
||||
from src.phase_z2_mapper import load_frame_contracts
|
||||
|
||||
# mdx03 식 — 첫 rank 가 catalog 등록 + use_as_is/light_edit/restructure(allowed)
|
||||
# 실제 catalog 등록 frame 사용 (catalog hardcode 의존 — 단 frame 32 중 어느 게 등록인지는 yaml 기반)
|
||||
catalog = load_frame_contracts()
|
||||
registered_template_ids = [k for k, v in catalog.items() if isinstance(v, dict)]
|
||||
assert len(registered_template_ids) >= 1, "catalog 등록 frame 1+ 필요 (mdx03 식 fixture)"
|
||||
|
||||
# rank 1 = registered frame + use_as_is (auto-renderable)
|
||||
# rank 2~3 = reject (catalog 등록 무관)
|
||||
first_registered = registered_template_ids[0]
|
||||
judgments = [
|
||||
_judgment(first_registered, "use_as_is", 0.95),
|
||||
_judgment("dummy_rank2", "reject", 0.3),
|
||||
_judgment("dummy_rank3", "reject", 0.2),
|
||||
]
|
||||
v4 = _make_v4_section(judgments)
|
||||
|
||||
_match, trace = lookup_v4_match_with_fallback(v4, "sec-1") # no explicit max_rank → policy
|
||||
assert trace["policy_applied"] == "default_max_rank", (
|
||||
f"normal case 에서 default 유지 기대, got {trace['policy_applied']}"
|
||||
)
|
||||
assert trace["effective_max_rank"] == trace["default_max_rank"]
|
||||
assert trace["usable_count"] >= 1
|
||||
|
||||
|
||||
# ─── Scenario B — extended case (rank-extended) ────────────────────
|
||||
|
||||
|
||||
def test_extended_case_with_no_usable_in_default_window_expands_to_ceiling():
|
||||
"""rank 1~3 window 에 0 usable 시 effective_max_rank=effective_extended_ceiling."""
|
||||
from src.phase_z2_pipeline import lookup_v4_match_with_fallback
|
||||
|
||||
# mdx05-2 식 — rank 1~3 미등록 (template_id 가 catalog 에 없음) + reject 라벨
|
||||
# rank 4~ 도 등록 안 됨 (fixture 단순화)
|
||||
# 다만 judgments_count=10 으로 충분 → effective_extended_ceiling = min(extended, 10) = 10
|
||||
judgments = [
|
||||
_judgment(f"unregistered_t{i}", "reject", 0.1 + i * 0.01) for i in range(10)
|
||||
]
|
||||
v4 = _make_v4_section(judgments)
|
||||
|
||||
_match, trace = lookup_v4_match_with_fallback(v4, "sec-1")
|
||||
assert trace["policy_applied"] == "extended_max_rank", (
|
||||
f"extended case 기대, got {trace['policy_applied']}"
|
||||
)
|
||||
assert trace["usable_count"] == 0
|
||||
assert trace["judgments_count"] == 10
|
||||
# Codex #2 정정: min(configured, 10) — configured 32 면 10, 5 면 5
|
||||
assert trace["effective_extended_ceiling"] == min(
|
||||
trace["configured_extended_max_rank"], 10
|
||||
)
|
||||
assert trace["effective_max_rank"] == trace["effective_extended_ceiling"]
|
||||
|
||||
|
||||
# ─── Scenario C — call site cleanup byte-identical (caller_override 제거 후 policy 활성) ─
|
||||
|
||||
|
||||
def test_default_call_site_now_uses_policy_after_cleanup():
|
||||
"""U3 cleanup 후 call site = no explicit max_rank → policy path 자동 활성.
|
||||
|
||||
이전: caller 가 max_rank=3 명시 → policy_applied=caller_override
|
||||
U3 후: caller 가 명시 X → policy_applied=default_max_rank (usable >= 1 시) or extended_max_rank
|
||||
"""
|
||||
from src.phase_z2_pipeline import lookup_v4_match_with_fallback
|
||||
judgments = [_judgment(f"unregistered_t{i}", "reject") for i in range(5)]
|
||||
v4 = _make_v4_section(judgments)
|
||||
|
||||
# caller 가 max_rank 명시 X (U3 cleanup 후 production caller 의 새 동작)
|
||||
_match, trace = lookup_v4_match_with_fallback(v4, "sec-1")
|
||||
assert trace["policy_applied"] in {"default_max_rank", "extended_max_rank"}
|
||||
assert trace["policy_applied"] != "caller_override", (
|
||||
"U3 cleanup 후 production caller = no explicit, policy path 활성 기대"
|
||||
)
|
||||
|
||||
|
||||
# ─── Scenario D — explicit caller_override 여전히 동작 (test path 보호) ────
|
||||
|
||||
|
||||
def test_explicit_caller_override_still_works_for_tests():
|
||||
"""test 에서 explicit max_rank=N 보낼 시 caller_override 그대로 동작 (backward compat)."""
|
||||
from src.phase_z2_pipeline import lookup_v4_match_with_fallback
|
||||
judgments = [_judgment(f"unregistered_t{i}", "reject") for i in range(10)]
|
||||
v4 = _make_v4_section(judgments)
|
||||
|
||||
_match, trace = lookup_v4_match_with_fallback(v4, "sec-1", max_rank=5)
|
||||
assert trace["policy_applied"] == "caller_override"
|
||||
assert trace["effective_max_rank"] == 5
|
||||
@@ -237,10 +237,10 @@ def test_restructure_reject_preserved_as_non_direct_evidence(patch_selector_deps
|
||||
by_rank = {c["rank"]: c for c in candidates}
|
||||
assert set(by_rank.keys()) == {1, 2, 3}
|
||||
|
||||
# rank-1 reject — non-direct, design_reference_only
|
||||
# rank-1 reject — non-direct, ai_adaptation_required (IMP-47B u1 policy correction)
|
||||
assert by_rank[1]["v4_label"] == "reject"
|
||||
assert by_rank[1]["filtered_for_direct_execution"] is True
|
||||
assert by_rank[1]["route_hint"] == "design_reference_only"
|
||||
assert by_rank[1]["route_hint"] == "ai_adaptation_required"
|
||||
|
||||
# rank-2 restructure — non-direct, ai_adaptation_required
|
||||
assert by_rank[2]["v4_label"] == "restructure"
|
||||
@@ -295,25 +295,79 @@ def test_existing_trace_shape_does_not_regress(patch_selector_deps):
|
||||
assert trace["selection_path"] == "rank_1"
|
||||
|
||||
|
||||
# ─── Case 7 : Step 9 production-source guard (Codex #20 blocker fix) ───
|
||||
# ─── Case 7 : Step 9 helper-call shape test (IMP-32 u5 — replaces source guard) ───
|
||||
|
||||
|
||||
def test_step9_production_emits_candidate_evidence_and_alias():
|
||||
"""Temporary production-source guard for IMP-05 Step 9 evidence fields.
|
||||
def test_build_application_plan_unit_emits_candidate_evidence_and_alias():
|
||||
"""IMP-32 u5 — direct helper-call shape test for Step 9 evidence fields.
|
||||
|
||||
Step 9 application-plan unit assembly is currently inline, so this test
|
||||
checks the exact production assignments until IMP-32 extracts a helper.
|
||||
Once that helper exists, replace this source-string guard with a direct
|
||||
helper-call test.
|
||||
Replaces the IMP-05 Case 7 `inspect.getsource(phase_z2_pipeline)` literal
|
||||
guard (introduced at commit `23d1b25` while Step 9 unit assembly was
|
||||
inline) with a direct call to `_build_application_plan_unit`, the helper
|
||||
extracted in IMP-32 u3. Verification axes preserved:
|
||||
|
||||
- candidate_evidence list identity sourced from `selection_trace["candidates"]`
|
||||
- fallback_chain compat-alias identity (same list object as candidate_evidence)
|
||||
- key order: candidate_evidence before fallback_chain
|
||||
- compat-alias comment preserved on the helper's fallback_chain line
|
||||
"""
|
||||
source = inspect.getsource(phase_z2_pipeline)
|
||||
candidate_line = '"candidate_evidence": selection_trace.get("candidates", [])'
|
||||
alias_line = '"fallback_chain": selection_trace.get("candidates", [])'
|
||||
from types import SimpleNamespace
|
||||
|
||||
assert candidate_line in source
|
||||
assert alias_line in source
|
||||
assert source.index(candidate_line) < source.index(alias_line)
|
||||
assert "compat alias; prefer candidate_evidence" in source
|
||||
from src.phase_z2_pipeline import _build_application_plan_unit
|
||||
|
||||
candidates_list = [
|
||||
{"rank": 1, "template_id": "MOCK_template_direct_a", "label": "use_as_is"},
|
||||
]
|
||||
selection_trace = {"candidates": candidates_list}
|
||||
|
||||
# Synthetic CompositionUnit-shape duck-typed input — matches V4Match attrs
|
||||
# used inside the helper (template_id / frame_id / frame_number / v4_rank /
|
||||
# confidence / label per src/phase_z2_pipeline.py V4Match dataclass).
|
||||
v4_candidate = SimpleNamespace(
|
||||
template_id="MOCK_template_direct_a",
|
||||
frame_id="MOCK_frame_001",
|
||||
frame_number=1,
|
||||
v4_rank=1,
|
||||
confidence=0.9,
|
||||
label="use_as_is",
|
||||
)
|
||||
unit = SimpleNamespace(
|
||||
source_section_ids=["S1"],
|
||||
v4_candidates=[v4_candidate],
|
||||
v4_rank=1,
|
||||
selection_path="rank_1",
|
||||
fallback_reason=None,
|
||||
frame_template_id="MOCK_template_direct_a",
|
||||
)
|
||||
|
||||
result = _build_application_plan_unit(
|
||||
unit=unit,
|
||||
zone_plan={},
|
||||
selection_trace=selection_trace,
|
||||
plan_record=None,
|
||||
v4_all_for_unit=[],
|
||||
layout_preset="Type A",
|
||||
layout_candidates_list=[],
|
||||
)
|
||||
|
||||
# IMP-05 L2 — candidate_evidence is the primary field, identity-bound to
|
||||
# selection_trace["candidates"] (not a copy).
|
||||
assert "candidate_evidence" in result
|
||||
assert result["candidate_evidence"] is candidates_list
|
||||
|
||||
# compat alias — fallback_chain references the SAME list object as
|
||||
# candidate_evidence (verified by `is` identity, not equality).
|
||||
assert "fallback_chain" in result
|
||||
assert result["fallback_chain"] is candidates_list
|
||||
|
||||
# key order — candidate_evidence MUST precede fallback_chain in the
|
||||
# returned dict to preserve documented L2 ordering.
|
||||
keys = list(result.keys())
|
||||
assert keys.index("candidate_evidence") < keys.index("fallback_chain")
|
||||
|
||||
# compat-alias comment preserved on the helper's fallback_chain line.
|
||||
helper_source = inspect.getsource(_build_application_plan_unit)
|
||||
assert "compat alias; prefer candidate_evidence" in helper_source
|
||||
|
||||
|
||||
# ─── Case 8 : Step 20 slide-status qualifier fields presence + defensive default
|
||||
@@ -380,3 +434,125 @@ def test_step20_slide_status_qualifier_fields_present_with_defensive_defaults():
|
||||
# Defensive defaults — 0 + [] when summary present but empty
|
||||
assert status_c["fallback_selection_count"] == 0
|
||||
assert status_c["selection_paths"] == []
|
||||
|
||||
|
||||
# ─── Case 9 : IMP-30 u1 — opt-in provisional synthesis on chain_exhausted ───
|
||||
|
||||
|
||||
def test_allow_provisional_default_off_preserves_imp05_behavior(patch_selector_deps):
|
||||
"""IMP-30 u1 — default ``allow_provisional=False`` keeps chain_exhausted
|
||||
returning ``(None, trace)`` exactly as IMP-05 specified. Regression guard
|
||||
for IMP-05 close commit 23d1b25.
|
||||
"""
|
||||
v4 = _make_v4([
|
||||
_j(1, "MOCK_template_restructure_a", "MOCK_frame_001", "restructure"),
|
||||
_j(2, "MOCK_template_reject_a", "MOCK_frame_002", "reject"),
|
||||
])
|
||||
|
||||
match, trace = lookup_v4_match_with_fallback(
|
||||
v4, "S1", raw_content="- a\n- b\n- c\n"
|
||||
)
|
||||
|
||||
assert match is None
|
||||
assert trace["selection_path"] == "chain_exhausted"
|
||||
assert trace.get("provisional") is None
|
||||
assert trace["selected_rank"] is None
|
||||
assert trace["selected_template_id"] is None
|
||||
|
||||
|
||||
def test_allow_provisional_synthesizes_rank_1_on_chain_exhausted(patch_selector_deps):
|
||||
"""IMP-30 u1 — opt-in ``allow_provisional=True`` synthesizes a provisional
|
||||
rank-1 match when the rank-1..3 chain is exhausted (all restructure/reject).
|
||||
Downstream first-render invariant uses this to render a "needs adaptation"
|
||||
zone instead of aborting.
|
||||
"""
|
||||
v4 = _make_v4([
|
||||
_j(1, "MOCK_template_restructure_a", "MOCK_frame_001", "restructure"),
|
||||
_j(2, "MOCK_template_reject_a", "MOCK_frame_002", "reject"),
|
||||
])
|
||||
|
||||
match, trace = lookup_v4_match_with_fallback(
|
||||
v4, "S1", raw_content="- a\n- b\n- c\n",
|
||||
allow_provisional=True,
|
||||
)
|
||||
|
||||
# Provisional rank-1 synthesized from the rank-1 judgment
|
||||
assert match is not None
|
||||
assert match.provisional is True
|
||||
assert match.template_id == "MOCK_template_restructure_a"
|
||||
assert match.frame_id == "MOCK_frame_001"
|
||||
assert match.label == "restructure"
|
||||
assert match.v4_rank == 1
|
||||
assert match.selection_path == "provisional_rank_1"
|
||||
# fallback_reason mirrors the chain-exhaust reason
|
||||
assert match.fallback_reason is not None
|
||||
assert "phase_z_status_not_allowed" in match.fallback_reason
|
||||
|
||||
# Top-level trace mirrors reflect provisional selection
|
||||
assert trace["selection_path"] == "provisional_rank_1"
|
||||
assert trace["selected_rank"] == 1
|
||||
assert trace["selected_template_id"] == "MOCK_template_restructure_a"
|
||||
assert trace["selected_frame_id"] == "MOCK_frame_001"
|
||||
assert trace["selected_label"] == "restructure"
|
||||
assert trace["fallback_used"] is True
|
||||
assert trace["provisional"] is True
|
||||
|
||||
# Original candidate skip reasons are preserved (not rewritten by synthesis)
|
||||
by_rank = {c["rank"]: c for c in trace["candidates"]}
|
||||
assert by_rank[1]["decision"] == "skipped"
|
||||
assert by_rank[1]["reason"] == "phase_z_status_not_allowed:extract_matched_zone"
|
||||
assert by_rank[2]["decision"] == "skipped"
|
||||
assert by_rank[2]["reason"] == "phase_z_status_not_allowed:fallback_candidate"
|
||||
|
||||
|
||||
def test_allow_provisional_no_op_when_normal_selection_succeeds(patch_selector_deps):
|
||||
"""IMP-30 u1 — ``allow_provisional=True`` is a no-op when normal selection
|
||||
succeeds. The rank-1 (or rank-N fallback) result MUST be non-provisional.
|
||||
"""
|
||||
v4 = _make_v4([
|
||||
_j(1, "MOCK_template_direct_a", "MOCK_frame_001", "use_as_is"),
|
||||
])
|
||||
|
||||
match, trace = lookup_v4_match_with_fallback(
|
||||
v4, "S1", raw_content="- a\n- b\n- c\n",
|
||||
allow_provisional=True,
|
||||
)
|
||||
|
||||
assert match is not None
|
||||
assert match.provisional is False
|
||||
assert match.selection_path == "rank_1"
|
||||
assert trace["selection_path"] == "rank_1"
|
||||
assert trace.get("provisional") is None
|
||||
|
||||
|
||||
def test_allow_provisional_no_op_when_no_v4_section(patch_selector_deps):
|
||||
"""IMP-30 u1 — when no V4 section is resolved (no rank-1 judgment to
|
||||
synthesize from), ``allow_provisional=True`` MUST still return
|
||||
``(None, trace)``. u3/u4 handle this case with a placeholder zone or
|
||||
empty-shell terminal slide.
|
||||
"""
|
||||
v4 = {"mdx_sections": {}} # no section at all
|
||||
|
||||
match, trace = lookup_v4_match_with_fallback(
|
||||
v4, "S1", raw_content="- a\n- b\n- c\n",
|
||||
allow_provisional=True,
|
||||
)
|
||||
|
||||
assert match is None
|
||||
assert trace["fallback_reason"] == "no_v4_section"
|
||||
|
||||
|
||||
def test_allow_provisional_no_op_when_empty_judgments(patch_selector_deps):
|
||||
"""IMP-30 u1 — when the V4 section exists but ``judgments_full32`` is
|
||||
empty, ``allow_provisional=True`` MUST still return ``(None, trace)``.
|
||||
No synthetic rank-1 can be fabricated from nothing.
|
||||
"""
|
||||
v4 = {"mdx_sections": {"S1": {"judgments_full32": []}}}
|
||||
|
||||
match, trace = lookup_v4_match_with_fallback(
|
||||
v4, "S1", raw_content="- a\n- b\n- c\n",
|
||||
allow_provisional=True,
|
||||
)
|
||||
|
||||
assert match is None
|
||||
assert trace["fallback_reason"] == "empty_v4_judgments"
|
||||
|
||||
@@ -0,0 +1,109 @@
|
||||
"""IMP-38 U1 — v4_fallback_policy.yaml loader test.
|
||||
|
||||
Verify:
|
||||
- load_v4_fallback_policy() returns dict with expected keys
|
||||
- yaml parsed correctly (usable_threshold, default_max_rank, extended_max_rank, policy_type)
|
||||
- graceful fallback when yaml missing → _V4_FALLBACK_POLICY_DEFAULT
|
||||
- _V4_FALLBACK_POLICY_CACHE pattern (lazy load, mirror of _CATALOG_CACHE)
|
||||
- load_frame_contracts() shape unchanged (separate yaml, catalog 오염 X)
|
||||
|
||||
4 round 합의 (#67):
|
||||
- Codex #1: separate yaml (not frame_contracts.yaml top-level)
|
||||
- Codex #3: load_frame_contracts() shape 변경 X
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
PROJECT_ROOT = Path(__file__).parent.parent
|
||||
V4_POLICY_PATH = PROJECT_ROOT / "templates" / "phase_z2" / "catalog" / "v4_fallback_policy.yaml"
|
||||
CATALOG_PATH = PROJECT_ROOT / "templates" / "phase_z2" / "catalog" / "frame_contracts.yaml"
|
||||
|
||||
|
||||
def _reset_caches():
|
||||
"""Reset module-level caches for test isolation."""
|
||||
import src.phase_z2_mapper as mapper
|
||||
mapper._V4_FALLBACK_POLICY_CACHE = None
|
||||
mapper._CATALOG_CACHE = None
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def clean_caches():
|
||||
_reset_caches()
|
||||
yield
|
||||
_reset_caches()
|
||||
|
||||
|
||||
def test_v4_fallback_policy_yaml_exists():
|
||||
"""IMP-38 U1 — separate yaml file must exist."""
|
||||
assert V4_POLICY_PATH.exists(), (
|
||||
f"v4_fallback_policy.yaml not found at {V4_POLICY_PATH}. "
|
||||
"IMP-38 U1 expects separate yaml (Codex #1 corr — not frame_contracts.yaml top-level)."
|
||||
)
|
||||
|
||||
|
||||
def test_load_v4_fallback_policy_returns_dict_with_expected_keys():
|
||||
"""load_v4_fallback_policy() must return dict with policy keys."""
|
||||
from src.phase_z2_mapper import load_v4_fallback_policy
|
||||
policy = load_v4_fallback_policy()
|
||||
assert isinstance(policy, dict)
|
||||
expected_keys = {"policy_type", "usable_threshold", "default_max_rank", "extended_max_rank"}
|
||||
missing = expected_keys - set(policy.keys())
|
||||
assert not missing, f"missing keys in v4_fallback_policy: {missing}"
|
||||
|
||||
|
||||
def test_load_v4_fallback_policy_values_match_yaml():
|
||||
"""Loaded policy values must match v4_fallback_policy.yaml (initial commit)."""
|
||||
from src.phase_z2_mapper import load_v4_fallback_policy
|
||||
policy = load_v4_fallback_policy()
|
||||
assert policy["policy_type"] == "dynamic_usable_count_based"
|
||||
assert policy["usable_threshold"] == 1
|
||||
assert policy["default_max_rank"] == 3
|
||||
assert policy["extended_max_rank"] == 32
|
||||
|
||||
|
||||
def test_load_v4_fallback_policy_cache_pattern():
|
||||
"""_V4_FALLBACK_POLICY_CACHE pattern — second call returns same dict (lazy load)."""
|
||||
from src.phase_z2_mapper import load_v4_fallback_policy
|
||||
policy_a = load_v4_fallback_policy()
|
||||
policy_b = load_v4_fallback_policy()
|
||||
assert policy_a is policy_b, "cache pattern violated (should return same dict instance)"
|
||||
|
||||
|
||||
def test_load_v4_fallback_policy_graceful_when_yaml_missing():
|
||||
"""yaml 파일 없을 시 → _V4_FALLBACK_POLICY_DEFAULT (extended_max_rank=3, byte-identical pre-IMP-38)."""
|
||||
import src.phase_z2_mapper as mapper
|
||||
with patch.object(mapper, "V4_FALLBACK_POLICY_PATH", PROJECT_ROOT / "tests" / "__nonexistent_policy.yaml"):
|
||||
# reset cache to force reload via patched path
|
||||
mapper._V4_FALLBACK_POLICY_CACHE = None
|
||||
policy = mapper.load_v4_fallback_policy()
|
||||
assert policy["default_max_rank"] == 3
|
||||
assert policy["extended_max_rank"] == 3, (
|
||||
"graceful fallback must keep extended==default (byte-identical pre-IMP-38)"
|
||||
)
|
||||
|
||||
|
||||
def test_load_frame_contracts_shape_unchanged():
|
||||
"""Codex #3 LOCK — load_frame_contracts() must still return template_id → entry dict."""
|
||||
from src.phase_z2_mapper import load_frame_contracts, load_v4_fallback_policy
|
||||
catalog = load_frame_contracts()
|
||||
policy = load_v4_fallback_policy()
|
||||
|
||||
# catalog 의 key 가 모두 frame entry (dict with template_id/frame_id) 여야 함
|
||||
for key, entry in catalog.items():
|
||||
assert isinstance(entry, dict), f"catalog entry {key} should be dict"
|
||||
assert "template_id" in entry, f"catalog entry {key} missing template_id (policy bleed?)"
|
||||
|
||||
# policy keys 는 catalog 에 안 들어감
|
||||
policy_keys = {"policy_type", "usable_threshold", "default_max_rank", "extended_max_rank"}
|
||||
catalog_top_keys = set(catalog.keys())
|
||||
bleed = policy_keys & catalog_top_keys
|
||||
assert not bleed, (
|
||||
f"policy keys leaked into frame_contracts.yaml: {bleed}. "
|
||||
"Codex #1 corr violated — policy must stay in separate v4_fallback_policy.yaml."
|
||||
)
|
||||
Reference in New Issue
Block a user