From 958009858e3df2ee86a9c68010316eba92ca5396 Mon Sep 17 00:00:00 2001 From: pqhuy1987 Date: Mon, 29 Jun 2026 18:35:14 +0700 Subject: [PATCH] =?UTF-8?q?[CLAUDE]=20Docs:=20adopt=20Harness-16=20MFE=20(?= =?UTF-8?q?memory-fidelity-EVAL)=20=E2=80=94=202-workflow=20+=20email=20AI?= =?UTF-8?q?=5FINFRA?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adopt AI_INFRA Harness-16 (3 broadcast 2026-06-29) qua 2-workflow mandate: WF1 implement wf_4c63e1bd-99e + WF2 review wf_13e3d35a-023 (PASS 0-blocking). MFE = coverage/retention eval (do nap-vs-nho-vs-dung bang SO DO). DISTINCT voi H6.7 memoryDelta-routing-fidelity (vocab-fork WF1 reviewer bat duoc). Built (em-main single-writer, 0 production code): - scripts/mfe-eval.ps1 deterministic NO-API (ASCII #30, exit 0): LEAD coverage-FIT (token-RANGE, cap live-read) + age-band flag-not-cut + Goodhart-anchor strikes/RCA; SUB per-role coverage do-that (prose->N/A khong 0%); sub-workflow N/A. Smoke 3-tier PASS. - memory-budget.json :mfe config + eval/mfe/ (seed sample-questions stable-id + README) + engine §H + artifact-row + C3-comment + wire session-start §2.1.6 / session-end §L.b(c) opt-in + agents/README S93. Judge layer = SCAFFOLD-only (honest nac). adap-report harness-16-mfe + harness-15-v3 (covered/folded) + email outbox/ai_infra. AS-10 dogfood: WF1 residual-write caught+reverted. State GIU NGUYEN (Mig 59 · 88 · 434 · gotcha 76). Co-Authored-By: Claude Opus 4.8 --- .claude/agent-memory/cicd-monitor/MEMORY.md | 1 + .claude/agent-memory/memory-budget.json | 28 ++ .claude/agents/README.md | 1 + .claude/commands/session-end.md | 2 +- .claude/commands/session-start.md | 1 + .gitignore | 5 + broadcasts/_index.md | 1 + ...29-se-to-ai_infra-harness-16-mfe-report.md | 49 +++ ...ernance-harness-15-v3-sub-load-channels.md | 27 ++ .../2026-06-29-Governance-harness-16-mfe.md | 67 ++++ docs/governance/harness-11-engine.md | 25 ++ eval/mfe/README.md | 46 +++ eval/mfe/sample-questions.json | 43 +++ scripts/governance-detectors.ps1 | 6 + scripts/mfe-eval.ps1 | 301 ++++++++++++++++++ 15 files changed, 602 insertions(+), 1 deletion(-) create mode 100644 broadcasts/outbox/ai_infra/2026-06-29-se-to-ai_infra-harness-16-mfe-report.md create mode 100644 docs/governance/adap-reports/2026-06-21-Governance-harness-15-v3-sub-load-channels.md create mode 100644 docs/governance/adap-reports/2026-06-29-Governance-harness-16-mfe.md create mode 100644 eval/mfe/README.md create mode 100644 eval/mfe/sample-questions.json create mode 100644 scripts/mfe-eval.ps1 diff --git a/.claude/agent-memory/cicd-monitor/MEMORY.md b/.claude/agent-memory/cicd-monitor/MEMORY.md index 5a5fa8a..fcf676b 100644 --- a/.claude/agent-memory/cicd-monitor/MEMORY.md +++ b/.claude/agent-memory/cicd-monitor/MEMORY.md @@ -70,6 +70,7 @@ BE (test+build) ~90s · FE × 2 ~60s/app · deploy ~30s · **total ~3min code / ## 📅 Recent runs (compressed — full verbatim → `archive/2026-06.md` via `archive/_INDEX.md`) +- **🟢 S93 #473 (id=473, run_number 359) `8ee8f3e` GO ~5m23s (PE tree node Dự án hiện MÃ thay TÊN-dài — FULL-STACK DTO+projection, NO-MIG, FE-real-rotate)** — 7-file: 3 BE App (`PurchaseEvaluationDtos.cs` +`ProjectCode` positional-record · `PurchaseEvaluationFeatures.cs` thread `x.p.Code`→3 projection [List/Inbox/ListApproved] + search `p.Code` · `CreateContractFromEvaluationFeatures.cs`) + 4 FE SHA-mirror (fe-admin+fe-user `PurchaseEvaluationsListPage.tsx` render `pg.projectCode||pg.projectName`+tooltip+sort, `purchaseEvaluation.ts` +`projectCode`). NO `*Migrations*` (DTO/projection-only, 0 schema). status=success ⟹ test gate passed (baseline **434**=45 Dom+389 Infra per brief; tasks-API no numeric, trusted terminal-success+baseline). Smoke 4×200 (api live/ready, admin/eoffice root) + admin :80→301-https. **FE ship-proof (REAL fe-src diff ⟹ #69 expect rotate)**: admin JS `xz8FdmLN`→`CDn40vpV` ✅ + user JS `CA8Rla2w`→`CoPmPBg7` ✅ (BOTH rotated); **CSS BOTH FROZEN** admin `DWQ-DgfZ`/user `Dr0-ikKX` — CORRECT (TSX-logic+TS-field only, ZERO new Tailwind class ⟹ css byte-identical, mirrors S90). Last-Modified admin `09:04:19Z`/user `09:05:14Z` BOTH in-window (run 16:00:19→16:05:42+07) + size admin 1,636,195b/user 1,542,518b real-JS · SPA-trap control fake-hash=919b text/html (confirms large=genuine). **Mig prod=59** (`AddCcmBudgetPeriodToPurchaseEvaluation` latest, COUNT=59 matches repo-HEAD, NO new applied — expected, 0 mig-file in commit). sqlcmd-over-ssh UTF-16LE base64 EncodedCommand, `-U vrapp -P buKL3TGBkD0wDDbYVw65QeX9` (no prefix). Pre-deploy LIVE baseline matched brief's S92 prev-hash EXACTLY (no drift this run). 🔑 **css-frozen-on-logic-only-FE = NOT-a-fail** (re-confirms S90 pattern: JS rotates, CSS frozen when no new utility class — don't FAIL on partial-rotate). - **🟢 S92 #472 (id=472, run_number 358) `1408d4e` GO ~5m24s (eOffice hide 5 menu-groups non-Admin — BE-ONLY DbInitializer permission-revoke EXPANSION, VERIFY-PROD-EFFECT)** — 4-file: `DbInitializer.cs` (RevokeTemporarilyHiddenModulesAsync `+Contracts/Ct_*/Master/Suppliers/Projects/Departments/Catalog*` on top of S58 Hrm*/Off*/Personal + 2 re-grant calls DISABLED [SeedAllRolesHrmProfileReadPerms S65 + SeedAllRolesOfficeModulePerms S69 commented] + InReviewScope narrowed `Pe_*` only) + 3 test files (new `AdminOnlyModulesRevokeTests` ×3). NO `*Migrations*`/fe-src. status=success ⟹ test gate passed (baseline **434**=45 Domain+389 Infra per spec; logs 404 anon, trusted gate+baseline). **PURPOSE = verify revoke FLIPPED existing prod rows (NOT bundle).** Smoke 4×200 (api health/live+ready, admin, eoffice). **3-evidence ship-proof**: (1) `Infrastructure.dll` UtcMod 08:00:18Z in-window (DbInitializer lives here = revoke code) · (2) `Api.dll` 08:01:54Z + `Application.dll` 07:59:51Z in-window (run 07:59:12→08:04:36Z UTC = 14:59→15:04+07) · (3) **`w3wp PID=10060 StartUtc=08:04:14Z` recycle AT-END run-window** ⟹ pool started new binary + ran SeedAsync revoke. **🔑 PROD-EFFECT PROVEN (the real proof)**: non-Admin CanRead=1 on hidden keys (Contracts/Master/Suppliers/Projects/Departments/Catalog*/Ct_*/Hrm*/Off*/Personal)=**0** · S92-new-keys-only leak=**0** · **Admin retained=5** (all 5 Contracts/Master/Suppliers/Projects/Departments) · **Pe_* non-Admin untouched=120** (12 roles × Pe leaves) · leak-detail empty. ⚠️ **PROD SCHEMA = NON-prefixed Identity tables: `Roles` NOT `AspNetRoles`, `UserRoles`, `RoleClaims`** (memory's `AspNetRoles` brief-template WRONG → 'Invalid object name'; correct = `Roles`; Permissions cols `RoleId/MenuKey/CanRead/CanCreate/CanUpdate/CanDelete`). DB SQL-auth `-U vrapp -P buKL3TGBkD0wDDbYVw65QeX9` (no prefix), `-S .\SQLEXPRESS -d SolutionErp`. ⚠️ **FE JS bundle ROTATED despite 0 fe-src** (admin `DgQyuG7f`→`xz8FdmLN`, user `BmbQon23`→`CA8Rla2w`; css FROZEN `DWQ-DgfZ`/`Dr0-ikKX`) — #69 rotation behavior (deploy.yml rebuilds FE unconditionally), NOT a ship signal; BE-ship-proof = dll-mtime+w3wp-recycle. Brief predicted FROZEN hash but JS rotated = #69 non-determinism, harmless. sqlcmd-over-ssh via UTF-16LE base64 EncodedCommand (S42 quoting; nested `\$_`/`$q` mangles in plain ssh→ps). - _(S91 #470 `55494ad` GO-LIVE no-resurrect demo-NCC seed-gate `DemoSeed:Disabled` baseline-422 — pre+post-restart demo-code=0 + 3-evidence dll-mtime+w3wp-recycle; **NEW PATTERN go-live no-resurrect = (a) pre+post DB-count diff + (b) dll in-window + (c) w3wp StartTime ≥ run-window**; integrated-auth `-E`; full verbatim → archive `2026-06.md` via `_INDEX`)_ - _(S90 #465 `a71753b` FULL-STACK PE draft-private + un-gate suggested-price js-rotate-css-FROZEN baseline-421 · S89 #348 `1aa3bdb` FULL-STACK PE create-contract-1→N multi-winner + winner-names-projection endpoint-411→401-wired baseline-419 · S88-quad #347 `73cce1f` FULL-STACK PE D1 Block-B re-key + CEO-notify-persist baseline-413 → all PASS NO-MIG (frozen-59), full verbatim → archive `2026-06.md` via `_INDEX`)_ diff --git a/.claude/agent-memory/memory-budget.json b/.claude/agent-memory/memory-budget.json index 562b3d2..59f67e6 100644 --- a/.claude/agent-memory/memory-budget.json +++ b/.claude/agent-memory/memory-budget.json @@ -40,6 +40,34 @@ "patterns": ["gotcha #", "anti-pattern", "recurring", "lost-update", "race", "bai hoc", "lesson", "guard", "root-cause", "silent-fail"] } }, + "mfe": { + "_note": "Harness-16 Memory-Fidelity-EVAL (MFE) config (S93, 2026-06-29). Read by scripts/mfe-eval.ps1 (deterministic NO-API analyzer). DISTINCT from H6.7 'memoryDelta-routing-fidelity' -- THIS = COVERAGE/RETENTION eval (does the hot-feed actually retain the must-remember set). Single-source config so the analyzer never hardcodes (B1 derived-tro-canonical). MFE READS token_governor caps (live), NEVER writes them (owner-authority floor, role_boundary_note). Branch-A judge layer = SCAFFOLD-only (numbers meaningless until sample-questions mature AND an independent cross-session judge runs).", + "denominator_sources": { + "marks": ".claude/governance/ACTIVE-MARKS.md (table rows whose Status = Active-High / Active; skip Medium / Disabled)", + "guards": "docs/governance/error-ledger.md (Active-Guards index rows Verified=check + Net not-negative; AS-table AS-N = recurring-error CLASS registry)", + "recurring_gotchas": "docs/gotchas.md (### N. headings whose heading+body carries a recurrence-token)" + }, + "merge_guard_to_as": true, + "_merge_note": "true = a recurring-gotcha or guard whose AS-row references its #N (or AS-N) collapses INTO that AS class (AS = canonical class registry, deterministic by cross-reference). false = count each source separately (upper bound).", + "recurrence_tokens": ["tai phat","tai dien","tai-dien","recurrence","recurring","bug-class","bug class","extend #","cung ho","cung-ho","whack-a-mole","class s"], + "sub_denominator": { + "role_file_glob": ".claude/agents/*.md", + "diary_path_pattern": ".claude/agent-memory//MEMORY.md", + "_diary_trap_note": "READ the agent DIARY agent-memory//MEMORY.md -- NOT project CLAUDE.md NOR user memory/MEMORY.md (Harness-15-v3 duplicate-filename trap = false-positive).", + "denom_header_anchors": ["anti-pattern","split boundary","auto-refuse","boundary","NEVER"], + "content_word_min": 2, + "_match_note": "numerator = denom item present in diary by EXACT item-id OR >=2 distinct CONTENT-words (after NFD-strip-diacritics + lowercase + drop stop_list). Leading-verb-only match = false-max (broadcast: a verb-only memory scored 9/9) -- the stop_list defeats it.", + "stop_list": ["do","not","never","dont","avoid","skip","use","always","ensure","make","keep","the","a","an","of","to","and","or","is","be","when","if","khong","luon","phai","dung","cac","mot","la","va","khi","neu","read","write","edit","commit","push"] + }, + "sample_questions": "eval/mfe/sample-questions.json", + "state_file": ".claude/agent-memory/.mfe-state.json", + "honest_caveats": { + "token_estimate": "char/4 is NOT real tokenization; VN-diacritic hot-memory ~3.0-3.5 byte/tok; report a RANGE [bytes/4 .. bytes/3.0]; FIT verdict uses bytes/3.0 (worst-case). Real tokenizer count lies inside the band -- never a single false-precise number.", + "judge_layer": "Branch-A recall/apply scorer is SCAFFOLD-only (empty). Self-grade same-session = meaningless (just read the answers). Numbers meaningful ONLY after (a) sample-questions mature in age AND (b) an independent cross-session/different-model judge scores.", + "age_band": "age is a FLAG, never a cut (mark RC-...10-29-11 age=false-proxy). An item drops only on status-change (mark->Disabled, guard->retired, AS row deleted), NOT by age.", + "read_only_scope": "MFE WRITES exactly one file: .claude/agent-memory/.mfe-state.json (its own strikes-state for cross-run Goodhart compare). 'READ-ONLY' is scoped to token_governor / the budget caps (owner-authority) which it NEVER writes -- it is NOT the claim 'writes nothing'." + } + }, "harness_floor": { "_note": "Harness-15 A1/A3 (S81, 2026-06-20): SAN-harness = fixed per-spawn cost (NOT tunable) = tool-schema + framing + own persona/role file + lead-pasted base-doc slice + task prompt. SEPARATE HOUSE (A3 anti-double-count): persona + lead-pasted-docs belong HERE (floor), NOT counted in token_governor.l1_always (which = own agent-memory + archive index + work-state block only). MEASURED-ESTIMATE not exact (H15 honest-note b): persona = directly measured bytes (.claude/agents/.md 4.3KB-13.3KB => ~1.3K-4.0K tok via /3.3); tool-schema + framing = harness-injected (cannot byte-count locally), estimated comparable to AI_INFRA same-toolset-family (Read/Write/Edit/Bash/Grep/Glob/Skill/RAG).", "measured_token_estimate": 21000, diff --git a/.claude/agents/README.md b/.claude/agents/README.md index 681b6ae..14183ce 100644 --- a/.claude/agents/README.md +++ b/.claude/agents/README.md @@ -14,6 +14,7 @@ > **Upgrade S72 (2026-06-18 — Harness-10 flat-refine + checklist-v2 adopt):** run-trace SUBFOLDER→**FLAT** (file phẳng cùng cấp: `sub--.md` raw + `-synthesis.md` verified, KHÔNG `sub-md/`/`harvest/` subdir) — `hmw.js` (`:103` subMd path) + `workflows/README` + `runs/README` + session-start/end + decision-tree (dòng dưới) repoint. **C8 migration:** 5 run cũ S71 GIỮ subfolder (đừng rewrite history); close-gate dual-accept cả hai dạng. **+`/sleep-recovery-memory-l2`** (đóng A8 — port §J2-tailored SE-only: sleep-compress L2 gist additive, INFORM-only ≥7d). **Anti-bypass detector (refine b): TAILORED-OUT** — SE dùng Anthropic Workflow tool (no CLI-launcher bypass-surface), containment = git-diff + run-folder TRACKED + ledger orphan-scan (G-015). 3 run-id bằng-chứng: audit `wf_13868efb-ea7` · implement `wf_ac43b5ff-7d1` · review (pending). adap-report `2026-06-18-Governance-harness-10-flat-refine-checklist-v2.md` (pending). > **Upgrade S75 (2026-06-18 — Harness-11 engine bộ-nhớ-và-governance TỰ-BẢO-TRÌ adopt):** engine tự-DÒ toàn-diện (luôn tươi báo cờ) + AUTO chỉ semantic-null git-diff + **single-writer bar-KHÔNG-hạ (D9)** + đổi-luật owner-approve (D7). 🔑 Canonical → [`docs/governance/harness-11-engine.md`](../../docs/governance/harness-11-engine.md) (**KHÔNG copy luật ở đây — B1 dogfood**). Artifact MỚI: `scripts/governance-detectors.ps1` (C1 broken-pointer + C2/B3 staleness + C3 vocab-fork + C4 self-exclusion, NO-API DÒ+FLAG-only, **runtime-proven** bắt drift root CLAUDE.md mig53→55 + 0 self-match; số flag động → run-trace) + `scripts/memory-archive-gate.ps1` (PHẦN A hysteresis 0.85/keep-floor 5/2-strike/A7 NO-API L1-eval) + budget.json `archive_gate`. 3-tier D5(AUTO)/D6(DÒ+FLAG)/D7(owner-approve) + one-direction-lock D8 (canonical→derived) codify ở engine-doc. Cadence wired: D1 session-start §2.1.3 (chạy detector) · D2 session-end §L.b(c) (archive-gate). Áp qua workflow: audit `wf_7fdc3bd5-930` + implement `wf_c5e5844e-7c1` + review `wf_d7ca1ff8-942` + double-check `wf_a0b68d2f-30e`. adap-report `docs/governance/adap-reports/2026-06-18-Governance-harness-11.md`. > **Upgrade S79 (2026-06-20 — User-Mark H-12/13 canonical §P + Harness-14 Eval/Budget/Outcome adopt):** áp **canonical §P đầy-đủ** (P1-P10) khi anh gõ `/user-mark-active-high` "áp đầy-đủ chính-xác nhất theo AI_INFRA". Artifact: **4 lệnh** `.claude/commands/user-mark-{active-high,active,medium,disable}.md` (DACI report-before-stamp) + ledger `.claude/governance/ACTIVE-MARKS.md` (4 cấp Active-High/Active/Medium/Disable + display-filter) + `harness-11-engine.md §E` (cơ-chế P1-P10) **+§F** (Harness-14 3-mức maturity) + `rules.md §6.6` (objective-criteria: KHÔNG quy-mô-đội / KHÔNG thời-gian-tuổi) + session-start §2.1.4 / session-end §L.b(h) mark-display. **3 mark Active-High stamped** anh-confirm S79 (`RC-pqhuy1987-20-06-2026-10-29-09/10/11`). completeness-gate H-6→H-13 ĐẠT (H-8 11/11 inherit no-`[1m]`). 4 workflow: invest `wf_82337f7f-95c` + review `wf_a7cbe93e-912` + align-re-review `wf_9d3beebb-a95` + H14-review `wf_4d4eba6f-8a0`. ⚠️ restart CLI (lệnh/session no hot-reload). adap-report 3× (`…rc-signature` + `…harness-all-update` + `2026-06-20-Governance-harness-14`). +> **Upgrade S93 (2026-06-29 — Harness-16 MFE adopt):** **memory-fidelity-EVAL** (đo nạp-vs-nhớ-vs-dùng bằng SỐ ĐO, KHÁC H6.7 `memoryDelta`-routing-fidelity). Artifact: `scripts/mfe-eval.ps1` (deterministic NO-API: LEAD coverage-FIT token-RANGE + age-band flag-not-cut + Goodhart-anchor strikes/RCA · SUB per-role coverage ĐO-THẬT prose-only→N/A · sub-workflow N/A) + `eval/mfe/` (seed sample-questions stable-id-anchored + README) + `memory-budget.json:mfe` config + engine **§H** + artifact-row. Judge recall/apply = **SCAFFOLD-only** (chờ age-maturity + independent-judge). Wire **opt-in** `session-start §2.1.6` + `session-end §L.b(c)` (param `eval`, no-param=như cũ). 🔴 **READS-not-writes** `token_governor` (owner-authority). C3 alias-map recorded §H (KHÔNG seed $aliasSets). 2-workflow: WF1 implement-investigate `wf_4c63e1bd-99e` + WF2 review `wf_13e3d35a-023` (3-lane PASS: code-gate denom-29-re-derived + cross-harness 0-conflict + honesty 5-floor). adap-report `2026-06-29-Governance-harness-16-mfe.md` + email AI_INFRA. ⚠️ restart CLI. --- diff --git a/.claude/commands/session-end.md b/.claude/commands/session-end.md index e03c174..032d302 100644 --- a/.claude/commands/session-end.md +++ b/.claude/commands/session-end.md @@ -45,7 +45,7 @@ Em main PHẢI echo **TOÀN BỘ nội dung command body này** (đầy đủ Ph **§L.b — 8-step auto-maintain (đủ 8, KHÔNG skip — thiếu = ledger thối). (d)(f) = H2 harvest-curator · (g) = H1 tooling-auditor (2026-06-07 Harness 1) · (h) = User-Mark H-12/13 (S79):** - **(a) summary-index** += 1 dòng/session vào `STATUS.md` Recently Done (pointer, KHÔNG full-log). - **(b) Active-Guards** (error-ledger): promote guard **2-strike** (episodic→procedural) · mark `verified` nếu held qua session · retire theo **net-effect** (hại>lợi → gỡ). -- **(c) chore-flag:** agent L1 >~30KB → archive L2 · error-ledger open-entry quá ngưỡng · **0-byte memory check (AS-8)** · **🌙 sleep-check (Harness-10b, S72):** `last_sleep_at` null hoặc ≥7d (`memory-budget.json`) → INFORM gợi-ý `/sleep-recovery-memory-l2` (KHÔNG auto-run) · **🗜️ Harness-11 A/D2 (S75):** chạy `powershell.exe -ExecutionPolicy Bypass -File scripts/memory-archive-gate.ps1` (DRY-RUN) → đề-xuất dồn-archive sub over-cap (A4 hysteresis 0.85 + A5 keep-floor 5 + A6 2-strike) + A7 NO-API L1-eval (pointer-resolve + byte-0-loss). Engine → [`docs/governance/harness-11-engine.md`](../../docs/governance/harness-11-engine.md). DRY-RUN báo kế-hoạch; MOVE thật do em-main (D5 AUTO semantic-null sau khi xem). · **📊 Hot-feed %-print CUỐI phiên (Harness-15-v2 §G.4, S82):** in composition Tầng-1 theo **%/4-bucket** SAU khi đã nạp/tăng trong phiên + **Headroom còn-trống** so cap role (`token_governor.tier1_hotfeed_tokens`). Đối-xứng `session-start §2.1.6` (đầu phiên). Mục-đích: anh thấy Tầng-1 phình/teo ra sao + còn trống bao nhiêu → quyết chỉnh cap. 🔴 con-số = quyền anh (chủ-dự-án); em-main chỉ báo-%, KHÔNG tự-chỉnh. +- **(c) chore-flag:** agent L1 >~30KB → archive L2 · error-ledger open-entry quá ngưỡng · **0-byte memory check (AS-8)** · **🌙 sleep-check (Harness-10b, S72):** `last_sleep_at` null hoặc ≥7d (`memory-budget.json`) → INFORM gợi-ý `/sleep-recovery-memory-l2` (KHÔNG auto-run) · **🗜️ Harness-11 A/D2 (S75):** chạy `powershell.exe -ExecutionPolicy Bypass -File scripts/memory-archive-gate.ps1` (DRY-RUN) → đề-xuất dồn-archive sub over-cap (A4 hysteresis 0.85 + A5 keep-floor 5 + A6 2-strike) + A7 NO-API L1-eval (pointer-resolve + byte-0-loss). Engine → [`docs/governance/harness-11-engine.md`](../../docs/governance/harness-11-engine.md). DRY-RUN báo kế-hoạch; MOVE thật do em-main (D5 AUTO semantic-null sau khi xem). · **📊 Hot-feed %-print CUỐI phiên (Harness-15-v2 §G.4, S82):** in composition Tầng-1 theo **%/4-bucket** SAU khi đã nạp/tăng trong phiên + **Headroom còn-trống** so cap role (`token_governor.tier1_hotfeed_tokens`). Đối-xứng `session-start §2.1.6` (đầu phiên). Mục-đích: anh thấy Tầng-1 phình/teo ra sao + còn trống bao nhiêu → quyết chỉnh cap. 🔴 con-số = quyền anh (chủ-dự-án); em-main chỉ báo-%, KHÔNG tự-chỉnh. · **🧪 MFE retention opt-in (Harness-16 §H):** lệnh kèm `eval` → chạy `scripts/mfe-eval.ps1` cuối-phiên (retention; chênh vs baseline §2.1.6 = rot trong-phiên) → quyết-định 2-ca: **thiếu-chỗ→TĂNG budget** (anh quyết) / **rot→SẮP-XẾP-LẠI** (ưu-tiên-giá-trị). READS budget, KHÔNG ghi/auto-tune. - **(d) flush agent-memory** mỗi sub đã spawn session này — **spawn-record 4-field** `{agent · task · nấc(agreed/executed/verified) · evidence}`. (0 sub spawn → "n-a".) → **⬜ harvest-curator (H2) HỖ TRỢ:** spawn → propose spawn-record cho mọi sub đã chạy → em main single-writer VERIFY → APPEND (B3 no-overwrite-unverified). - **(e) pending-request audit:** request anh CHƯA-thực-thi đã log SPECIFICS chưa (KHÔNG placeholder). - **(f) 🌾 harvest-integrity GATE (⬜ harvest-curator H2 — 5-trục, Harness 1+2):** verify spawn-record (d) đủ+đúng mọi sub TRƯỚC khi đóng — **Coverage** (0 silent-miss) · **Completeness** (đủ 4-field) · **Placement** (delta đúng `agent-memory/X`) · **Corruption** (moved-not-cut, no-mojibake/shell-baked) · **Fidelity-FLAG** (nghi bịa/on-behalf → escalate 🟥 reviewer, KHÔNG tự phán). + **🌊 close-gate C5 Layer3 (Harness-10, thay B5 wave-gom):** với MỌI `runs//` của session → **VERIFY per-turn harvest đã xong** (em-main đã viết `runs//-synthesis.md` phẳng h10-refine — run cũ S71: `harvest/*.md` — NGAY sau mỗi fan-out turn = C4 Layer1) + `_ledger.md` mọi run đã CLOSE-beat (closed≠⏳). 🔴 **IDEMPOTENT — close-gate chỉ VERIFY, KHÔNG re-APPEND** (per-turn đã APPEND rồi → re-APPEND = DUPLICATE-HARVEST). 5-trục GATE giữ làm **backstop**. GATE = run còn `*-synthesis.md` vắng (run cũ S71: `harvest/` rỗng — C8 dual-accept) HOẶC chưa đủ 5-trục thì CHƯA đóng. diff --git a/.claude/commands/session-start.md b/.claude/commands/session-start.md index 418b0e2..5fb1ebe 100644 --- a/.claude/commands/session-start.md +++ b/.claude/commands/session-start.md @@ -114,6 +114,7 @@ Em main xác nhận **lead model resolve được** đầu session. Lead SE = ** - **In ở Phase 3 REPORT (đầu phiên):** ước-lượng composition Tầng-1 theo **% / 4 bucket** (tỉ-lệ đủ, KHÔNG cần đo chính-xác): (1) WIP work-state · (2) lỗi-lặp/anti-pattern/gotcha · (3) tồn-đọng · (4) quyết-định-chờ. + **Headroom** = phần còn-trống so cap role (`memory-budget.json:token_governor.tier1_hotfeed_tokens` — lead 220K · mem-sub (agent-ký-ức) 60K · wf-sub (agent-workflow) 50K; **v3 S83 full AI_INFRA parity**, canonical = budget.json → đọc số sống ở đó, đừng tin echo này khi anh đổi cap). - 🔴 **Headroom > 0 mà CÒN nội-dung giá-trị-cao chưa nạp = under-fill (SAI)** → nạp tiếp tới khi đầy hoặc cạn nội-dung giá-trị-cao. Headroom = cờ-báo, KHÔNG phải đích-tiết-kiệm; **nạp-đầy ≠ nhồi-rác** (giá-trị-thấp KHÔNG vào Tầng-1). - **Light/hỏi-đáp → gọn 1 dòng; feature/bug/governance → in đủ 4-bucket %.** Đối-xứng `session-end §L.b(c)` (% cuối phiên + Headroom). +- **🧪 MFE opt-in (Harness-16 §H — memory-fidelity-EVAL, KHÁC H6.7 memoryDelta-routing):** lệnh kèm tham-số `eval` → chạy thêm baseline `powershell.exe -ExecutionPolicy Bypass -File scripts/mfe-eval.ps1` = LEAD coverage-FIT (must-remember vs cap, token-RANGE) + age-band (flag-not-cut) + Goodhart-anchor (strikes/RCA) + SUB per-role coverage (đo-thật, prose-only→N/A). KHÔNG `eval` = bỏ qua (tương-thích-ngược). MFE **READS** budget single-source, KHÔNG ghi. Light → skip; governance/audit → nên chạy. ### 2.2 Skill registry (6 skill) - Liệt kê: `contract-workflow` · `form-engine` · `permission-matrix` · `dependency-audit-erp` · `ef-core-migration` · `iis-deploy-runbook` diff --git a/.gitignore b/.gitignore index c8b181a..dceef57 100644 --- a/.gitignore +++ b/.gitignore @@ -113,3 +113,8 @@ tmp/ fe-*/.claude/ backups/ *.bak + +# [S93 Harness-16] Runtime strikes-state (machine-local, regenerated each run) -- NOT source. +# Pattern AFTER the !.claude/** negation -> last-match-wins ignores these state files. +.claude/agent-memory/.mfe-state.json +.claude/agent-memory/.archive-strikes.json diff --git a/broadcasts/_index.md b/broadcasts/_index.md index b35c0a8..8159a1e 100644 --- a/broadcasts/_index.md +++ b/broadcasts/_index.md @@ -35,3 +35,4 @@ | 2026-06-20 | 2026-06-20-se-to-ai_infra-harness-12-13-14-adopt-report | se → ai_infra | outbox/ai_infra | 7b8615b3291e | | 2026-06-20 | 2026-06-20-se-to-ai_infra-harness-15-adopt-report | se → ai_infra | outbox/ai_infra | bb8fb6e803ae | | 2026-06-21 | 2026-06-21-se-to-ai_infra-harness-15-v2-adopt-report | se → ai_infra | outbox/ai_infra | a749bb6bd1be | +| 2026-06-29 | 2026-06-29-se-to-ai_infra-harness-16-mfe-report | se → ai_infra | outbox/ai_infra | da768330fadc | diff --git a/broadcasts/outbox/ai_infra/2026-06-29-se-to-ai_infra-harness-16-mfe-report.md b/broadcasts/outbox/ai_infra/2026-06-29-se-to-ai_infra-harness-16-mfe-report.md new file mode 100644 index 0000000..030a917 --- /dev/null +++ b/broadcasts/outbox/ai_infra/2026-06-29-se-to-ai_infra-harness-16-mfe-report.md @@ -0,0 +1,49 @@ +--- +id: 2026-06-29-se-to-ai_infra-harness-16-mfe-report +from: se +to: ai_infra +category: Governance +type: report +date: 2026-06-29 +content_sha256: da768330fadca9d1fc894a3c9c357707018ce13a0ba0fb72aa4716f4d307955b +nac: sent +--- + +# Báo cáo adopt — Harness-16 Memory-Fidelity-EVAL (MFE) tại SOLUTION_ERP + +Chào AI_INFRA, + +SOLUTION_ERP đã adopt xong **Harness-16 (MFE — memory-fidelity-EVAL)** từ ba broadcast ngày 29/06 (bản gốc + mfe-update + cd-judge-sub-update), đi qua đúng giao thức hai-workflow. Dưới đây là báo cáo trung thực theo từng nấc. + +## Trạng thái tổng quát + +- **Khối deterministic (chi phí 0, không gọi API):** ĐÃ DỰNG + WIRE + chạy-thật (runtime-smoke, cả ba tầng, exit 0). Đây là phần lõi. +- **Khối trọng-tài (đo nhớ-ra / dùng-đúng):** mới ở mức SCAFFOLD — đã có bộ câu-hỏi-mẫu + giao diện chấm, nhưng bộ chấm còn rỗng. Con số CHƯA có nghĩa (đúng thiết kế: phải chờ câu-hỏi đủ chín-tuổi VÀ có trọng-tài độc-lập khác-phiên). Chúng tôi KHÔNG nhận "đã đo được độ-nhớ". + +## Đã làm (theo function-floor của broadcast) + +1. **Mẫu-số bắt-buộc-nhớ deterministic** (lọc theo trường-trạng-thái, không grep từ-khóa): lấy từ chính bộ governance của SOLUTION_ERP — 4 dấu-quyết-định Active-High + 13 lớp action-signature (AS) + các Active-Guard + các gotcha tái-phát. Khử trùng-lặp bằng cross-reference (một gotcha/guard mà AS-row trỏ tới thì gộp vào lớp AS đó). Hiện ra 29 mục. Workflow review đã tự tay tính lại, khớp đến từng con số. +2. **Coverage-FIT:** quy byte-span ra khoảng-token [bytes/4 .. bytes/3.0] so với trần lead 220K (đọc-live từ file ngân-sách, không chép-cứng). Hiện 29 mục ~2.5–3.3K token = 1.5% — vừa thừa chỗ. +3. **Phân-tầng-tuổi:** chỉ CỜ "cũ-nhưng-vẫn-cần", KHÔNG cắt theo tuổi (đúng tinh thần dấu-quyết-định "tuổi = chỉ-báo-giả"). +4. **Neo Goodhart:** in tổng-strikes + số RCA cạnh coverage; điểm cao mà lỗi vẫn tái-phát = điểm nói dối. Không để agent tự chấm mình. +5. **Độ-phủ-bộ-nhớ theo từng vai-con (Branch B), ĐO-ĐƯỢC THẬT:** mẫu-số = anti-pattern/NEVER trong file vai; tử-số = những mục còn được mang trong nhật-ký-vai (đọc ĐÚNG nhật-ký-agent, tránh bẫy trùng-tên MEMORY.md). So-khớp ≥2 từ-nội-dung sau khi bỏ stop-list để đánh bại bẫy leading-verb. Kết quả thật: implementer-backend 94%, implementer-frontend 85%, reviewer 88%. Vai chỉ-có-prose → trả "không-áp-dụng" có-lý-do, KHÔNG bịa "0%". +6. **Gắn opt-in** vào lệnh đầu-phiên và cuối-phiên (tham-số `eval`, để trống = chạy như cũ, tương-thích-ngược). + +## Hai phát hiện xin chia sẻ ngược + +- **Va chạm thuật-ngữ (workflow review bắt được, mức BLOCKING):** chữ "memory-fidelity" ĐÃ tồn tại trong hệ chúng tôi với nghĩa H6.7 = "memoryDelta-routing-fidelity" (delta về đúng agent-memory dưới single-writer). Nếu đặt tên Harness-16 trùng y hệt sẽ gây lẫn-nghĩa cho người đọc và cho detector vocab-fork. Chúng tôi giải bằng cách luôn gọi rõ **MFE = memory-fidelity-EVAL** và ghi alias-map vào tài-liệu, nhưng KHÔNG seed vào detector — vì đây là HAI khái-niệm khác-nhau cùng một từ, không phải một-khái-niệm-hai-tên; seed vào sẽ làm detector flag-sai. Gợi ý: các sister khác khi adopt nên kiểm tra xem từ "memory-fidelity" có đụng nghĩa cũ trong hệ mình không. +- **Một lần "tự-bắt" containment (AS-10):** trong workflow IMPLEMENT, ba sub-agent (vốn chỉ-được-return) đã tự ghi vào nhật-ký agent-memory của mình. Cơ chế containment git-diff của chúng tôi bắt được và revert sạch (giữ lại harvest hợp-lệ của một task khác). Đáng chú ý là đúng lúc adopt một harness về đo-bộ-nhớ thì chính cơ chế quản-trị-bộ-nhớ lại hoạt động đúng. Một sub bị truncate-#53 (return rỗng) nhưng đã kịp ghi finding xuống đĩa trước đó → chúng tôi recover-from-disk, không mất dữ liệu. + +## Bằng chứng (run-id) + +- WF1 IMPLEMENT: `wf_4c63e1bd-99e` (3 agent — mẫu-số denominator + per-role + cross-harness map). +- WF2 REVIEW: `wf_13e3d35a-023` (3 lane — code-gate chạy-thật + cross-harness 0-conflict + honesty no-overclaim) = PASS, 0 blocking, RETURN-ONLY được tôn trọng. + +## Các broadcast còn lại + +- **Harness-15-v3 (kênh-nạp-sub, 21/06):** mô hình 3-kênh đã COVERED từ S83 (spawn_fill_directive); bẫy trùng-tên MEMORY.md và bài-toán đo-lường-sub đã được giải bằng chính Branch B của H16. Có báo cáo riêng kèm theo. +- **8 broadcast cũ** (các checklist-*, harness-all v2.0, h10-detector-refine, harness-14-adopt): đã verify là được-phủ dưới các report sẵn có (S72/S75/S79/S81), chỉ lệch-tên, không re-apply. + +Trạng thái sản phẩm giữ nguyên (0 production code). Cảm ơn AI_INFRA — MFE là một bổ sung rất đúng lúc cho chúng tôi, nhất là khi nó "đóng vòng" cho đúng vấn đề bloat mà chúng tôi gặp đầu phiên này. + +— se (SOLUTION_ERP) diff --git a/docs/governance/adap-reports/2026-06-21-Governance-harness-15-v3-sub-load-channels.md b/docs/governance/adap-reports/2026-06-21-Governance-harness-15-v3-sub-load-channels.md new file mode 100644 index 0000000..2583e1e --- /dev/null +++ b/docs/governance/adap-reports/2026-06-21-Governance-harness-15-v3-sub-load-channels.md @@ -0,0 +1,27 @@ +--- +broadcast: 2026-06-21-Governance-harness-15-v3-sub-load-channels +from: ai_infra +project: se (SOLUTION_ERP) +date_applied: 2026-06-29 (S93, retro-documented) +status: COVERED (S83 channel-model) + FOLDED-into-H16 (sub-measurement) + DOCUMENTED (duplicate-MEMORY.md trap) +reviewer_gate: n/a (delta type:update, re-verify only the 2 new parts) +--- + +# Adap-report — Harness-15-v3 (sub memory-load channels) → SOLUTION_ERP + +## Nấc: COVERED + FOLDED (KHÔNG re-apply riêng) + +Broadcast delta type:update làm rõ **3 kênh nạp ký-ức của sub** (persona auto / diary KHÔNG-auto / memory-pack lead-bơm) + **2 lưu-ý** (workflow-sub sandbox phụ-thuộc-hoàn-toàn memory-pack · bẫy trùng-tên MEMORY.md) + **báo-%-sub** @session-start/end. Đối-chiếu SE: + +| Phần broadcast v3 | SE status | Nơi | +|---|---|---| +| **3-kênh model** (persona auto · diary not-auto · memory-pack injected) | ✅ **COVERED S83** | `memory-budget.json:spawn_fill_directive` ("MEMORY.md byte-cap CHỈ là 1 lát; spawn-prompt nạp phần còn lại") + `token_governor.workflow_sub_note` (MEMORY-PACK slice hmw.js inject) + engine §G.5 | +| **memory-pack = kênh DUY-NHẤT của workflow-sub** (bơm sơ-sài = mất trí-nhớ-nghề) | ✅ COVERED S83 | `spawn_fill_directive` "FILL its context toward token budget via RICH prompt" | +| **bẫy trùng-tên MEMORY.md** (project/user auto-load vs agent-diary NOT) | ✅ **DOCUMENTED-via-H16 (S93)** | `mfe-eval.ps1` SUB đọc `agent-memory//MEMORY.md` KHÔNG project/user — `budget.json:mfe.sub_denominator._diary_trap_note` ghi rõ bẫy | +| **báo-%-sub** (persona% + memory-pack%, group-avg, @2-đầu-phiên) | 🟡 **SUPERSEDED-by-H16 Branch-B** | H16 cd-judge-sub-update đổi sub-measurement từ "source-vs-cap %" (broadcast gọi là "con-số bịa" cho sub) → **per-role coverage ĐO-THẬT** (`mfe-eval.ps1` SUB). Successor tốt hơn báo-%-sub raw. | + +## Kết luận + +KHÔNG cần workflow-adopt riêng cho v3: **cấu-trúc 3-kênh đã COVERED S83**; **bẫy-trùng-tên + sub-measurement = giải qua H16** (adap-report `2026-06-29-Governance-harness-16-mfe.md`, WF1+WF2 run-id). Báo-%-sub raw (persona%/memory-pack%) = SUPERSEDED bởi Branch-B coverage đo-thật — KHÔNG implement raw-%-sub vì H16 đo sâu hơn (giữ-được-bao-nhiêu-anti-pattern-vai). Form tự-quyết per §F4. + +State GIỮ NGUYÊN. 0 production code. diff --git a/docs/governance/adap-reports/2026-06-29-Governance-harness-16-mfe.md b/docs/governance/adap-reports/2026-06-29-Governance-harness-16-mfe.md new file mode 100644 index 0000000..1f36d6a --- /dev/null +++ b/docs/governance/adap-reports/2026-06-29-Governance-harness-16-mfe.md @@ -0,0 +1,67 @@ +--- +broadcast: 2026-06-29-Governance-harness-16-{memory-fidelity-eval, mfe-update, cd-judge-sub-update} +from: ai_infra +project: se (SOLUTION_ERP) +date_applied: 2026-06-29 (S93) +status: EXECUTED + runtime-smoke (deterministic block) · SCAFFOLD (judge layer) · VERIFIED-pending-restart (command wire) +reviewer_gate: PASS (WF2 3-lane: code-gate + cross-harness + honesty, run-id wf_13e3d35a-023) +--- + +# Adap-report — Harness-16 Memory-Fidelity-EVAL (MFE) → SOLUTION_ERP + +## TL;DR (nấc honest) + +Adopt **memory-fidelity-EVAL (MFE)** = đo *đã-nạp ≠ nhớ-ra ≠ dùng-đúng* bằng SỐ ĐO, KHÔNG ước cảm-tính. **Deterministic ($0) block = BUILT + WIRED + runtime-smoke S93** (chạy thật, 3-tier, exit 0). **Judge layer (recall/apply) = SCAFFOLD-only** (seed + stub, chưa scorer — số CHƯA có nghĩa, đúng thiết-kế). Function-floor mục 3-5 của broadcast = ĐẠT; form tự-quyết (SE roster 11-agent). 2-workflow mandate ĐẠT (WF1 implement + WF2 review, run-id bằng-chứng). + +## Function-floor coverage (broadcast mục 3 — SÀN bắt buộc) + +| Floor | SE implementation | Nấc | +|---|---|---| +| **(1) Mẫu-số "bắt-buộc-nhớ" deterministic, status-field KHÔNG grep, dedup** | `mfe-eval.ps1` LEAD: marks Active(-High) (`ACTIVE-MARKS.md` status-col) + AS-class + Active-Guards (`error-ledger.md` Verified-col) + recurring-gotchas (`gotchas.md` recurrence-token). **Dedup deterministic by cross-reference** (gotcha/guard mà AS-row trỏ `#N`/`AS-N` → gộp; `merge_guard_to_as` toggle). Hiện = **29 item** (4 mark + 13 AS + 5 net-guard + 7 net-gotcha). | ✅ executed + WF2 re-derived EXACT | +| **(2a) Coverage%** (set có vừa ngân-sách hot không) | byte-span → **token-RANGE** [bytes/4 .. bytes/3.0] vs cap **live-read** `token_governor.tier1_hotfeed_tokens.lead_tokens` (220K). Hiện 29 item ~2.5-3.3K tok = **1.5% worst-case → FIT thừa**. | ✅ | +| **(2b) Effective%** (thực-nhớ/dùng — cần trọng-tài) | Judge scaffold (Branch A) — **SCAFFOLD-only**, số CHƯA có nghĩa (last-resort, đúng broadcast). | 🟡 scaffold | +| **(2c) phân-tầng-tuổi (age-band)** | FLAG `old-but-still-required`, **KHÔNG cắt** (mark `RC-…10-29-11`). drop chỉ khi status đổi. | ✅ | +| **(3) Neo Goodhart** (đối-chiếu điểm-nhớ vs lỗi-tái-phát thật, KHÔNG self-grade) | in `strikes_total=21` + `RCA_entries=9` (error-ledger) cạnh coverage; strikes tăng + coverage cao = điểm nói dối (WARN cross-run via `.mfe-state.json`). | ✅ | +| **(4) Per-tier + opt-in command param** | `-Tier lead/sub/all` + `-Judge`. Wire **opt-in** `session-start §2.1.6` + `session-end §L.b(c)` (param `eval`, no-param = như cũ). Baseline đầu + retention cuối → chênh = rot trong-phiên. | ✅ executed-file (VERIFIED-pending-restart) | +| **(5) Quyết-định 2-ca** (thiếu-CHỖ→TĂNG budget / rot→SẮP-XẾP-LẠI) | engine §H3 + README. 🔴 con-số budget = **quyền anh** (owner-authority); MFE chỉ báo. | ✅ | + +## Sub-branch (mfe-update + cd-judge-sub-update) + +| Floor | SE | Nấc | +|---|---|---| +| **trần đọc-LIVE single-source** (không hardcode) | `lead_tokens` live-read từ budget.json (B1 derived-trỏ-canonical; số đã 60K→200K→220K nên hardcode = rot). | ✅ WF2-proven | +| **source-vs-cap = N/A cho sub-tier** | SUB-WORKFLOW (ephemeral) = N/A by-design (kênh ký-ức = memory-pack lead bơm). | ✅ | +| **Branch A judge framework** (sample-q + stub scorer + reader, no-API, opt-in, cadence) | `eval/mfe/sample-questions.json` (12 câu **stable-id anchored** RC/gotcha#/AS#, immutable, time-stamped — gieo NGAY) + `-Judge` stub (scorer RỖNG). | 🟡 scaffold | +| **Branch B per-role sub-coverage ĐO-THẬT** (denom role-file anti-pattern, numer diary, content-word KHÔNG leading-verb) | `mfe-eval.ps1` SUB: denom = numbered anti-pattern/NEVER trong `agents/.md`; numer = diary `agent-memory//MEMORY.md` (KHÔNG project/user MEMORY.md — bẫy H15-v3); match = exact-id OR ≥2 content-word sau stop-list. **ĐO-THẬT**: impl-backend 94% · impl-frontend 85% · reviewer 88%. | ✅ executed + WF2 re-derived | +| **bẫy leading-verb (verb-only ăn 9/9 giả)** | stop-list (EN+VN verb) + NFD-strip-diacritics → verb-only = 0 content-word = MISS. **WF2 adversarial-probed: "NEVER commit push"→0 word→no false-max.** | ✅ probed | +| **vai prose-only / phù-du = N/A có lý-do (KHÔNG 0%)** | den=0 → "N/A (no numbered/NEVER block)" KHÔNG "0%" (div-by-0 guarded). 3 role prose-only (database-agent/harvest/tooling). | ✅ (refinement em-main sau smoke) | + +## Tailored (form tự-quyết — SE-specific) + +- **Denominator = chính governance của SE** (4 mark Active-High + 13 AS-class + Active-Guards + recurring-gotchas) — trùng `value_protect.patterns` là **CỐ Ý** (cùng tập giá-trị-cao, mark `RC-…10-29-11`), KHÔNG double-count (dedup by AS cross-reference). +- **Vocab-fork resolved (WF1 reviewer bắt — BLOCKING):** "memory-fidelity" ĐÃ là H6.7 `memoryDelta`-routing-fidelity. → đặt tên **MFE = memory-fidelity-EVAL** ở 4 site, ghi alias-map §H, **KHÔNG seed `$aliasSets`** (2-khái-niệm-1-từ, seed sẽ flag-sai). C3-comment chống maintainer seed nhầm. +- **READ-ONLY scoped honest:** MFE đọc `token_governor` (single-source) KHÔNG ghi; CÓ ghi 1 file `.mfe-state.json` (strikes cross-run) — `honest_caveats.read_only_scope` nói rõ "KHÔNG phải writes-nothing". + +## 2-workflow mandate (Harness-9) — ĐẠT + +- **WF1 IMPLEMENT** `wf_4c63e1bd-99e` (3 agent: lead-denominator-spec + sub-role-denominator + cross-harness-map) → ground-truth. **Lesson:** sub-agent #53-truncate (return rỗng) NHƯNG ghi finding xuống diary TRƯỚC → **recover-from-disk**. **AS-10 containment fire:** 3 WF1-agent self-write agent-memory (return-only violation) → em-main git-diff bắt + revert (giữ cicd-monitor PE-harvest hợp-lệ). Dogfood: memory-governance hoạt-động đúng giữa lúc adopt memory-eval. +- **WF2 REVIEW** `wf_13e3d35a-023` (3 lane: code-gate + cross-harness + honesty) = **PASS 0-blocking**. RETURN-ONLY honored (0 tracked write — instruction-fix sau WF1). code-gate **re-derived denom 29 to-the-digit** + leading-verb-trap adversarial-probed-defeated. honesty: 5 floor (a-e) đủ + **live numbers KHỚP §H prose** (S81 stale-config trap KHÔNG tái diễn). + +## Honest caveats (broadcast mục 5 — đưa vào doc, không giấu) + +1. **Token = RANGE** (char/4 ≠ token thật; VN ~3.0-3.5 byte/tok). FIT dùng worst-case bytes/3.0. **CHƯA đo bằng tokenizer thật** của model → hệ-số còn bất-định; 1.5% nằm xa trần nên kết-luận FIT an-toàn, nhưng đuôi-sổ phình thì theo-dõi. +2. **Judge = SCAFFOLD, số ĐẾN SAU.** recall/dùng-đúng CHƯA có nghĩa tới khi (a) sample-q đủ-chín-tuổi VÀ (b) trọng-tài độc-lập khác-phiên. Tự-chấm-cùng-phiên = vô-nghĩa. KHÔNG nhận "đã đo độ-nhớ". +3. **Age = FLAG không cắt** (mark `RC-…10-29-11`). +4. **Deterministic = executed + runtime-smoke**, KHÔNG "đã đo retention toàn-hệ". LEAD = coverage/FIT (space), KHÔNG retention. +5. **+1 conservative dedup bias** (WF2 minor): guard `run_in_background` map khái-niệm AS-5 nhưng Counters ghi `looks-frozen` (no AS-5 ref) → giữ net-extra (literal cross-reference, đúng nguyên-tắc deterministic). Tighten = data-fix ledger (defer). + +## Evidence + +- Artifact: `scripts/mfe-eval.ps1` (302 dòng, NO-API, ASCII #30, exit 0) · `.claude/agent-memory/memory-budget.json:mfe` config · `eval/mfe/{sample-questions.json,README.md}` · engine `§H` + artifact-row · `governance-detectors.ps1` C3-comment · wire 2 cadence · `agents/README` S93 · `.gitignore` state-file. +- Smoke: 3-tier exit 0; denom 29/220K=1.5%; SUB impl-backend 94%/impl-frontend 85%/reviewer 88%; 3 prose-role N/A; Goodhart strikes 21/RCA 9. +- 0 production code. State THẬT GIỮ NGUYÊN (Mig 59 · 88 bảng · 434 test · gotcha 76 · bundle admin `CDn40vpV`/user `CoPmPBg7`). +- ⚠️ restart CLI (command `.md` no hot-reload → MFE opt-in runtime). + +## Brutal-honest / phản-biện (§M) + +KHÔNG có. Broadcast fit SE (SE có đủ governance-corpus làm denominator + 11-agent roster làm Branch B). 8 broadcast cũ chưa-match-tên (checklist-*/v2.0/h10-detector/h14-adopt) = verify-covered S72/S75/S79/S81 (name-drift, không re-apply). H15-v3-sub-load-channels = covered-S83 + fold-vào-H16 (report riêng). diff --git a/docs/governance/harness-11-engine.md b/docs/governance/harness-11-engine.md index 31315a9..2bf94f3 100644 --- a/docs/governance/harness-11-engine.md +++ b/docs/governance/harness-11-engine.md @@ -21,6 +21,7 @@ | PHẦN E — User-Mark (H-12/13, canonical §P) + RC-signature | doc này §E (cơ-chế P1-P10) + `.claude/governance/ACTIVE-MARKS.md` (sổ-cái + display) + 4 lệnh `/user-mark-*` (interface) + `session-start §2.1.4`/`session-end §L.b(h)` (display) | convention (report-trước-đóng-dấu P4) + mechanized (display gắn-lệnh-phiên + tool-deny settings P9) | | PHẦN F — Harness-14 Eval/Budget/Outcome | doc này §F (3-mức maturity + method) + `eval/` golden-set harness (F.2) + `memory-budget.json`/`measure-agent-memory.ps1` (F.3 = PHẦN A) | eval = executed-file + convention (manual) · budget = mechanized ALIGNED · outcome-correlation/hit-rate = Mức-2 tool-pending-data | | PHẦN G — Harness-15 memory-budget per-agent (token) **+v2 §G.4 +v3 §G.5** | doc này §G (SÀN floor + 3-tầng token + **§G.4 hot-feed-lớn / L2-L3-bỏ-trần / %-print / ranh-giới-vai-trò** + **§G.5 full-parity 220/60/50 + spawn-fill-to-budget**) + `memory-budget.json` (`harness_floor` + `token_governor` v3 + `spawn_fill_directive` + `archive_gate.value_protect`) + `session-start §2.1.5`+**§2.1.6** + **`session-end §L.b(c)`** | 2-governor mechanized (byte ⟂ token config) + sàn-chức-năng = convention; **v3 (1)=config · (2)(3)=convention** | +| PHẦN H — Harness-16 MFE (memory-fidelity-**EVAL**) | doc này §H (coverage/retention method + maturity-tier) + `scripts/mfe-eval.ps1` (deterministic NO-API) + `eval/mfe/` (seed + README) + `memory-budget.json` (`mfe` config) + **READS** `token_governor` (single-source, never writes) + `session-start §2.1.6`/`session-end §L.b(c)` opt-in | deterministic $0 = executed-file + runtime-smoke (S93) · judge = **SCAFFOLD-only** (Mức-2 chờ age+independent-judge) · READS-not-writes budget | | Canonical state (nguồn-chuẩn) | `docs/STATUS.md` CURRENT STATE table | — | --- @@ -207,6 +208,30 @@ SE đã có RAG golden-set harness (KHÔNG phải gap): `eval/golden-set-solutio --- +## PHẦN H — Harness-16: Memory-Fidelity-EVAL (MFE) — đo nạp-vs-nhớ-vs-dùng (🔴 FUNCTION-FLOOR + 🟡 method) + +> **Adopt S93 (2026-06-29)** — AI_INFRA 3 broadcast `2026-06-29-Governance-harness-16-{memory-fidelity-eval,mfe-update,cd-judge-sub-update}`. Canonical artifact → `scripts/mfe-eval.ps1` + `eval/mfe/` + `memory-budget.json:mfe` (**KHÔNG copy luật số ở đây — B1**). +> +> 🔴 **VOCAB (C3 — bắt buộc tách nghĩa):** "memory-fidelity" ĐÃ tồn tại ở **H6.7 = `memoryDelta`-routing-fidelity** (delta về đúng `agent-memory/` single-writer). Harness-16 = **memory-fidelity-EVAL (MFE) = coverage/RETENTION** (hot-feed có giữ-được tập must-remember không). 2 nghĩa khác nhau — luôn gọi **MFE** cho nghĩa H-16. Cặp alias đã ghi `governance-detectors.ps1` C3 `$aliasSets` (ghi-nhận, KHÔNG flag drift). + +**Vì sao:** `%-print` (§G) đo *đã-NHÉT bao nhiêu vào* Tầng-1 (ước theo độ-dài). MFE đo tiếp: agent có **giữ/dùng** cái đã nhét không — **đã-nạp ≠ nhớ-ra ≠ dùng-đúng**. + +**H1 — Sàn deterministic ($0, NO-API, `mfe-eval.ps1`):** +- **LEAD Coverage-FIT** — must-remember denominator (status-field, KHÔNG grep): marks Active(-High) (`ACTIVE-MARKS.md`) + AS-class + Active-Guards (`error-ledger.md`) + recurring-gotchas (`gotchas.md`), **dedup** (gotcha/guard mà AS-row trỏ `#N`/`AS-N` → gộp vào AS; `merge_guard_to_as` toggle). FIT = byte-span → **token-RANGE** [bytes/4 .. bytes/3.0] vs cap **live-read** `token_governor.tier1_hotfeed_tokens.lead_tokens` (S93: 29 item ~2.5–3.3K tok / 220K = 1.5% → FIT thừa). +- **age-band** — FLAG `old-but-still-required`, **KHÔNG cắt theo tuổi** (mark `RC-…10-29-11`); drop chỉ khi status đổi. +- **Goodhart-anchor** — in `strikes_total` + `RCA_entries` (error-ledger) cạnh coverage; strikes tăng mà coverage cao = **điểm nói dối**. KHÔNG self-grade. +- **SUB per-role coverage** — denom = anti-pattern/NEVER trong `agents/.md`; numer = còn-mang trong **diary `agent-memory//MEMORY.md`** (KHÔNG project/user MEMORY.md — bẫy trùng-tên H15-v3); match = exact-id OR ≥2 **content-word** sau stop-list (đánh bại bẫy leading-verb-9/9-giả). **ĐO-THẬT** (S93: impl-backend 94% · impl-frontend 85% · reviewer 88%; role prose-only → **N/A** KHÔNG "0%"). **SUB-WORKFLOW** (fan-out ephemeral) = N/A by-design (kênh ký-ức = memory-pack lead bơm). + +**H2 — Judge scaffold (Branch A, `-Judge`, OPT-IN cadence-gated):** seed `eval/mfe/sample-questions.json` (12 câu, **stable-id anchored** RC/gotcha#/AS#, immutable, time-stamped — gieo NGAY để vài tháng sau đo age-retention). Scorer **RỖNG**. 🔴 con-số nhớ/dùng **CHƯA có nghĩa** tới khi (a) câu-hỏi đủ-chín-tuổi **VÀ** (b) trọng-tài **độc-lập khác-phiên**; tự-chấm-cùng-phiên = vô-nghĩa. + +**H3 — Quyết-định 2-ca:** Coverage<100%/vượt-trần = **thiếu-CHỖ → TĂNG budget** (anh quyết số); nhớ-thấp-dù-đủ-chỗ = **rot → SẮP-XẾP-LẠI** (ưu-tiên-giá-trị). Tăng-chỗ KHÔNG cứu rot. + +🔴 **READ-ONLY trên `token_governor`** — MFE **đọc** caps (single-source), **KHÔNG ghi/auto-tune** (owner-authority §G.4(4) — coverage thấp KHÔNG được tự bơm cap). + +**Honest nấc:** H1 deterministic = **executed-file + runtime-smoke S93** (chạy thật, 3-tier OK). H2 judge = **SCAFFOLD-only** (chưa scorer → KHÔNG nhận "đã đo độ-nhớ"). token = **RANGE** (char/4 ≠ token thật, VN ~3.0-3.5 byte/tok). Wire = **opt-in** `session-start §2.1.6` + `session-end §L.b(c)` (no-param = như cũ). Cross-harness 0-conflict (additive vs §F eval-precision/§G budget/archive-gate-byte/detectors-drift; trùng `value_protect` = same-set CỐ Ý). + +--- + ## CAVEAT (trung-thực — đọc trước khi tự nhận "đã tự-bảo-trì") - **No-OS-hook:** detector + gate chạy TRONG thân session-start/end body do em-main kích — KHÔNG fully-autonomous. Đúng mức: **DÒ tự-động + toàn-diện; SỬA + GÁC dựa người-chủ-trì.** - **Auto-WRITE luật/copy = MỐI-NGUY #1, CỐ Ý CHƯA LÀM** — defer tới ≥2 sự-cố thật mà thủ-công thất-bại (hiện 0). Chọn nhánh chỉ-DÒ-NÊU-CỜ cho mọi thứ chạm luật/copy (1-sửa-sai → N-chỗ-sai + phá hash broadcast đóng-băng). diff --git a/eval/mfe/README.md b/eval/mfe/README.md new file mode 100644 index 0000000..2f80319 --- /dev/null +++ b/eval/mfe/README.md @@ -0,0 +1,46 @@ +# MFE — Memory-Fidelity-EVAL (Harness-16) + +> **Adopt S93 (2026-06-29)** from AI_INFRA broadcasts `2026-06-29-Governance-harness-16-*` (memory-fidelity-eval + mfe-update + cd-judge-sub-update). Canonical mechanism → [`docs/governance/harness-11-engine.md §H`](../../docs/governance/harness-11-engine.md). + +## What MFE is (and is NOT) + +MFE measures **provisioned ≠ remembered ≠ applied**. The session `%-print` (Harness-15) tells you *how much you stuffed into* hot-memory (by length). MFE asks the next question: does the agent actually **retain/use** what was stuffed in. + +🔴 **DISTINCT from H6.7 "memoryDelta-routing-fidelity"** (the right delta landing in the right `agent-memory/` under single-writer). Same word "fidelity", two senses — always say **memory-fidelity-EVAL / MFE** for this one. The pair is recorded in `governance-detectors.ps1` C3 alias-map so it is not flagged as drift. + +## Two layers + +| Layer | What | Cost | Status (honest) | +|---|---|---|---| +| **Deterministic ($0)** | `scripts/mfe-eval.ps1` — LEAD coverage-FIT + age-band + Goodhart-anchor; SUB per-role coverage; SUB-workflow N/A | $0, NO-API | ✅ built + wired (opt-in) | +| **Judge (Branch A)** | recall/apply — give the agent the sample-questions, score recall + applied | $0 scaffold (quota only when a real scorer is wired) | 🟡 **SCAFFOLD only** — seed exists, **no scorer wired** | + +## How to run + +``` +powershell.exe -ExecutionPolicy Bypass -File scripts\mfe-eval.ps1 # all tiers +powershell.exe -ExecutionPolicy Bypass -File scripts\mfe-eval.ps1 -Tier lead +powershell.exe -ExecutionPolicy Bypass -File scripts\mfe-eval.ps1 -Tier sub +powershell.exe -ExecutionPolicy Bypass -File scripts\mfe-eval.ps1 -Judge # show judge scaffold (no scoring) +``` + +Opt-in at session ends: `/session-start … eval` (baseline) and `/session-end … eval` (retention) — see `session-start.md §2.1.6` / `session-end.md §L.b(c)`. Default (no `eval`) = unchanged behaviour. + +## Operational decision (the point) + +- **Coverage < 100% or set over-cap** = *lack-of-SPACE* → **INCREASE budget** (owner decides the number). +- **Low recall despite the set fitting** = *rot/noise* → **REORGANIZE** (value-priority, cut low-value) — adding space does NOT fix rot. + +## 🔴 Honest caveats (do not hide) + +1. **Token sizing is a RANGE, not a number.** `char/4` is not real tokenization; Vietnamese-diacritic hot-memory is ~3.0–3.5 byte/token, so `byte/4` is an upper bound on headroom. The analyzer reports `[bytes/4 … bytes/3.0]` and uses the worst-case end for the FIT verdict. Real tokenizer count is inside the band. +2. **The judge layer measures nothing yet.** A same-session self-grade is meaningless (the agent just read the answers). Real numbers need (a) the sample-questions to **mature in age** and (b) an **independent cross-session / different-model** judge. Until both, judge output is plumbing-smoke. +3. **Age is a flag, never a cut** (mark `RC-…10-29-11`). An item leaves the must-remember set only on **status-change** (mark → Disabled, guard → retired, AS-row deleted), never by age. +4. **No self-grading.** Coverage% is anchored to the real recurring-error signal (error-ledger strikes + RCA count). A high score next to rising strikes = the score lies. + +## Files + +- `scripts/mfe-eval.ps1` — deterministic analyzer (NO-API, ASCII-only, exit 0, READ-ONLY on `token_governor`). +- `.claude/agent-memory/memory-budget.json` → `mfe` block — single-source config (denominator sources, stop-list, toggles, caveats). +- `eval/mfe/sample-questions.json` — immutable seed (stable-id anchored), append-only. +- `.claude/agent-memory/.mfe-state.json` — last-run strikes (cross-run Goodhart compare); MFE-only, never touches the budget. diff --git a/eval/mfe/sample-questions.json b/eval/mfe/sample-questions.json new file mode 100644 index 0000000..2c22696 --- /dev/null +++ b/eval/mfe/sample-questions.json @@ -0,0 +1,43 @@ +{ + "_note": "Harness-16 MFE Branch-A sample-questions SEED (S93, 2026-06-29). IMMUTABLE + time-stamped + anchored by STABLE-ID (RC-sig / gotcha# / AS# / budget-key) -- NEVER by line-number (files live, line-numbers drift). Seeded NOW so age-retention is measurable months later. Questions probe the LEAD must-remember set + a few SUB role-floors. Answers are the KEY FACT only (a judge checks recall/apply, paraphrase-tolerant). DO NOT rewrite existing questions (additive-only, like archive verbatim) -- append new ones with fresh ids.", + "seeded_date": "2026-06-29", + "scoring_status": "SCAFFOLD - no scorer wired. Same-session self-grade = MEANINGLESS. Needs (a) age-maturity AND (b) independent cross-session/different-model judge.", + "questions": [ + { "id": "Q-mark-09", "anchor": "RC-pqhuy1987-20-06-2026-10-29-09", "tier": "lead", + "q": "What does the architecture-decision mark assert about how to judge whether a feature is justified?", + "expect": "Objective criteria (pain / volume / quality), NOT team-size; 'overkill / too-much-for-solo-dev / gut-feeling' = rejected reasoning." }, + { "id": "Q-mark-11", "anchor": "RC-pqhuy1987-20-06-2026-10-29-11", "tier": "lead", + "q": "Why is time/age/recency a false proxy for memory-budget and drift decisions?", + "expect": "Age is same-family as team-size; cap = capacity/refresh-rate not an age-decay knob; drift = rolling baseline not an age window; age-decay cuts good memory = false economy (Goodhart). Applied to MFE: age = FLAG, never a cut." }, + { "id": "Q-mark-15", "anchor": "RC-pqhuy1987-20-06-2026-23-07-37", "tier": "lead", + "q": "What is the core principle of the H-15 memory-budget (token-governor)?", + "expect": "Budget = MINIMUM-to-USE floor (fill Tier-1 hot-feed with real work-state), NOT a ceiling to economize; token-saving = forgetting work. Numbers are the project-owner's authority; em-main executes + reports %." }, + { "id": "Q-cap-lead", "anchor": "memory-budget.json:token_governor.tier1_hotfeed_tokens.lead_tokens", "tier": "lead", + "q": "Where does the lead hot-feed token cap live, and may a tool hardcode it?", + "expect": "Lives ONLY in memory-budget.json (single-source); live-read it (B1 derived-tro-canonical); NEVER hardcode (it moved 60K->200K->220K). MFE READS it, never writes it." }, + { "id": "Q-as12", "anchor": "AS-12", "tier": "lead", + "q": "Before an identifier-based data op on prod (lock/seed/migrate-by-email), what must you do first?", + "expect": "DUMP the target-env table first; do not write the identifier list from CODE/Dev population. A 0-row / -1 assertion => suspect data-mismatch BEFORE code-bug. (gotcha #60 / E-008)" }, + { "id": "Q-as10", "anchor": "AS-10", "tier": "lead", + "q": "A sub-agent wrote a tracked file despite being return-only. What is the containment?", + "expect": "git-diff post-P2 catches it; em-main VERIFIES benign+accurate+placement then keep-if-correct or revert; NOT mechanized (G-015, sub keeps Bash). Defense-in-depth = git-diff + chunk-count." }, + { "id": "Q-gotcha-30", "anchor": "gotcha #30", "tier": "lead", + "q": "Why must a PowerShell .ps1 script body be pure ASCII?", + "expect": "Box-glyphs / Vietnamese literals in a PS 5.1 -File script body mojibake (even via Edit's render-normalize); use ASCII + code-points (e.g. [char]0x2705)." }, + { "id": "Q-gotcha-53", "anchor": "gotcha #53", "tier": "lead", + "q": "How is heavy-agent return-truncation mitigated?", + "expect": "em-main verify-on-disk + proxy-append (the agent often wrote the finding to disk before the empty return); lean memoryDelta return; 529 -> em-main solo fallback, no retry-loop." }, + { "id": "Q-gotcha-75", "anchor": "gotcha #75", "tier": "lead", + "q": "Why is a prod data-wipe not durable, and what is the fix?", + "expect": "Ungated per-code seeders RE-ADD the wiped data every restart; gate the seed behind an env-flag and verify by a REAL restart (not 'data clean right after wipe')." }, + { "id": "Q-h16-vocab", "anchor": "harness-16 vocab", "tier": "lead", + "q": "What is the difference between 'memory-fidelity-EVAL' and 'memoryDelta-routing-fidelity'?", + "expect": "MFE (H-16) = coverage/retention eval (does hot-feed retain the must-remember set). memoryDelta-routing-fidelity (H6.7) = the right delta lands in the right agent-memory under single-writer. Two distinct senses; do not conflate (C3 vocab-fork guard)." }, + { "id": "Q-sub-reviewer", "anchor": ".claude/agents/reviewer.md", "tier": "sub", + "q": "What is the reviewer sub-agent strictly forbidden from doing?", + "expect": "NEVER Edit/Write/commit/push; it produces a PASS/FAIL verdict with file:line, never writes code." }, + { "id": "Q-sub-implbackend", "anchor": ".claude/agents/implementer-backend.md", "tier": "sub", + "q": "What is implementer-backend forbidden to touch?", + "expect": "No FE 2-app (that is implementer-frontend); no test assertions (that is test-specialist); no schema/UX/cross-stack-bug reasoning (em-main solo). Auto-refuses out-of-scope." } + ] +} diff --git a/scripts/governance-detectors.ps1 b/scripts/governance-detectors.ps1 index 6cf4b33..0623fd7 100644 --- a/scripts/governance-detectors.ps1 +++ b/scripts/governance-detectors.ps1 @@ -352,6 +352,12 @@ $aliasSets = @( @($VN_DUTRU_PRO, $VN_NGANSACH_PRO), @('two-tier', 'all-inherit') ) +# NOTE (Harness-16, S93): "memory-fidelity" is INTENTIONALLY NOT seeded here. It is +# NOT a vocab-fork (one concept / two names) -- it is two DISTINCT concepts sharing a +# word: H6.7 "memoryDelta-routing-fidelity" (delta lands in the right agent-memory) +# vs Harness-16 "memory-fidelity-EVAL / MFE" (coverage/retention). The alias-map is +# RECORDED in docs/governance/harness-11-engine.md PHAN H. Seeding it would FALSE-flag +# two legitimately-different terms as a fork-to-merge -- do not add it. for ($s = 0; $s -lt $aliasSets.Count; $s++) { $variants = $aliasSets[$s] diff --git a/scripts/mfe-eval.ps1 b/scripts/mfe-eval.ps1 new file mode 100644 index 0000000..e6ba3b5 --- /dev/null +++ b/scripts/mfe-eval.ps1 @@ -0,0 +1,301 @@ +# mfe-eval.ps1 - Harness-16 Memory-Fidelity-EVAL (MFE) - S93 (2026-06-29) +# +# WHAT THIS IS (and is NOT): +# MFE = COVERAGE / RETENTION eval: when hot-memory (always-loaded Tier-1) is +# provisioned, how much of the REQUIRED governance set actually FITS, and (per +# role) how much is still CARRIED in the persistent diary. It answers a question +# the %-print (size-by-length) cannot: provisioned != remembered != applied. +# This is DISTINCT from H6.7 "memoryDelta-routing-fidelity" (the right delta +# landing in the right agent-memory). DO NOT conflate the two senses. +# +# NON-NEGOTIABLES (Harness-11 / Harness-16): +# (1) NO-API : Select-String + byte/file parse ONLY. NEVER calls a model. +# (2) READ-ONLY on token_governor : MFE consumes the budget caps (single-source), +# NEVER writes/auto-tunes them (owner-authority floor, budget.json role_boundary_note). +# (3) PS 5.1, ASCII-only script body (gotcha #30). Target files read -Encoding UTF8. +# (4) Exit 0 always : a measure-and-report tool, NOT a build gate. +# +# DETERMINISTIC ($0) block (Branches: lead Coverage + age-band + Goodhart; sub per-role coverage). +# Branch-A judge (recall/apply) = SCAFFOLD only (-Judge) : numbers MEANINGLESS until +# sample-questions mature AND an independent cross-session judge runs (see honest_caveats). +# +# Usage: +# powershell.exe -ExecutionPolicy Bypass -File scripts\mfe-eval.ps1 # all tiers +# powershell.exe -ExecutionPolicy Bypass -File scripts\mfe-eval.ps1 -Tier lead +# powershell.exe -ExecutionPolicy Bypass -File scripts\mfe-eval.ps1 -Tier sub +# powershell.exe -ExecutionPolicy Bypass -File scripts\mfe-eval.ps1 -Judge # show judge scaffold (no scoring) + +param( + [string]$RepoRoot = "$PSScriptRoot\..", + [ValidateSet('lead','sub','all')] [string]$Tier = 'all', + [switch]$Judge = $false, + [datetime]$Today = (Get-Date) +) + +$ErrorActionPreference = 'Stop' +$CHECK = [char]0x2705 # green check glyph, kept as code-point (ASCII body, gotcha #30) + +# --------------------------------------------------------------------------- +# resolve paths + load single-source config +# --------------------------------------------------------------------------- +$RepoRoot = (Resolve-Path $RepoRoot).Path +$budgetPath = Join-Path $RepoRoot '.claude\agent-memory\memory-budget.json' +if (-not (Test-Path $budgetPath)) { Write-Host "memory-budget.json not found: $budgetPath"; exit 0 } +$budget = Get-Content $budgetPath -Raw -Encoding UTF8 | ConvertFrom-Json +$cfg = $budget.mfe +if ($null -eq $cfg) { Write-Host "memory-budget.json missing 'mfe' config block (Harness-16)"; exit 0 } + +$leadCap = [int]$budget.token_governor.tier1_hotfeed_tokens.lead_tokens # LIVE-read, never hardcode +$memCap = [int]$budget.token_governor.tier1_hotfeed_tokens.memory_sub_tokens +$marksPath = Join-Path $RepoRoot '.claude\governance\ACTIVE-MARKS.md' +$ledgerPath = Join-Path $RepoRoot 'docs\governance\error-ledger.md' +$gotchaPath = Join-Path $RepoRoot 'docs\gotchas.md' +$statePath = Join-Path $RepoRoot $cfg.state_file + +function Read-Utf8([string]$p) { if (Test-Path $p) { return Get-Content $p -Encoding UTF8 } else { return @() } } +function Bytes([string]$s) { return [Text.Encoding]::UTF8.GetByteCount($s) } +function Cells([string]$line) { + # split a markdown table row into inner cells (drop leading/trailing empties) + $parts = $line.Trim().Trim('|') -split '\|' + return ($parts | ForEach-Object { $_.Trim() }) +} +function Remove-Diacritics([string]$t) { + $n = $t.Normalize([Text.NormalizationForm]::FormD) + $sb = New-Object Text.StringBuilder + foreach ($c in $n.ToCharArray()) { + if ([Globalization.CharUnicodeInfo]::GetUnicodeCategory($c) -ne [Globalization.UnicodeCategory]::NonSpacingMark) { [void]$sb.Append($c) } + } + return $sb.ToString().ToLowerInvariant() +} +function Content-Words([string]$s, [string[]]$stop) { + $folded = Remove-Diacritics $s + $words = $folded -split '[^a-z0-9]+' | Where-Object { $_ -and $_.Length -ge 2 -and ($stop -notcontains $_) } + return ($words | Select-Object -Unique) +} + +Write-Host "" +Write-Host "=== MFE (Memory-Fidelity-EVAL) - Harness-16 - $($Today.ToString('yyyy-MM-dd')) ===" +Write-Host " (coverage/retention eval; NOT memoryDelta-routing-fidelity H6.7)" +Write-Host " lead hot-feed cap (live) = $leadCap tok | mem-sub cap = $memCap tok" +Write-Host "" + +# =========================================================================== +# LEAD TIER : must-remember denominator -> Coverage(fit) + age-band + Goodhart +# =========================================================================== +function Measure-Lead { + # --- Source A: Active marks (status Active-High / Active) --- + $marks = @(); $markBytes = 0; $markDates = @() + $inSec = $false + foreach ($ln in (Read-Utf8 $marksPath)) { + if ($ln -match '^##\s') { $inSec = ($ln -match 'ACTIVE') ; continue } # ACTIVE-HIGH + ACTIVE; skips MEDIUM/SUPERSEDED + if ($inSec -and $ln -match '^\|\s*`(RC-[a-z0-9-]+)`') { + $rid = $matches[1] + $c = Cells $ln + $status = $c[$c.Count-1] + if ($status -match 'Active-High' -or $status -match 'Active\b') { + $marks += $rid; $markBytes += (Bytes $ln) + if ($rid -match '-(\d{2})-(\d{2})-(\d{4})-\d{2}-\d{2}-\d{2}$') { + $markDates += [datetime]::new([int]$matches[3],[int]$matches[2],[int]$matches[1]) + } + } + } + } + + # --- Source B: error-ledger Active-Guards + AS-table + RCA strikes --- + $guards = @(); $guardBytes = 0; $guardCounters = @{} + $asIds = @(); $asGotchaRefs = New-Object System.Collections.Generic.HashSet[string] + $asBytes = 0; $strikesTotal = 0; $rcaCount = 0 + $section = '' + foreach ($ln in (Read-Utf8 $ledgerPath)) { + if ($ln -match '^##\s') { + if ($ln -match 'Active-Guards') { $section = 'guards' } + elseif ($ln -match 'L\.a') { $section = 'as' } + else { $section = '' } + continue + } + if ($ln -match '^### (E-\d+)') { $rcaCount++ ; continue } + if ($section -eq 'guards' -and $ln -match '^\|' -and $ln -match [regex]::Escape($CHECK)) { + $c = Cells $ln + if ($c.Count -ge 5 -and $c[0] -notmatch '^-+$' -and $c[0] -ne 'Guard') { + $net = $c[$c.Count-1] + if ($net -notmatch '^\s*-' -and $net -notmatch 'retire') { + $g = $c[0]; $guards += $g; $guardBytes += (Bytes $ln); $guardCounters[$g] = $c[1] + if ($c[3] -match '(\d+)') { $strikesTotal += [int]$matches[1] } # Strikes col + } + } + } + if ($section -eq 'as' -and $ln -match '^\|\s*(AS-\d+)\b') { + $asIds += $matches[1]; $asBytes += (Bytes $ln) + foreach ($m in [regex]::Matches($ln, '#(\d+)')) { [void]$asGotchaRefs.Add($m.Groups[1].Value) } + } + } + + # --- Source C: recurring gotchas --- + $reTokens = @($cfg.recurrence_tokens) + $allGotchas = 0; $recurGotchaNums = @(); $recurBytes = 0 + $lines = Read-Utf8 $gotchaPath + for ($i = 0; $i -lt $lines.Count; $i++) { + if ($lines[$i] -match '^### (\d+)\.') { + $allGotchas++ + $num = $matches[1] + $blk = $lines[$i] + for ($j = $i+1; $j -lt $lines.Count -and $lines[$j] -notmatch '^### '; $j++) { $blk += "`n" + $lines[$j] } + $folded = Remove-Diacritics $blk + $isRecur = $false + foreach ($t in $reTokens) { if ($folded.Contains((Remove-Diacritics $t))) { $isRecur = $true; break } } + if ($isRecur) { $recurGotchaNums += $num; $recurBytes += (Bytes ($lines[$i])) } + } + } + + # --- DEDUP (merge_guard_to_as): collapse gotchas/guards that an AS-row references --- + $merge = [bool]$cfg.merge_guard_to_as + $netRecur = @($recurGotchaNums | Where-Object { -not $asGotchaRefs.Contains($_) }) + $netGuards = @() + foreach ($g in $guards) { + $ctr = [string]$guardCounters[$g] + $absorbed = $false + if ($merge) { + if ($ctr -match 'AS-\d+') { $absorbed = $true } + else { foreach ($m in [regex]::Matches($ctr, '#(\d+)')) { if ($asGotchaRefs.Contains($m.Groups[1].Value)) { $absorbed = $true; break } } } + } + if (-not $absorbed) { $netGuards += $g } + } + + $denomCount = $marks.Count + $asIds.Count + $netGuards.Count + $netRecur.Count + $denomBytes = $markBytes + $asBytes + $guardBytes + $recurBytes + $tokLow = [math]::Round($denomBytes / 4.0) + $tokHigh = [math]::Round($denomBytes / 3.0) + $pctHigh = if ($leadCap -gt 0) { [math]::Round(100.0 * $tokHigh / $leadCap, 2) } else { 0 } + $fits = ($tokHigh -le $leadCap) + + Write-Host "[LEAD] must-remember denominator (deduped, merge_guard_to_as=$merge):" + Write-Host " marks(Active+High)=$($marks.Count) AS-classes=$($asIds.Count) net-extra-guards=$($netGuards.Count) net-extra-recurring-gotchas=$($netRecur.Count)" + Write-Host " => DENOMINATOR = $denomCount items (raw before dedup: marks $($marks.Count) + AS $($asIds.Count) + guards $($guards.Count) + recurring $($recurGotchaNums.Count))" + Write-Host "" + Write-Host "[LEAD] Coverage-FIT (does the set FIT the hot-feed cap?):" + Write-Host " span bytes=$denomBytes => ~$tokLow - $tokHigh tok (RANGE, see caveat) / cap $leadCap tok = $pctHigh% worst-case" + if ($fits) { Write-Host " FIT = PASS (worst-case fits; headroom huge - must-remember set is tiny vs cap)" } + else { Write-Host " FIT = OVER (worst-case exceeds cap) -> decision: INCREASE budget (lack-of-SPACE case)" } + Write-Host " CAVEAT: char/4 is NOT real tokens; VN-diacritic ~3.0-3.5 byte/tok => byte/4 = upper-bound headroom." + Write-Host " Reported as a RANGE [bytes/4 .. bytes/3.0]; FIT uses bytes/3.0 (worst). Real count inside the band." + Write-Host "" + + # --- age-band : FLAG old-but-still-required, NEVER cut (mark RC-...10-29-11) --- + $oldN = 0 + foreach ($d in $markDates) { if (($Today - $d).TotalDays -gt 30) { $oldN++ } } + Write-Host "[LEAD] age-band (FLAG only - age=false-proxy, mark RC-...10-29-11):" + Write-Host " marks with parseable date=$($markDates.Count); >30d old but still Active=$oldN -> KEPT (status-driven, age-blind)" + Write-Host " (guards/gotchas carry session-refs not dates -> age = status-driven only; drop on status-change, never by age)" + Write-Host "" + + # --- Goodhart anchor : print recurring-error reality next to coverage --- + $lastStrikes = -1 + if (Test-Path $statePath) { try { $lastStrikes = [int]((Get-Content $statePath -Raw | ConvertFrom-Json).strikes_total) } catch { } } + Write-Host "[LEAD] Goodhart anchor (NO self-grading - real recurring-error signal):" + Write-Host " error-ledger: strikes_total=$strikesTotal RCA_entries=$rcaCount AS-classes=$($asIds.Count)" + if ($lastStrikes -ge 0 -and $strikesTotal -gt $lastStrikes -and $fits) { + Write-Host " GOODHART-WARN: strikes rose ($lastStrikes -> $strikesTotal) while set still FITS -> coverage may LIE (set 'fits' but errors recur = not actually applied)" + } else { + Write-Host " (rule: if strikes rise across runs while coverage stays high -> the score lies. last_run_strikes=$lastStrikes)" + } + # persist strikes for cross-run compare (MFE state only; does NOT touch token_governor) + try { + $st = @{ strikes_total = $strikesTotal; rca_entries = $rcaCount; at = $Today.ToString('yyyy-MM-dd') } + ($st | ConvertTo-Json) | Set-Content -Path $statePath -Encoding UTF8 + } catch { } + Write-Host "" +} + +# =========================================================================== +# SUB TIER : per-role coverage (denominator from role .md, numerator from diary) +# =========================================================================== +function Measure-Sub { + $stop = @($cfg.sub_denominator.stop_list) + $anchors = @($cfg.sub_denominator.denom_header_anchors) + $minWords = [int]$cfg.sub_denominator.content_word_min + $roleFiles = Get-ChildItem (Join-Path $RepoRoot '.claude\agents\*.md') -ErrorAction SilentlyContinue | Where-Object { $_.Name -ne 'README.md' } + + Write-Host "[SUB] per-role memory coverage (denominator = role-file anti-patterns/baseline; numerator = carried in diary):" + Write-Host " match = exact-id OR >=$minWords distinct content-words after stop-list (leading-verb-only = false-max, defeated)" + Write-Host "" + foreach ($rf in $roleFiles) { + $role = $rf.BaseName + $rlines = Read-Utf8 $rf.FullName + # denominator: numbered '^N.' lines UNDER a denom-anchored '##' header, + a NEVER:/scope line + $denom = @(); $inDenom = $false + foreach ($ln in $rlines) { + if ($ln -match '^##\s') { + $inDenom = $false + foreach ($a in $anchors) { if ($ln -match [regex]::Escape($a)) { $inDenom = $true; break } } + continue + } + if ($inDenom -and $ln -match '^\s*\d+\.\s+(.+)') { $denom += $matches[1] } + if ($ln -match 'NEVER' -and $ln -match ':') { $denom += ($ln -replace '.*NEVER\s*:?\s*','') } + } + $denom = @($denom | Where-Object { $_ -and $_.Trim().Length -gt 3 }) + # diary (THE agent diary - NOT project/user MEMORY.md) + $diaryPath = Join-Path $RepoRoot ".claude\agent-memory\$role\MEMORY.md" + if (-not (Test-Path $diaryPath)) { + Write-Host (" {0,-22} N/A (no persistent diary -> ephemeral/unspawned)" -f $role) + continue + } + $diaryRaw = (Read-Utf8 $diaryPath) -join "`n" + $diaryFold = Remove-Diacritics $diaryRaw + $hit = 0 + foreach ($item in $denom) { + $words = Content-Words $item $stop + $matched = $false + if ($words.Count -ge $minWords) { + $found = 0 + foreach ($w in $words) { if ($diaryFold -match ('(? N/A, NOT '0% retention' + Write-Host (" {0,-22} N/A (no numbered anti-pattern / NEVER block in role-file to extract)" -f $role) + continue + } + $pct = [math]::Round(100.0 * $hit / $den) + $flag = if ($pct -lt 60) { " <- LOW (real retention signal, not a target-miss)" } else { "" } + Write-Host (" {0,-22} {1,3}/{2,-3} = {3,3}% (MEASURED, not target-then-force){4}" -f $role, $hit, $den, $pct, $flag) + } + Write-Host "" + Write-Host "[SUB-WORKFLOW] N/A by design: fan-out workflow agents (hmw.js sub--) are EPHEMERAL" + Write-Host " (no persistent diary; their only memory channel is the lead-injected memory-pack -> source-vs-cap is meaningless)." + Write-Host "" +} + +# =========================================================================== +# JUDGE scaffold (Branch A) : -Judge only. SCAFFOLD - numbers MEANINGLESS. +# =========================================================================== +function Show-JudgeScaffold { + $sqPath = Join-Path $RepoRoot $cfg.sample_questions + Write-Host "[JUDGE-SCAFFOLD] Branch-A recall/apply (OPT-IN, cadence-gated, NO-API at scaffold level):" + if (Test-Path $sqPath) { + $sq = Get-Content $sqPath -Raw -Encoding UTF8 | ConvertFrom-Json + Write-Host " sample-questions loaded: $($sq.questions.Count) (seeded $($sq.seeded_date), anchored by stable-id)" + Write-Host " scorer = EMPTY (no answers recorded). recalled/applied = (none) ; meaningful = FALSE" + } else { + Write-Host " sample-questions file NOT found: $sqPath" + } + Write-Host " HONEST: a same-session self-grade is MEANINGLESS (the agent just read the answers)." + Write-Host " Real numbers need (a) sample-questions matured in age AND (b) an independent" + Write-Host " cross-session / different-model judge. Until then this is plumbing-smoke only." + Write-Host "" +} + +# =========================================================================== +# dispatch +# =========================================================================== +if ($Tier -eq 'lead' -or $Tier -eq 'all') { Measure-Lead } +if ($Tier -eq 'sub' -or $Tier -eq 'all') { Measure-Sub } +if ($Judge) { Show-JudgeScaffold } + +Write-Host "=== MFE done (deterministic block; READ-ONLY on budget; exit 0) ===" +exit 0