feat(ai): deterministic element-kind hints for manifest sub-agents
ab-v9.1 misses were attention failures, not vocabulary failures: kinds named literally in the brief (toolbar 5/10) still got hand-composed, while anti-pattern-named kinds rarely missed. Token-match the subtask label+elements text (plus the whole brief for single-section plans) against the catalog, expand via a data-grounded synonym table, and inject up to 6 nominations as an ELEMENT HINTS block with composite- first and no-hand-compose rules. ab-v9.2 matrix: 51 FAIL->PASS vs 4 noise-shaped regressions; all four arms >=83% M3-pass, M5 adoption 96%.
This commit is contained in:
parent
461b77a054
commit
fa4130c133
|
|
@ -16,6 +16,7 @@ pub mod design_system;
|
|||
pub mod design_type;
|
||||
pub mod intent;
|
||||
pub mod manifest;
|
||||
pub mod manifest_hints;
|
||||
pub mod model_profile;
|
||||
pub mod parse;
|
||||
pub mod plan;
|
||||
|
|
|
|||
218
crates/op-orchestrator/src/manifest_hints.rs
Normal file
218
crates/op-orchestrator/src/manifest_hints.rs
Normal file
|
|
@ -0,0 +1,218 @@
|
|||
//! Deterministic element-kind nomination for the manifest sub-agent prompt.
|
||||
//!
|
||||
//! ab-v9.1 (openpencil-docs/model-benchmarks/2026-06-10-ab-v9-manifest)
|
||||
//! showed element-adoption misses concentrate on briefs that DESCRIBE a
|
||||
//! component instead of naming it (`setting_row` 10/10 missed) — yet the
|
||||
//! literal kind tokens are almost always present in the planner's own
|
||||
//! subtask text ("Notification Settings Row" / "toggle switch"), and even
|
||||
//! a literal "toolbar" in the brief was missed 5/10. Recognition fails on
|
||||
//! attention, not vocabulary — so this module does the recognition
|
||||
//! deterministically: token-match catalog kind names against the subtask
|
||||
//! text (plus a small synonym table for visual idioms that never carry
|
||||
//! the kind's name) and surface the matches as ELEMENT HINTS lines in the
|
||||
//! sub-agent user prompt.
|
||||
|
||||
use std::collections::BTreeSet;
|
||||
|
||||
/// Hard cap on hint count — hints are an anchor, not a checklist; past a
|
||||
/// handful they read as noise and dilute the strong matches.
|
||||
const MAX_HINTS: usize = 6;
|
||||
|
||||
/// Visual idiom → catalog kind, for kinds whose names don't appear
|
||||
/// literally in the way people describe them. Grounded in the ab-v3
|
||||
/// corpus briefs that missed: each phrase is how the brief actually
|
||||
/// said it, kept generic enough to transfer.
|
||||
const SYNONYM_PHRASES: &[(&str, &str)] = &[
|
||||
("verification code", "otp_input"),
|
||||
("one time password", "otp_input"),
|
||||
("passcode", "otp_input"),
|
||||
("pending invitation", "invite_row"),
|
||||
("stacked avatar", "avatar_group"),
|
||||
("overlapping avatar", "avatar_group"),
|
||||
("avatar tile", "avatar_group"),
|
||||
("facepile", "avatar_group"),
|
||||
("profile hero", "profile_header"),
|
||||
("profile page hero", "profile_header"),
|
||||
("status indicator", "status_badge"),
|
||||
("status pill", "status_badge"),
|
||||
("status chip", "status_badge"),
|
||||
("status dot", "status_badge"),
|
||||
("metric cell", "metric_comparison"),
|
||||
("kpi cell", "metric_comparison"),
|
||||
("settings menu", "setting_row"),
|
||||
("preference row", "setting_row"),
|
||||
("hover hint", "tooltip"),
|
||||
];
|
||||
|
||||
/// Nominate catalog kinds the text plausibly asks for, most specific
|
||||
/// first, capped at [`MAX_HINTS`].
|
||||
///
|
||||
/// A kind token-matches when ALL of its `_`-separated name tokens appear
|
||||
/// as words in the text (case- and trailing-plural-insensitive), so
|
||||
/// "phone settings menu — one row" fires `setting_row` and "data-table
|
||||
/// body row" fires `data_table_row`. Synonym phrases match as
|
||||
/// space-bounded substrings (also tolerating a plural on the last word).
|
||||
/// Specificity = matched token / phrase word count; multi-token kinds
|
||||
/// outrank single-token ones so a composite (`setting_row`) lists before
|
||||
/// its part (`switch`).
|
||||
pub fn nominate_kinds(text: &str, kinds: &[String]) -> Vec<String> {
|
||||
let canon = canonical_text(text);
|
||||
if canon.is_empty() {
|
||||
return Vec::new();
|
||||
}
|
||||
let tokens: BTreeSet<String> = canon.split_whitespace().map(normalize_token).collect();
|
||||
|
||||
// (specificity, kind) — token matches first.
|
||||
let mut scored: Vec<(usize, &str)> = Vec::new();
|
||||
for kind in kinds {
|
||||
let name_tokens: Vec<String> = kind.split('_').map(normalize_token).collect();
|
||||
if !name_tokens.is_empty() && name_tokens.iter().all(|t| tokens.contains(t)) {
|
||||
scored.push((name_tokens.len(), kind.as_str()));
|
||||
}
|
||||
}
|
||||
|
||||
let padded = format!(" {canon} ");
|
||||
for (phrase, kind) in SYNONYM_PHRASES {
|
||||
// Catalog drift guard: never hint a kind the catalog can't build.
|
||||
if !kinds.iter().any(|k| k == kind) {
|
||||
continue;
|
||||
}
|
||||
let hit =
|
||||
padded.contains(&format!(" {phrase} ")) || padded.contains(&format!(" {phrase}s "));
|
||||
if hit {
|
||||
let words = phrase.split_whitespace().count();
|
||||
match scored.iter_mut().find(|(_, k)| k == kind) {
|
||||
Some(entry) => entry.0 = entry.0.max(words),
|
||||
None => scored.push((words, kind)),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
scored.sort_by(|a, b| b.0.cmp(&a.0).then(a.1.cmp(b.1)));
|
||||
scored.truncate(MAX_HINTS);
|
||||
scored.into_iter().map(|(_, k)| k.to_string()).collect()
|
||||
}
|
||||
|
||||
/// Lowercase, every non-alphanumeric run collapsed to a single space —
|
||||
/// so "data-table", "data_table", and "Data Table" all canonicalize the
|
||||
/// same way for both token and phrase matching.
|
||||
fn canonical_text(text: &str) -> String {
|
||||
let mut out = String::with_capacity(text.len());
|
||||
let mut last_space = true;
|
||||
for c in text.chars() {
|
||||
if c.is_ascii_alphanumeric() {
|
||||
out.push(c.to_ascii_lowercase());
|
||||
last_space = false;
|
||||
} else if !last_space {
|
||||
out.push(' ');
|
||||
last_space = true;
|
||||
}
|
||||
}
|
||||
out.trim_end().to_string()
|
||||
}
|
||||
|
||||
/// Trailing-plural-insensitive token form, applied symmetrically to both
|
||||
/// text and kind-name tokens ("settings" and "setting" both → "setting";
|
||||
/// "progress" keeps its double-s).
|
||||
fn normalize_token(t: &str) -> String {
|
||||
let t = t.to_ascii_lowercase();
|
||||
if t.len() > 3 && t.ends_with('s') && !t.ends_with("ss") {
|
||||
t[..t.len() - 1].to_string()
|
||||
} else {
|
||||
t
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn known() -> Vec<String> {
|
||||
op_mcp::element_manifest::known_element_kinds()
|
||||
}
|
||||
|
||||
/// Kind-name tokens in the text fire the kind, plural- and
|
||||
/// punctuation-insensitive (the mobile-setting-row stable failure).
|
||||
#[test]
|
||||
fn token_match_fires_on_literal_kind_words() {
|
||||
let hints = nominate_kinds(
|
||||
"Design ONLY one row of a phone settings menu — the Notifications preference",
|
||||
&known(),
|
||||
);
|
||||
assert!(hints.contains(&"setting_row".to_string()), "{hints:?}");
|
||||
|
||||
let hints = nominate_kinds(
|
||||
"a single body row of a desktop dashboard customers data-table",
|
||||
&known(),
|
||||
);
|
||||
assert!(hints.contains(&"data_table_row".to_string()), "{hints:?}");
|
||||
}
|
||||
|
||||
/// Synonym phrases cover idioms that never carry the kind name
|
||||
/// (otp_input / avatar_group / status_badge stable failures).
|
||||
#[test]
|
||||
fn synonym_phrases_fire_without_kind_words() {
|
||||
let cases: &[(&str, &str)] = &[
|
||||
("a 6-digit verification code input", "otp_input"),
|
||||
("a stacked-avatar presence indicator", "avatar_group"),
|
||||
("a status indicator: a small green dot", "status_badge"),
|
||||
("one row of a Pending invitations table", "invite_row"),
|
||||
];
|
||||
for (text, kind) in cases {
|
||||
let hints = nominate_kinds(text, &known());
|
||||
assert!(hints.contains(&kind.to_string()), "{text} → {hints:?}");
|
||||
}
|
||||
}
|
||||
|
||||
/// Composite kinds (more name tokens) rank before their parts so the
|
||||
/// "most specific wins" prompt clause reads in hint order.
|
||||
#[test]
|
||||
fn multi_token_kinds_rank_before_single_token() {
|
||||
let hints = nominate_kinds(
|
||||
"Notification Settings Row with a trailing toggle switch",
|
||||
&known(),
|
||||
);
|
||||
let row = hints.iter().position(|k| k == "setting_row").unwrap();
|
||||
let sw = hints.iter().position(|k| k == "switch").unwrap();
|
||||
assert!(row < sw, "{hints:?}");
|
||||
}
|
||||
|
||||
/// Hint volume is capped — a noun-soup brief can't flood the prompt.
|
||||
#[test]
|
||||
fn hints_cap_at_six() {
|
||||
let hints = nominate_kinds(
|
||||
"tag badge switch checkbox avatar divider spinner toast tabs link",
|
||||
&known(),
|
||||
);
|
||||
assert_eq!(hints.len(), MAX_HINTS);
|
||||
}
|
||||
|
||||
/// No matches (or empty text) → empty, so the prompt block is omitted.
|
||||
#[test]
|
||||
fn no_match_returns_empty() {
|
||||
assert!(nominate_kinds("a pricing page", &known()).is_empty());
|
||||
assert!(nominate_kinds(" ", &known()).is_empty());
|
||||
}
|
||||
|
||||
/// Every synonym target must exist in the live catalog; a rename in
|
||||
/// op-mcp should fail here, not silently stop hinting.
|
||||
#[test]
|
||||
fn synonym_targets_resolve_in_catalog() {
|
||||
let kinds = known();
|
||||
for (phrase, kind) in SYNONYM_PHRASES {
|
||||
assert!(
|
||||
kinds.iter().any(|k| k == kind),
|
||||
"synonym \"{phrase}\" targets unknown kind {kind}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// The drift guard drops synonym hits whose kind is missing from the
|
||||
/// provided catalog slice instead of hinting an unbuildable kind.
|
||||
#[test]
|
||||
fn synonym_skipped_when_kind_absent() {
|
||||
let only_switch = vec!["switch".to_string()];
|
||||
let hints = nominate_kinds("a 6-digit verification code input", &only_switch);
|
||||
assert!(hints.is_empty(), "{hints:?}");
|
||||
}
|
||||
}
|
||||
|
|
@ -484,6 +484,31 @@ fn build_subagent_prompt_with_manifest(
|
|||
})
|
||||
.unwrap_or_default();
|
||||
|
||||
// Deterministic element nomination (ab-v9.1 de-randomization lever):
|
||||
// weak models miss catalog kinds whose names sit literally in the
|
||||
// brief (setting_row 10/10, toolbar 5/10 despite the word "toolbar").
|
||||
// Token-matching the subtask's own text pulls that recognition step
|
||||
// out of the model's attention. Multi-section plans scan only the
|
||||
// subtask-local text so one section's nouns don't pollute another's
|
||||
// hints; single-section plans scan the whole brief too.
|
||||
let element_hints = if manifest_on {
|
||||
let mut hint_src = subtask.label.clone();
|
||||
if let Some(items) = subtask.elements.as_ref() {
|
||||
hint_src.push(' ');
|
||||
hint_src.push_str(items);
|
||||
}
|
||||
if plan.subtasks.len() <= 1 {
|
||||
hint_src.push(' ');
|
||||
hint_src.push_str(&req.prompt);
|
||||
}
|
||||
crate::manifest_hints::nominate_kinds(
|
||||
&hint_src,
|
||||
&op_mcp::element_manifest::known_element_kinds(),
|
||||
)
|
||||
} else {
|
||||
Vec::new()
|
||||
};
|
||||
|
||||
// Three constraints differ by output protocol: the raw-JSONL path has
|
||||
// the model author its own root frame + ids; the manifest path forbids
|
||||
// exactly that (system-assigned ids, system-owned section root).
|
||||
|
|
@ -525,6 +550,16 @@ CRITICAL LAYOUT CONSTRAINTS:\n\
|
|||
output_rule,
|
||||
);
|
||||
|
||||
if !element_hints.is_empty() {
|
||||
user_prompt.push_str(&format!(
|
||||
"\n\nELEMENT HINTS: this section's brief matches these catalog kinds: {}.\n\
|
||||
- Check each hinted kind in the catalog and declare it as its own {{\"el\":...}} line unless it clearly does not fit this section.\n\
|
||||
- Most specific wins: when a composite hinted kind already contains a smaller hinted one (a setting_row already has its switch), declare ONLY the composite.\n\
|
||||
- NEVER hand-compose a hinted kind out of section/text/icon_font lines.",
|
||||
element_hints.join(", ")
|
||||
));
|
||||
}
|
||||
|
||||
if plan.root_frame.width <= 480.0 {
|
||||
user_prompt.push_str(
|
||||
"\n\nMOBILE STATUS BAR: A status bar (time, signal, wifi, battery) has already been pre-inserted as the first child of the root page frame. Do NOT generate any status bar, system chrome, or OS-level indicators. Start your content directly.",
|
||||
|
|
|
|||
|
|
@ -745,3 +745,47 @@ fn subagent_prompt_manifest_mode_swaps_output_protocol() {
|
|||
assert!(!off.system_prompt.contains(MANIFEST_FORMAT_ONLY));
|
||||
assert!(!off.system_prompt.contains(MANIFEST_SKILL_ONLY));
|
||||
}
|
||||
|
||||
/// Manifest mode nominates catalog kinds from the subtask's own text and
|
||||
/// injects them as an ELEMENT HINTS block; raw mode and hint-less
|
||||
/// subtasks stay clean (ab-v9.1 adoption de-randomization).
|
||||
#[test]
|
||||
fn subagent_prompt_manifest_mode_injects_element_hints() {
|
||||
let mut st = subtask();
|
||||
st.label = "Notification Settings Row".into();
|
||||
st.elements = Some("bell icon, text stack, iOS toggle switch".into());
|
||||
let build = |st: &crate::plan::Subtask, manifest_on: bool| {
|
||||
build_subagent_prompt_with_manifest(
|
||||
st,
|
||||
&plan(),
|
||||
&req(),
|
||||
AbortFlag::new(),
|
||||
false,
|
||||
false,
|
||||
manifest_on,
|
||||
)
|
||||
};
|
||||
|
||||
let cr = build(&st, true);
|
||||
assert!(
|
||||
cr.user_prompt.contains("ELEMENT HINTS:"),
|
||||
"hint block loads"
|
||||
);
|
||||
assert!(
|
||||
cr.user_prompt.contains("setting_row"),
|
||||
"composite kind hinted"
|
||||
);
|
||||
assert!(cr.user_prompt.contains("switch"), "part kind hinted");
|
||||
|
||||
let raw = build(&st, false);
|
||||
assert!(
|
||||
!raw.user_prompt.contains("ELEMENT HINTS"),
|
||||
"raw JSONL mode has no catalog to hint from"
|
||||
);
|
||||
|
||||
let none = build(&subtask(), true);
|
||||
assert!(
|
||||
!none.user_prompt.contains("ELEMENT HINTS"),
|
||||
"no matches must omit the block, not emit it empty"
|
||||
);
|
||||
}
|
||||
|
|
|
|||
Loading…
Reference in a new issue