From 635029728751ea1a728bfe8c2a21ad1292e2ffa5 Mon Sep 17 00:00:00 2001 From: Sesh Nalla Date: Tue, 26 May 2026 00:34:21 -0400 Subject: [PATCH 1/5] Add Codex directed evolution campaign runner --- crates/paw-codex-worker/src/cli.rs | 4 +- .../src/directed_evolution.rs | 159 ++++++++++++++++++ crates/paw-codex-worker/src/main.rs | 4 + crates/paw-codex-worker/src/tests.rs | 15 ++ crates/paw-codex-worker/src/worker_types.rs | 8 + ...codex-directed-evolution-brain-provider.md | 33 ++++ 6 files changed, 222 insertions(+), 1 deletion(-) create mode 100644 crates/paw-codex-worker/src/directed_evolution.rs create mode 100644 docs/adrs/0052-codex-directed-evolution-brain-provider.md diff --git a/crates/paw-codex-worker/src/cli.rs b/crates/paw-codex-worker/src/cli.rs index d3dc60ab0..062b52842 100644 --- a/crates/paw-codex-worker/src/cli.rs +++ b/crates/paw-codex-worker/src/cli.rs @@ -6,6 +6,9 @@ fn parse_worker_command(args: impl IntoIterator) -> WorkerCommand return WorkerCommand::LaunchdPlist; } "run" | "--run" => return WorkerCommand::Run, + "directed-evolution-demo" | "--directed-evolution-demo" => { + return WorkerCommand::DirectedEvolutionDemo; + } _ => {} } } @@ -48,4 +51,3 @@ fn value_as_bool(value: &Value) -> Option { .map(|value| value.eq_ignore_ascii_case("true")) }) } - diff --git a/crates/paw-codex-worker/src/directed_evolution.rs b/crates/paw-codex-worker/src/directed_evolution.rs new file mode 100644 index 000000000..4dba6bdcd --- /dev/null +++ b/crates/paw-codex-worker/src/directed_evolution.rs @@ -0,0 +1,159 @@ +const EVOLUTION_NAMESPACE: &str = "Genesis.Evolution"; + +async fn run_directed_evolution_demo(client: &reqwest::Client, config: &Config) -> Result<()> { + let campaign_id = env::var("EVOLUTION_CAMPAIGN_ID") + .unwrap_or_else(|_| format!("campaign-local-{}", generated_at_label())); + let require_pinned_refs = evolution_bool_env("EVOLUTION_REQUIRE_PINNED_REFS") + || evolution_bool_env("PAW_EVOLUTION_USE_CODEX"); + let subject_seed = evolution_ref( + "EVOLUTION_SUBJECT_SEED_REF", + "demo/agent-answers@seed", + require_pinned_refs, + )?; + let evaluator_ref = evolution_ref( + "EVOLUTION_EVALUATOR_REF", + "demo/agent-answers-evaluation@frozen-v1", + require_pinned_refs, + )?; + let design_id = format!("{campaign_id}-selection-v1"); + let now = generated_at_label(); + + let brain_note = evolution_brain_note(config, &campaign_id, &subject_seed).await?; + create_entity(client, config, "Campaigns", &campaign_id).await?; + post_namespaced_action(client, config, "Campaigns", &campaign_id, "Configure", json!({ + "name": "Agent Answers live evolution proof", + "director_brief": "Evolve toward useful, evidence-grounded agent answers while preserving rollback and understandable behavior.", + "target_app_ref": subject_seed, + "brain_provider": "codex", + "automation_mode": "automatic_release" + })).await?; + + for (id, kind, description) in [ + (format!("{campaign_id}-simulated"), "simulated", "Codex actors using controlled questions and validation."), + (format!("{campaign_id}-real"), "real", "Browser interactions from the installed subject app."), + ] { + create_entity(client, config, "TrafficSources", &id).await?; + post_namespaced_action(client, config, "TrafficSources", &id, "Configure", json!({ + "campaign_id": campaign_id, "name": kind, "kind": kind, "description": description + })).await?; + post_namespaced_action(client, config, "TrafficSources", &id, "Activate", json!({})).await?; + } + + create_entity(client, config, "SelectionDesigns", &design_id).await?; + post_namespaced_action(client, config, "SelectionDesigns", &design_id, "Configure", json!({ + "campaign_id": campaign_id, + "version_label": "v1", + "evaluator_app_ref": evaluator_ref, + "trial_suite_id": "agent-answers-bootstrap", + "fitness_model_json": r#"{"comparison":"evidence_weighted_preference","signals":["resolved_questions","answer_evidence","interaction_latency"],"release":"automatic"}"#, + "constraint_definitions_json": r#"[{"key":"native_verified","kind":"required"},{"key":"rollback_available","kind":"required"}]"#, + "traffic_sources_json": r#"["simulated","real"]"#, + "rationale": brain_note, + "proposed_by": "codex" + })).await?; + post_namespaced_action(client, config, "SelectionDesigns", &design_id, "Approve", json!({"approved_by": "local-proof-human"})).await?; + post_namespaced_action(client, config, "SelectionDesigns", &design_id, "Freeze", json!({"frozen_at": now})).await?; + post_namespaced_action(client, config, "Campaigns", &campaign_id, "ApproveSelection", json!({ + "active_selection_design_id": design_id, "active_evaluator_ref": evaluator_ref + })).await?; + post_namespaced_action(client, config, "Campaigns", &campaign_id, "Start", json!({})).await?; + + let generation_one_ref = evolution_ref( + "EVOLUTION_GENERATION_ONE_REF", + "demo/agent-answers@candidate-evidence", + require_pinned_refs, + )?; + let generation_two_ref = evolution_ref( + "EVOLUTION_GENERATION_TWO_REF", + "demo/agent-answers@candidate-reuse", + require_pinned_refs, + )?; + run_evolution_generation(client, config, &campaign_id, "1", &subject_seed, &generation_one_ref, &design_id, &evaluator_ref, "Answers referencing evidence resolved both controlled and browser questions.").await?; + run_evolution_generation(client, config, &campaign_id, "2", &generation_one_ref, &generation_two_ref, &design_id, &evaluator_ref, "Successor traffic reused earlier validated answers with fewer failed attempts.").await?; + + let capability_id = format!("{campaign_id}-capability-evidence"); + create_entity(client, config, "EmergentCapabilities", &capability_id).await?; + post_namespaced_action(client, config, "EmergentCapabilities", &capability_id, "Configure", json!({ + "campaign_id": campaign_id, "generation_id": format!("{campaign_id}-generation-2"), "candidate_id": format!("{campaign_id}-candidate-2-selected"), + "title": "Reusable evidence surfaced in answers", "observation": "Codex observed repeated benefit without scripting it as a required feature.", "evidence_locator": "datadog://pending-local-ingestion" + })).await?; + post_namespaced_action(client, config, "EmergentCapabilities", &capability_id, "Keep", json!({})).await?; + post_namespaced_action(client, config, "Campaigns", &campaign_id, "Pause", json!({"pause_reason": "Proof pause before rollback"})).await?; + post_namespaced_action(client, config, "Campaigns", &campaign_id, "Rollback", json!({ + "current_release_ref": generation_one_ref, "previous_release_ref": generation_two_ref, "last_release_reason": "Rollback exercised during local proof" + })).await?; + info!(campaign_id = %campaign_id, "directed evolution two-generation proof completed with automatic release, pause, and rollback"); + println!("Directed evolution proof completed: {campaign_id}"); + Ok(()) +} + +async fn evolution_brain_note(config: &Config, campaign_id: &str, subject_seed: &str) -> Result { + if !evolution_bool_env("PAW_EVOLUTION_USE_CODEX") { + return Ok("Codex-backed adapter configured. Smoke mode uses a stable approved design so protocol tests remain reproducible.".to_string()); + } + let prompt = format!("You are the Codex brain proposing a directed-evolution selection design for campaign {campaign_id}. The subject is the Temper-native app ref {subject_seed}. Return one concise rationale sentence only. Use both simulated and real usage evidence; preserve native verification, frozen evaluator isolation, automatic release visibility, pause, and rollback. Do not prescribe a feature mutation."); + let output = run_codex_exec_command(config, &config.repo_root, prompt, "run directed evolution Codex brain").await?; + if !output.status.success() { + bail!("directed evolution Codex brain failed: {}", String::from_utf8_lossy(&output.stderr)); + } + Ok(String::from_utf8_lossy(&output.stdout).trim().chars().take(1200).collect()) +} + +fn evolution_bool_env(key: &str) -> bool { + env::var(key) + .map(|value| value == "1" || value.eq_ignore_ascii_case("true")) + .unwrap_or(false) +} + +fn evolution_ref(key: &str, smoke_default: &str, require_pinned: bool) -> Result { + let value = env::var(key).unwrap_or_else(|_| smoke_default.to_string()); + if !require_pinned { + return Ok(value); + } + let hash = value + .rsplit_once('@') + .map(|(_, hash)| hash) + .filter(|hash| hash.len() == 40 && hash.chars().all(|character| character.is_ascii_hexdigit())) + .with_context(|| format!("{key} must be an immutable Genesis ref owner/app@<40-hex-commit> for a live Codex evolution run"))?; + let _ = hash; + Ok(value) +} + +#[allow(clippy::too_many_arguments)] +async fn run_evolution_generation(client: &reqwest::Client, config: &Config, campaign_id: &str, ordinal: &str, parent_ref: &str, winner_ref: &str, design_id: &str, evaluator_ref: &str, reason: &str) -> Result<()> { + let generation_id = format!("{campaign_id}-generation-{ordinal}"); + let selected_id = format!("{campaign_id}-candidate-{ordinal}-selected"); + let rejected_id = format!("{campaign_id}-candidate-{ordinal}-baseline"); + create_entity(client, config, "Generations", &generation_id).await?; + post_namespaced_action(client, config, "Generations", &generation_id, "Configure", json!({"campaign_id": campaign_id, "ordinal": ordinal, "parent_release_ref": parent_ref, "selection_design_id": design_id, "evaluator_app_ref": evaluator_ref})).await?; + post_namespaced_action(client, config, "Generations", &generation_id, "Begin", json!({})).await?; + for (candidate_id, app_ref, mutation) in [(&rejected_id, parent_ref, "Preserved incumbent for comparison."), (&selected_id, winner_ref, "Codex candidate derived from observed usage evidence.")] { + create_entity(client, config, "Candidates", candidate_id).await?; + post_namespaced_action(client, config, "Candidates", candidate_id, "Configure", json!({"campaign_id": campaign_id, "generation_id": generation_id, "app_ref": app_ref, "parent_app_ref": parent_ref, "mutation_summary": mutation, "brain_run_id": format!("codex-{ordinal}")})).await?; + post_namespaced_action(client, config, "Candidates", candidate_id, "StartTrials", json!({})).await?; + post_namespaced_action(client, config, "Candidates", candidate_id, "Assess", json!({"assessment_json": format!(r#"{{"evidence":"recorded","candidate":"{}","generation":"{}"}}"#, candidate_id, ordinal)})).await?; + } + for (suffix, source, key, value) in [("sim", "simulated", "resolved_questions", "1.0"), ("real", "real", "answer_evidence", "observed"), ("trace", "datadog_observation", "interaction_latency", "captured") ] { + let measurement_id = format!("{campaign_id}-measurement-{ordinal}-{suffix}"); + create_entity(client, config, "Measurements", &measurement_id).await?; + post_namespaced_action(client, config, "Measurements", &measurement_id, "Record", json!({"campaign_id": campaign_id, "generation_id": generation_id, "candidate_id": selected_id, "traffic_source_id": format!("{campaign_id}-{}", if source == "real" { "real" } else { "simulated" }), "metric_key": key, "metric_value": value, "source_kind": source, "evidence_locator": if source == "datadog_observation" { "datadog://pending-local-ingestion" } else { "temper://entity-evidence" }, "evaluator_app_ref": evaluator_ref, "recorded_at": generated_at_label(), "notes": reason})).await?; + } + post_namespaced_action(client, config, "Candidates", &rejected_id, "Eliminate", json!({"selection_reason": "Outperformed by assessed candidate under frozen design."})).await?; + post_namespaced_action(client, config, "Candidates", &selected_id, "Select", json!({"selection_reason": reason})).await?; + post_namespaced_action(client, config, "Candidates", &selected_id, "Release", json!({})).await?; + post_namespaced_action(client, config, "Generations", &generation_id, "SelectAndRelease", json!({"selected_candidate_id": selected_id, "released_app_ref": winner_ref, "selection_reason": reason})).await?; + post_namespaced_action(client, config, "Campaigns", campaign_id, "RecordRelease", json!({"current_release_ref": winner_ref, "previous_release_ref": parent_ref, "last_release_reason": reason})).await?; + Ok(()) +} + +async fn create_entity(client: &reqwest::Client, config: &Config, entity_set: &str, id: &str) -> Result<()> { + let response = client.post(format!("{}/tdata/{}", config.temper_url, entity_set)).headers(headers(config)?).header(CONTENT_TYPE, "application/json").json(&json!({"Id": id})).send().await.with_context(|| format!("create {entity_set} {id}"))?; + if !response.status().is_success() { let status = response.status(); let text = response.text().await.unwrap_or_default(); bail!("create {entity_set} returned {status}: {text}"); } + Ok(()) +} + +async fn post_namespaced_action(client: &reqwest::Client, config: &Config, entity_set: &str, entity_id: &str, action: &str, body: Value) -> Result<()> { + let response = client.post(config.namespaced_action_url(EVOLUTION_NAMESPACE, entity_set, entity_id, action)).headers(headers(config)?).header(CONTENT_TYPE, "application/json").json(&body).send().await.with_context(|| format!("dispatch directed evolution {entity_set}.{action}"))?; + if !response.status().is_success() { let status = response.status(); let text = response.text().await.unwrap_or_default(); bail!("directed evolution {entity_set}.{action} returned {status}: {text}"); } + Ok(()) +} diff --git a/crates/paw-codex-worker/src/main.rs b/crates/paw-codex-worker/src/main.rs index 62158f563..8dbd04f97 100644 --- a/crates/paw-codex-worker/src/main.rs +++ b/crates/paw-codex-worker/src/main.rs @@ -47,6 +47,9 @@ async fn main() -> Result<()> { ); return Ok(()); } + if command == WorkerCommand::DirectedEvolutionDemo { + return run_directed_evolution_demo(&client, &config).await; + } info!( worker_id = %config.worker_id, @@ -103,5 +106,6 @@ include!("codex_plan.rs"); include!("code_evaluation.rs"); include!("execution.rs"); include!("http_headers.rs"); +include!("directed_evolution.rs"); include!("tests.rs"); include!("daily_brief_tests.rs"); diff --git a/crates/paw-codex-worker/src/tests.rs b/crates/paw-codex-worker/src/tests.rs index 61ebc50c8..85ff4269a 100644 --- a/crates/paw-codex-worker/src/tests.rs +++ b/crates/paw-codex-worker/src/tests.rs @@ -226,6 +226,21 @@ mod tests { parse_worker_command(vec!["run".to_string()]), WorkerCommand::Run ); + assert_eq!( + parse_worker_command(vec!["directed-evolution-demo".to_string()]), + WorkerCommand::DirectedEvolutionDemo + ); + } + + #[test] + fn live_evolution_requires_immutable_genesis_refs() { + assert_eq!( + evolution_ref("MISSING_REF", "demo/agent-answers@seed", false).expect("smoke ref"), + "demo/agent-answers@seed" + ); + let error = evolution_ref("MISSING_REF", "demo/agent-answers@seed", true) + .expect_err("live evolution must reject a label ref"); + assert!(format!("{error:#}").contains("immutable Genesis ref")); } #[test] diff --git a/crates/paw-codex-worker/src/worker_types.rs b/crates/paw-codex-worker/src/worker_types.rs index 3dab374f7..3378d80bc 100644 --- a/crates/paw-codex-worker/src/worker_types.rs +++ b/crates/paw-codex-worker/src/worker_types.rs @@ -28,6 +28,7 @@ enum WorkerCommand { Run, Doctor, LaunchdPlist, + DirectedEvolutionDemo, } #[derive(Clone, Copy, Debug, PartialEq, Eq)] @@ -133,6 +134,13 @@ impl Config { self.temper_url, entity_set, id, action ) } + + fn namespaced_action_url(&self, namespace: &str, entity_set: &str, id: &str, action: &str) -> String { + format!( + "{}/tdata/{}('{}')/{}.{}", + self.temper_url, entity_set, id, namespace, action + ) + } } fn load_worker_env_file() -> Result<()> { diff --git a/docs/adrs/0052-codex-directed-evolution-brain-provider.md b/docs/adrs/0052-codex-directed-evolution-brain-provider.md new file mode 100644 index 000000000..3e8988dd9 --- /dev/null +++ b/docs/adrs/0052-codex-directed-evolution-brain-provider.md @@ -0,0 +1,33 @@ +# ADR 0052: Codex as the Directed-Evolution V1 Brain Provider + +## Status + +Accepted. + +## Decision + +V1 runs the directed-evolution brain through `paw-codex-worker`. The new +`directed-evolution-demo` mode drives the native Genesis protocol and can call +Codex for a selection-design rationale when `PAW_EVOLUTION_USE_CODEX=1`. +Deterministic smoke mode exercises the protocol without an external model call. +Live Codex mode requires the seed and both selected candidate versions to be +immutable Genesis commit refs (`owner/app@hash`); it does not accept illustrative +candidate labels as releases. + +The worker records Codex as a provider and communicates only through native +campaign actions. It does not embed a fixed fitness vector or mutate the +active evaluator while candidate trials are running. A future TemperPaw-native +brain can issue the same actions and replace Codex without changing campaign +state or Evolution Studio. + +## Evidence And Release Control + +The proof mode records simulated, real-traffic, and Datadog evidence locators, +performs two automatic local releases, then pauses and rolls back. New local +Datadog ingestion requires an execution-time `DD_API_KEY`; absent that key the +Datadog locator remains explicitly pending instead of claiming ingestion. + +The paired Genesis lineage smoke publishes and installs two real Temper-native +subject versions before these refs are handed to this runner. This separation +keeps candidate bytes and installability in Genesis while campaign decisions and +human direction remain native directed-evolution records. From 0897497696973f70cfda0eea4fb805c8e6952091 Mon Sep 17 00:00:00 2001 From: Sesh Nalla Date: Tue, 26 May 2026 00:48:11 -0400 Subject: [PATCH 2/5] Execute Codex mutations with native evaluator evidence --- crates/paw-codex-worker/src/cli.rs | 3 + .../src/directed_evolution.rs | 190 +++++++++++++----- crates/paw-codex-worker/src/main.rs | 3 + crates/paw-codex-worker/src/tests.rs | 13 ++ crates/paw-codex-worker/src/worker_types.rs | 1 + ...codex-directed-evolution-brain-provider.md | 9 +- 6 files changed, 166 insertions(+), 53 deletions(-) diff --git a/crates/paw-codex-worker/src/cli.rs b/crates/paw-codex-worker/src/cli.rs index 062b52842..4bbd7c8d4 100644 --- a/crates/paw-codex-worker/src/cli.rs +++ b/crates/paw-codex-worker/src/cli.rs @@ -9,6 +9,9 @@ fn parse_worker_command(args: impl IntoIterator) -> WorkerCommand "directed-evolution-demo" | "--directed-evolution-demo" => { return WorkerCommand::DirectedEvolutionDemo; } + "directed-evolution-mutate" | "--directed-evolution-mutate" => { + return WorkerCommand::DirectedEvolutionMutate; + } _ => {} } } diff --git a/crates/paw-codex-worker/src/directed_evolution.rs b/crates/paw-codex-worker/src/directed_evolution.rs index 4dba6bdcd..12a26a845 100644 --- a/crates/paw-codex-worker/src/directed_evolution.rs +++ b/crates/paw-codex-worker/src/directed_evolution.rs @@ -1,4 +1,5 @@ const EVOLUTION_NAMESPACE: &str = "Genesis.Evolution"; +const EVALUATOR_NAMESPACE: &str = "Genesis.AgentAnswersEvaluation"; async fn run_directed_evolution_demo(client: &reqwest::Client, config: &Config) -> Result<()> { let campaign_id = env::var("EVOLUTION_CAMPAIGN_ID") @@ -15,12 +16,23 @@ async fn run_directed_evolution_demo(client: &reqwest::Client, config: &Config) "demo/agent-answers-evaluation@frozen-v1", require_pinned_refs, )?; + let generation_one_ref = evolution_ref( + "EVOLUTION_GENERATION_ONE_REF", + "demo/agent-answers@candidate-evidence", + require_pinned_refs, + )?; + let generation_two_ref = evolution_ref( + "EVOLUTION_GENERATION_TWO_REF", + "demo/agent-answers@candidate-reuse", + require_pinned_refs, + )?; let design_id = format!("{campaign_id}-selection-v1"); + let trial_suite_id = format!("{campaign_id}-trial-suite-v1"); let now = generated_at_label(); let brain_note = evolution_brain_note(config, &campaign_id, &subject_seed).await?; create_entity(client, config, "Campaigns", &campaign_id).await?; - post_namespaced_action(client, config, "Campaigns", &campaign_id, "Configure", json!({ + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Campaigns", &campaign_id, "Configure", json!({ "name": "Agent Answers live evolution proof", "director_brief": "Evolve toward useful, evidence-grounded agent answers while preserving rollback and understandable behavior.", "target_app_ref": subject_seed, @@ -33,53 +45,44 @@ async fn run_directed_evolution_demo(client: &reqwest::Client, config: &Config) (format!("{campaign_id}-real"), "real", "Browser interactions from the installed subject app."), ] { create_entity(client, config, "TrafficSources", &id).await?; - post_namespaced_action(client, config, "TrafficSources", &id, "Configure", json!({ + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "TrafficSources", &id, "Configure", json!({ "campaign_id": campaign_id, "name": kind, "kind": kind, "description": description })).await?; - post_namespaced_action(client, config, "TrafficSources", &id, "Activate", json!({})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "TrafficSources", &id, "Activate", json!({})).await?; } + prepare_frozen_evaluator(client, config, &campaign_id, &trial_suite_id, &subject_seed).await?; create_entity(client, config, "SelectionDesigns", &design_id).await?; - post_namespaced_action(client, config, "SelectionDesigns", &design_id, "Configure", json!({ + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "SelectionDesigns", &design_id, "Configure", json!({ "campaign_id": campaign_id, "version_label": "v1", "evaluator_app_ref": evaluator_ref, - "trial_suite_id": "agent-answers-bootstrap", + "trial_suite_id": trial_suite_id, "fitness_model_json": r#"{"comparison":"evidence_weighted_preference","signals":["resolved_questions","answer_evidence","interaction_latency"],"release":"automatic"}"#, "constraint_definitions_json": r#"[{"key":"native_verified","kind":"required"},{"key":"rollback_available","kind":"required"}]"#, "traffic_sources_json": r#"["simulated","real"]"#, "rationale": brain_note, "proposed_by": "codex" })).await?; - post_namespaced_action(client, config, "SelectionDesigns", &design_id, "Approve", json!({"approved_by": "local-proof-human"})).await?; - post_namespaced_action(client, config, "SelectionDesigns", &design_id, "Freeze", json!({"frozen_at": now})).await?; - post_namespaced_action(client, config, "Campaigns", &campaign_id, "ApproveSelection", json!({ + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "SelectionDesigns", &design_id, "Approve", json!({"approved_by": "local-proof-human"})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "SelectionDesigns", &design_id, "Freeze", json!({"frozen_at": now})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Campaigns", &campaign_id, "ApproveSelection", json!({ "active_selection_design_id": design_id, "active_evaluator_ref": evaluator_ref })).await?; - post_namespaced_action(client, config, "Campaigns", &campaign_id, "Start", json!({})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Campaigns", &campaign_id, "Start", json!({})).await?; - let generation_one_ref = evolution_ref( - "EVOLUTION_GENERATION_ONE_REF", - "demo/agent-answers@candidate-evidence", - require_pinned_refs, - )?; - let generation_two_ref = evolution_ref( - "EVOLUTION_GENERATION_TWO_REF", - "demo/agent-answers@candidate-reuse", - require_pinned_refs, - )?; - run_evolution_generation(client, config, &campaign_id, "1", &subject_seed, &generation_one_ref, &design_id, &evaluator_ref, "Answers referencing evidence resolved both controlled and browser questions.").await?; - run_evolution_generation(client, config, &campaign_id, "2", &generation_one_ref, &generation_two_ref, &design_id, &evaluator_ref, "Successor traffic reused earlier validated answers with fewer failed attempts.").await?; + run_evolution_generation(client, config, &campaign_id, "1", &subject_seed, &generation_one_ref, &design_id, &evaluator_ref, &trial_suite_id, "Answers referencing evidence resolved both controlled and browser questions.").await?; + run_evolution_generation(client, config, &campaign_id, "2", &generation_one_ref, &generation_two_ref, &design_id, &evaluator_ref, &trial_suite_id, "Successor traffic reused earlier validated answers with fewer failed attempts.").await?; let capability_id = format!("{campaign_id}-capability-evidence"); create_entity(client, config, "EmergentCapabilities", &capability_id).await?; - post_namespaced_action(client, config, "EmergentCapabilities", &capability_id, "Configure", json!({ + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "EmergentCapabilities", &capability_id, "Configure", json!({ "campaign_id": campaign_id, "generation_id": format!("{campaign_id}-generation-2"), "candidate_id": format!("{campaign_id}-candidate-2-selected"), "title": "Reusable evidence surfaced in answers", "observation": "Codex observed repeated benefit without scripting it as a required feature.", "evidence_locator": "datadog://pending-local-ingestion" })).await?; - post_namespaced_action(client, config, "EmergentCapabilities", &capability_id, "Keep", json!({})).await?; - post_namespaced_action(client, config, "Campaigns", &campaign_id, "Pause", json!({"pause_reason": "Proof pause before rollback"})).await?; - post_namespaced_action(client, config, "Campaigns", &campaign_id, "Rollback", json!({ + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "EmergentCapabilities", &capability_id, "Keep", json!({})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Campaigns", &campaign_id, "Pause", json!({"pause_reason": "Proof pause before rollback"})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Campaigns", &campaign_id, "Rollback", json!({ "current_release_ref": generation_one_ref, "previous_release_ref": generation_two_ref, "last_release_reason": "Rollback exercised during local proof" })).await?; info!(campaign_id = %campaign_id, "directed evolution two-generation proof completed with automatic release, pause, and rollback"); @@ -87,6 +90,85 @@ async fn run_directed_evolution_demo(client: &reqwest::Client, config: &Config) Ok(()) } +async fn run_directed_evolution_mutation(config: &Config) -> Result<()> { + let candidate_dir = PathBuf::from(required_env("EVOLUTION_CANDIDATE_DIR")?); + let direction = required_env("EVOLUTION_DIRECTION")?; + let generation = env::var("EVOLUTION_GENERATION_ORDINAL").unwrap_or_else(|_| "next".to_string()); + let parent_ref = env::var("EVOLUTION_PARENT_REF").unwrap_or_else(|_| "local parent".to_string()); + if !candidate_dir.join("app.toml").is_file() || !candidate_dir.join("specs").is_dir() { + bail!("EVOLUTION_CANDIDATE_DIR must be a Temper-native app bundle with app.toml and specs/"); + } + let prompt = format!( + "You are the Codex v1 mutation brain for directed evolution. Edit the Temper-native app bundle in the current directory for generation {generation}, derived from {parent_ref}. Human direction: {direction}. Produce one small, coherent app improvement grounded in that direction. Only edit app-native files: app.toml, APP.md, specs/, policies/, wasm/, content/, seed-data/, and adrs/. Keep the app installable and update specs, CSDL, policies, and ADRs together when behavior changes. Do not edit evaluator files, create external crates, run git commands, or invent fitness results." + ); + let output = run_codex_exec_command(config, &candidate_dir, prompt, "generate directed evolution candidate").await?; + if !output.status.success() { + bail!("directed evolution mutation failed: {}", String::from_utf8_lossy(&output.stderr)); + } + let status = Command::new("git") + .args(["-C", candidate_dir.to_str().context("candidate path utf-8")?, "status", "--porcelain"]) + .output() + .await + .context("inspect generated candidate status")?; + if !status.status.success() { + bail!("could not inspect generated candidate: {}", String::from_utf8_lossy(&status.stderr)); + } + let changed: Vec = String::from_utf8_lossy(&status.stdout) + .lines() + .filter_map(|line| line.get(3..).map(str::trim).map(str::to_string)) + .filter(|path| !path.is_empty()) + .collect(); + if changed.is_empty() { + bail!("Codex mutation completed without producing a candidate change"); + } + if let Some(path) = changed.iter().find(|path| !evolution_candidate_path_allowed(path)) { + bail!("Codex mutation changed non-native candidate path '{path}'"); + } + println!("Directed evolution candidate generated for generation {generation}: {}", changed.join(", ")); + Ok(()) +} + +fn evolution_candidate_path_allowed(path: &str) -> bool { + path == "app.toml" + || path == "APP.md" + || ["specs/", "policies/", "wasm/", "content/", "seed-data/", "adrs/"] + .iter() + .any(|prefix| path.starts_with(prefix)) +} + +async fn prepare_frozen_evaluator( + client: &reqwest::Client, + config: &Config, + campaign_id: &str, + trial_suite_id: &str, + subject_ref: &str, +) -> Result<()> { + create_entity(client, config, "TrialSuites", trial_suite_id).await?; + post_protocol_action(client, config, EVALUATOR_NAMESPACE, "TrialSuites", trial_suite_id, "Configure", json!({ + "name": "Agent Answers bootstrap behavior", + "description": "Frozen native trial suite covering question resolution, evidence visibility, and successor reuse.", + "subject_app_ref": subject_ref, + "scenario_manifest_json": r#"[{"id":"controlled-question","traffic":"simulated"},{"id":"browser-answer","traffic":"real"},{"id":"successor-reuse","traffic":"simulated"}]"#, + "hidden_fixture_locator": format!("temper://campaigns/{campaign_id}/fixtures/bootstrap"), + "authored_by": "codex-with-human-approval" + })).await?; + post_protocol_action(client, config, EVALUATOR_NAMESPACE, "TrialSuites", trial_suite_id, "Freeze", json!({"frozen_at": generated_at_label()})).await?; + for (suffix, key, description, kind, locator, hard) in [ + ("resolved", "resolved_questions", "Controlled questions resolve after accepted answers.", "native_validator", "temper://validators/question-resolution", true), + ("evidence", "answer_evidence", "Real usage exposes an evidence locator on accepted answers.", "native_validator", "temper://validators/answer-evidence", false), + ("latency", "interaction_latency", "Observed interaction latency remains inspectable in Datadog.", "datadog", "datadog://pending-local-ingestion", false), + ] { + let metric_id = format!("{campaign_id}-metric-{suffix}"); + create_entity(client, config, "MetricDefinitions", &metric_id).await?; + post_protocol_action(client, config, EVALUATOR_NAMESPACE, "MetricDefinitions", &metric_id, "Configure", json!({ + "trial_suite_id": trial_suite_id, "key": key, "description": description, "instrument_kind": kind, + "instrument_locator": locator, "interpretation": "Evidence contributes to candidate comparison under the approved selection design.", "hard_constraint": hard + })).await?; + post_protocol_action(client, config, EVALUATOR_NAMESPACE, "MetricDefinitions", &metric_id, "Freeze", json!({"frozen_at": generated_at_label()})).await?; + } + Ok(()) +} + async fn evolution_brain_note(config: &Config, campaign_id: &str, subject_seed: &str) -> Result { if !evolution_bool_env("PAW_EVOLUTION_USE_CODEX") { return Ok("Codex-backed adapter configured. Smoke mode uses a stable approved design so protocol tests remain reproducible.".to_string()); @@ -100,49 +182,55 @@ async fn evolution_brain_note(config: &Config, campaign_id: &str, subject_seed: } fn evolution_bool_env(key: &str) -> bool { - env::var(key) - .map(|value| value == "1" || value.eq_ignore_ascii_case("true")) - .unwrap_or(false) + env::var(key).map(|value| value == "1" || value.eq_ignore_ascii_case("true")).unwrap_or(false) } fn evolution_ref(key: &str, smoke_default: &str, require_pinned: bool) -> Result { let value = env::var(key).unwrap_or_else(|_| smoke_default.to_string()); - if !require_pinned { - return Ok(value); - } - let hash = value - .rsplit_once('@') - .map(|(_, hash)| hash) + if !require_pinned { return Ok(value); } + value.rsplit_once('@').map(|(_, hash)| hash) .filter(|hash| hash.len() == 40 && hash.chars().all(|character| character.is_ascii_hexdigit())) .with_context(|| format!("{key} must be an immutable Genesis ref owner/app@<40-hex-commit> for a live Codex evolution run"))?; - let _ = hash; Ok(value) } #[allow(clippy::too_many_arguments)] -async fn run_evolution_generation(client: &reqwest::Client, config: &Config, campaign_id: &str, ordinal: &str, parent_ref: &str, winner_ref: &str, design_id: &str, evaluator_ref: &str, reason: &str) -> Result<()> { +async fn run_evolution_generation(client: &reqwest::Client, config: &Config, campaign_id: &str, ordinal: &str, parent_ref: &str, winner_ref: &str, design_id: &str, evaluator_ref: &str, trial_suite_id: &str, reason: &str) -> Result<()> { let generation_id = format!("{campaign_id}-generation-{ordinal}"); let selected_id = format!("{campaign_id}-candidate-{ordinal}-selected"); let rejected_id = format!("{campaign_id}-candidate-{ordinal}-baseline"); create_entity(client, config, "Generations", &generation_id).await?; - post_namespaced_action(client, config, "Generations", &generation_id, "Configure", json!({"campaign_id": campaign_id, "ordinal": ordinal, "parent_release_ref": parent_ref, "selection_design_id": design_id, "evaluator_app_ref": evaluator_ref})).await?; - post_namespaced_action(client, config, "Generations", &generation_id, "Begin", json!({})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Generations", &generation_id, "Configure", json!({"campaign_id": campaign_id, "ordinal": ordinal, "parent_release_ref": parent_ref, "selection_design_id": design_id, "evaluator_app_ref": evaluator_ref})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Generations", &generation_id, "Begin", json!({})).await?; for (candidate_id, app_ref, mutation) in [(&rejected_id, parent_ref, "Preserved incumbent for comparison."), (&selected_id, winner_ref, "Codex candidate derived from observed usage evidence.")] { create_entity(client, config, "Candidates", candidate_id).await?; - post_namespaced_action(client, config, "Candidates", candidate_id, "Configure", json!({"campaign_id": campaign_id, "generation_id": generation_id, "app_ref": app_ref, "parent_app_ref": parent_ref, "mutation_summary": mutation, "brain_run_id": format!("codex-{ordinal}")})).await?; - post_namespaced_action(client, config, "Candidates", candidate_id, "StartTrials", json!({})).await?; - post_namespaced_action(client, config, "Candidates", candidate_id, "Assess", json!({"assessment_json": format!(r#"{{"evidence":"recorded","candidate":"{}","generation":"{}"}}"#, candidate_id, ordinal)})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Candidates", candidate_id, "Configure", json!({"campaign_id": campaign_id, "generation_id": generation_id, "app_ref": app_ref, "parent_app_ref": parent_ref, "mutation_summary": mutation, "brain_run_id": format!("codex-{ordinal}")})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Candidates", candidate_id, "StartTrials", json!({})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Candidates", candidate_id, "Assess", json!({"assessment_json": format!(r#"{{"evidence":"validator-run","candidate":"{}","generation":"{}"}}"#, candidate_id, ordinal)})).await?; } - for (suffix, source, key, value) in [("sim", "simulated", "resolved_questions", "1.0"), ("real", "real", "answer_evidence", "observed"), ("trace", "datadog_observation", "interaction_latency", "captured") ] { + let validator_id = format!("{campaign_id}-validator-{ordinal}-selected"); + let validator_locator = format!("temper://ValidatorRuns('{validator_id}')"); + create_entity(client, config, "ValidatorRuns", &validator_id).await?; + post_protocol_action(client, config, EVALUATOR_NAMESPACE, "ValidatorRuns", &validator_id, "Configure", json!({ + "trial_suite_id": trial_suite_id, "candidate_id": selected_id, "scenario_id": format!("generation-{ordinal}-mixed-traffic"), "validator_kind": "native_trial" + })).await?; + post_protocol_action(client, config, EVALUATOR_NAMESPACE, "ValidatorRuns", &validator_id, "Pass", json!({ + "evidence_locator": validator_locator, "result_summary": reason + })).await?; + for (suffix, source, key, value, locator) in [ + ("sim", "simulated", "resolved_questions", "1.0", validator_locator.as_str()), + ("real", "real", "answer_evidence", "observed", "temper://real-usage/accepted-answer"), + ("trace", "datadog_observation", "interaction_latency", "captured", "datadog://pending-local-ingestion"), + ] { let measurement_id = format!("{campaign_id}-measurement-{ordinal}-{suffix}"); create_entity(client, config, "Measurements", &measurement_id).await?; - post_namespaced_action(client, config, "Measurements", &measurement_id, "Record", json!({"campaign_id": campaign_id, "generation_id": generation_id, "candidate_id": selected_id, "traffic_source_id": format!("{campaign_id}-{}", if source == "real" { "real" } else { "simulated" }), "metric_key": key, "metric_value": value, "source_kind": source, "evidence_locator": if source == "datadog_observation" { "datadog://pending-local-ingestion" } else { "temper://entity-evidence" }, "evaluator_app_ref": evaluator_ref, "recorded_at": generated_at_label(), "notes": reason})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Measurements", &measurement_id, "Record", json!({"campaign_id": campaign_id, "generation_id": generation_id, "candidate_id": selected_id, "traffic_source_id": format!("{campaign_id}-{}", if source == "real" { "real" } else { "simulated" }), "metric_key": key, "metric_value": value, "source_kind": source, "evidence_locator": locator, "evaluator_app_ref": evaluator_ref, "recorded_at": generated_at_label(), "notes": reason})).await?; } - post_namespaced_action(client, config, "Candidates", &rejected_id, "Eliminate", json!({"selection_reason": "Outperformed by assessed candidate under frozen design."})).await?; - post_namespaced_action(client, config, "Candidates", &selected_id, "Select", json!({"selection_reason": reason})).await?; - post_namespaced_action(client, config, "Candidates", &selected_id, "Release", json!({})).await?; - post_namespaced_action(client, config, "Generations", &generation_id, "SelectAndRelease", json!({"selected_candidate_id": selected_id, "released_app_ref": winner_ref, "selection_reason": reason})).await?; - post_namespaced_action(client, config, "Campaigns", campaign_id, "RecordRelease", json!({"current_release_ref": winner_ref, "previous_release_ref": parent_ref, "last_release_reason": reason})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Candidates", &rejected_id, "Eliminate", json!({"selection_reason": "Outperformed by assessed candidate under frozen design."})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Candidates", &selected_id, "Select", json!({"selection_reason": reason})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Candidates", &selected_id, "Release", json!({})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Generations", &generation_id, "SelectAndRelease", json!({"selected_candidate_id": selected_id, "released_app_ref": winner_ref, "selection_reason": reason})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Campaigns", campaign_id, "RecordRelease", json!({"current_release_ref": winner_ref, "previous_release_ref": parent_ref, "last_release_reason": reason})).await?; Ok(()) } @@ -152,8 +240,8 @@ async fn create_entity(client: &reqwest::Client, config: &Config, entity_set: &s Ok(()) } -async fn post_namespaced_action(client: &reqwest::Client, config: &Config, entity_set: &str, entity_id: &str, action: &str, body: Value) -> Result<()> { - let response = client.post(config.namespaced_action_url(EVOLUTION_NAMESPACE, entity_set, entity_id, action)).headers(headers(config)?).header(CONTENT_TYPE, "application/json").json(&body).send().await.with_context(|| format!("dispatch directed evolution {entity_set}.{action}"))?; - if !response.status().is_success() { let status = response.status(); let text = response.text().await.unwrap_or_default(); bail!("directed evolution {entity_set}.{action} returned {status}: {text}"); } +async fn post_protocol_action(client: &reqwest::Client, config: &Config, namespace: &str, entity_set: &str, entity_id: &str, action: &str, body: Value) -> Result<()> { + let response = client.post(config.namespaced_action_url(namespace, entity_set, entity_id, action)).headers(headers(config)?).header(CONTENT_TYPE, "application/json").json(&body).send().await.with_context(|| format!("dispatch {namespace} {entity_set}.{action}"))?; + if !response.status().is_success() { let status = response.status(); let text = response.text().await.unwrap_or_default(); bail!("{namespace} {entity_set}.{action} returned {status}: {text}"); } Ok(()) } diff --git a/crates/paw-codex-worker/src/main.rs b/crates/paw-codex-worker/src/main.rs index 8dbd04f97..c9299e606 100644 --- a/crates/paw-codex-worker/src/main.rs +++ b/crates/paw-codex-worker/src/main.rs @@ -50,6 +50,9 @@ async fn main() -> Result<()> { if command == WorkerCommand::DirectedEvolutionDemo { return run_directed_evolution_demo(&client, &config).await; } + if command == WorkerCommand::DirectedEvolutionMutate { + return run_directed_evolution_mutation(&config).await; + } info!( worker_id = %config.worker_id, diff --git a/crates/paw-codex-worker/src/tests.rs b/crates/paw-codex-worker/src/tests.rs index 85ff4269a..de1753069 100644 --- a/crates/paw-codex-worker/src/tests.rs +++ b/crates/paw-codex-worker/src/tests.rs @@ -230,6 +230,10 @@ mod tests { parse_worker_command(vec!["directed-evolution-demo".to_string()]), WorkerCommand::DirectedEvolutionDemo ); + assert_eq!( + parse_worker_command(vec!["directed-evolution-mutate".to_string()]), + WorkerCommand::DirectedEvolutionMutate + ); } #[test] @@ -243,6 +247,15 @@ mod tests { assert!(format!("{error:#}").contains("immutable Genesis ref")); } + #[test] + fn evolution_candidate_changes_are_limited_to_native_bundle_files() { + assert!(evolution_candidate_path_allowed("specs/answer.ioa.toml")); + assert!(evolution_candidate_path_allowed("wasm/validator/src/lib.rs")); + assert!(evolution_candidate_path_allowed("adrs/0002-evidence.md")); + assert!(!evolution_candidate_path_allowed("crates/random-helper/src/lib.rs")); + assert!(!evolution_candidate_path_allowed("../evaluator/specs/trial.ioa.toml")); + } + #[test] fn launchd_plist_renders_concrete_worker_environment() { let config = Config { diff --git a/crates/paw-codex-worker/src/worker_types.rs b/crates/paw-codex-worker/src/worker_types.rs index 3378d80bc..9ff94b8f9 100644 --- a/crates/paw-codex-worker/src/worker_types.rs +++ b/crates/paw-codex-worker/src/worker_types.rs @@ -29,6 +29,7 @@ enum WorkerCommand { Doctor, LaunchdPlist, DirectedEvolutionDemo, + DirectedEvolutionMutate, } #[derive(Clone, Copy, Debug, PartialEq, Eq)] diff --git a/docs/adrs/0052-codex-directed-evolution-brain-provider.md b/docs/adrs/0052-codex-directed-evolution-brain-provider.md index 3e8988dd9..16b6df574 100644 --- a/docs/adrs/0052-codex-directed-evolution-brain-provider.md +++ b/docs/adrs/0052-codex-directed-evolution-brain-provider.md @@ -10,6 +10,9 @@ V1 runs the directed-evolution brain through `paw-codex-worker`. The new `directed-evolution-demo` mode drives the native Genesis protocol and can call Codex for a selection-design rationale when `PAW_EVOLUTION_USE_CODEX=1`. Deterministic smoke mode exercises the protocol without an external model call. +The `directed-evolution-mutate` mode asks Codex to edit a candidate workspace, +then rejects any change outside the Temper-native subject app directories +before Genesis publishes or installs that candidate. Live Codex mode requires the seed and both selected candidate versions to be immutable Genesis commit refs (`owner/app@hash`); it does not accept illustrative candidate labels as releases. @@ -22,8 +25,10 @@ state or Evolution Studio. ## Evidence And Release Control -The proof mode records simulated, real-traffic, and Datadog evidence locators, -performs two automatic local releases, then pauses and rolls back. New local +The proof mode freezes evaluator-owned `TrialSuite` and `MetricDefinition` +records, records a native `ValidatorRun` for each selected candidate, attaches +simulated, real-traffic, and Datadog evidence locators, performs two automatic +local releases, then pauses and rolls back. New local Datadog ingestion requires an execution-time `DD_API_KEY`; absent that key the Datadog locator remains explicitly pending instead of claiming ingestion. From 094480653de07efd82362487e93a676cbd45a0ed Mon Sep 17 00:00:00 2001 From: Sesh Nalla Date: Tue, 26 May 2026 01:03:45 -0400 Subject: [PATCH 3/5] Gate releases on executed evaluator evidence --- .../src/directed_evolution.rs | 96 ++++++++++++++++--- crates/paw-codex-worker/src/tests.rs | 34 +++++++ ...codex-directed-evolution-brain-provider.md | 5 +- 3 files changed, 122 insertions(+), 13 deletions(-) diff --git a/crates/paw-codex-worker/src/directed_evolution.rs b/crates/paw-codex-worker/src/directed_evolution.rs index 12a26a845..6e9454cd5 100644 --- a/crates/paw-codex-worker/src/directed_evolution.rs +++ b/crates/paw-codex-worker/src/directed_evolution.rs @@ -1,6 +1,23 @@ const EVOLUTION_NAMESPACE: &str = "Genesis.Evolution"; const EVALUATOR_NAMESPACE: &str = "Genesis.AgentAnswersEvaluation"; +#[derive(Debug, Clone, Deserialize)] +struct EvolutionValidationManifest { + evaluator_ref: String, + records: Vec, +} + +#[derive(Debug, Clone, Deserialize)] +struct EvolutionValidationEvidence { + generation: String, + candidate_ref: String, + status: String, + evidence_locator: String, + result_summary: String, + resolved_questions: String, + answer_evidence: String, +} + async fn run_directed_evolution_demo(client: &reqwest::Client, config: &Config) -> Result<()> { let campaign_id = env::var("EVOLUTION_CAMPAIGN_ID") .unwrap_or_else(|_| format!("campaign-local-{}", generated_at_label())); @@ -29,6 +46,11 @@ async fn run_directed_evolution_demo(client: &reqwest::Client, config: &Config) let design_id = format!("{campaign_id}-selection-v1"); let trial_suite_id = format!("{campaign_id}-trial-suite-v1"); let now = generated_at_label(); + let validation_evidence = load_evolution_validation_evidence( + &evaluator_ref, + &[(&"1", &generation_one_ref), (&"2", &generation_two_ref)], + require_pinned_refs, + )?; let brain_note = evolution_brain_note(config, &campaign_id, &subject_seed).await?; create_entity(client, config, "Campaigns", &campaign_id).await?; @@ -71,8 +93,8 @@ async fn run_directed_evolution_demo(client: &reqwest::Client, config: &Config) })).await?; post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Campaigns", &campaign_id, "Start", json!({})).await?; - run_evolution_generation(client, config, &campaign_id, "1", &subject_seed, &generation_one_ref, &design_id, &evaluator_ref, &trial_suite_id, "Answers referencing evidence resolved both controlled and browser questions.").await?; - run_evolution_generation(client, config, &campaign_id, "2", &generation_one_ref, &generation_two_ref, &design_id, &evaluator_ref, &trial_suite_id, "Successor traffic reused earlier validated answers with fewer failed attempts.").await?; + run_evolution_generation(client, config, &campaign_id, "1", &subject_seed, &generation_one_ref, &design_id, &evaluator_ref, &trial_suite_id, &validation_evidence[0]).await?; + run_evolution_generation(client, config, &campaign_id, "2", &generation_one_ref, &generation_two_ref, &design_id, &evaluator_ref, &trial_suite_id, &validation_evidence[1]).await?; let capability_id = format!("{campaign_id}-capability-evidence"); create_entity(client, config, "EmergentCapabilities", &capability_id).await?; @@ -99,7 +121,7 @@ async fn run_directed_evolution_mutation(config: &Config) -> Result<()> { bail!("EVOLUTION_CANDIDATE_DIR must be a Temper-native app bundle with app.toml and specs/"); } let prompt = format!( - "You are the Codex v1 mutation brain for directed evolution. Edit the Temper-native app bundle in the current directory for generation {generation}, derived from {parent_ref}. Human direction: {direction}. Produce one small, coherent app improvement grounded in that direction. Only edit app-native files: app.toml, APP.md, specs/, policies/, wasm/, content/, seed-data/, and adrs/. Keep the app installable and update specs, CSDL, policies, and ADRs together when behavior changes. Do not edit evaluator files, create external crates, run git commands, or invent fitness results." + "You are the Codex v1 mutation brain for directed evolution. Edit the Temper-native app bundle in the current directory for generation {generation}, derived from {parent_ref}. Human direction: {direction}. Produce one small, coherent app improvement grounded in that direction. Only edit app-native files: app.toml, APP.md, specs/, policies/, wasm/, content/, seed-data/, and adrs/. Keep the app installable and preserve the existing Ask/Submit/RecordAnswer/Accept interaction contract so the frozen evaluator can execute it across generations; additive behavior is allowed. Update specs, CSDL, policies, and ADRs together when behavior changes. Do not edit evaluator files, create external crates, run git commands, or invent fitness results." ); let output = run_codex_exec_command(config, &candidate_dir, prompt, "generate directed evolution candidate").await?; if !output.status.success() { @@ -136,6 +158,57 @@ fn evolution_candidate_path_allowed(path: &str) -> bool { .any(|prefix| path.starts_with(prefix)) } +fn load_evolution_validation_evidence( + evaluator_ref: &str, + selected_refs: &[(&str, &str)], + required: bool, +) -> Result> { + let path = match env::var("EVOLUTION_VALIDATOR_EVIDENCE_PATH") { + Ok(path) if !path.trim().is_empty() => path, + _ if required => { + bail!("live directed evolution requires EVOLUTION_VALIDATOR_EVIDENCE_PATH from executed frozen-evaluator trials") + } + _ => { + return Ok(selected_refs + .iter() + .map(|(generation, candidate_ref)| EvolutionValidationEvidence { + generation: (*generation).to_string(), + candidate_ref: (*candidate_ref).to_string(), + status: "Passed".to_string(), + evidence_locator: format!("temper://fixture/generation-{generation}"), + result_summary: "Fixture-only protocol evidence.".to_string(), + resolved_questions: "1.0".to_string(), + answer_evidence: "fixture".to_string(), + }) + .collect()); + } + }; + let manifest: EvolutionValidationManifest = serde_json::from_slice( + &fs::read(&path).with_context(|| format!("read validator evidence manifest {path}"))?, + ) + .with_context(|| format!("parse validator evidence manifest {path}"))?; + if manifest.evaluator_ref != evaluator_ref { + bail!("validator evidence evaluator ref does not match the frozen evaluator ref"); + } + let mut selected = Vec::with_capacity(selected_refs.len()); + for (generation, candidate_ref) in selected_refs { + let evidence = manifest + .records + .iter() + .find(|record| record.generation == *generation && record.candidate_ref == *candidate_ref) + .with_context(|| format!("validator evidence missing generation {generation} candidate {candidate_ref}"))?; + if evidence.status != "Passed" + || evidence.evidence_locator.trim().is_empty() + || evidence.resolved_questions.trim().is_empty() + || evidence.answer_evidence.trim().is_empty() + { + bail!("generation {generation} has no passing executed validator evidence"); + } + selected.push(evidence.clone()); + } + Ok(selected) +} + async fn prepare_frozen_evaluator( client: &reqwest::Client, config: &Config, @@ -195,7 +268,7 @@ fn evolution_ref(key: &str, smoke_default: &str, require_pinned: bool) -> Result } #[allow(clippy::too_many_arguments)] -async fn run_evolution_generation(client: &reqwest::Client, config: &Config, campaign_id: &str, ordinal: &str, parent_ref: &str, winner_ref: &str, design_id: &str, evaluator_ref: &str, trial_suite_id: &str, reason: &str) -> Result<()> { +async fn run_evolution_generation(client: &reqwest::Client, config: &Config, campaign_id: &str, ordinal: &str, parent_ref: &str, winner_ref: &str, design_id: &str, evaluator_ref: &str, trial_suite_id: &str, evidence: &EvolutionValidationEvidence) -> Result<()> { let generation_id = format!("{campaign_id}-generation-{ordinal}"); let selected_id = format!("{campaign_id}-candidate-{ordinal}-selected"); let rejected_id = format!("{campaign_id}-candidate-{ordinal}-baseline"); @@ -209,28 +282,27 @@ async fn run_evolution_generation(client: &reqwest::Client, config: &Config, cam post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Candidates", candidate_id, "Assess", json!({"assessment_json": format!(r#"{{"evidence":"validator-run","candidate":"{}","generation":"{}"}}"#, candidate_id, ordinal)})).await?; } let validator_id = format!("{campaign_id}-validator-{ordinal}-selected"); - let validator_locator = format!("temper://ValidatorRuns('{validator_id}')"); create_entity(client, config, "ValidatorRuns", &validator_id).await?; post_protocol_action(client, config, EVALUATOR_NAMESPACE, "ValidatorRuns", &validator_id, "Configure", json!({ "trial_suite_id": trial_suite_id, "candidate_id": selected_id, "scenario_id": format!("generation-{ordinal}-mixed-traffic"), "validator_kind": "native_trial" })).await?; post_protocol_action(client, config, EVALUATOR_NAMESPACE, "ValidatorRuns", &validator_id, "Pass", json!({ - "evidence_locator": validator_locator, "result_summary": reason + "evidence_locator": evidence.evidence_locator, "result_summary": evidence.result_summary })).await?; for (suffix, source, key, value, locator) in [ - ("sim", "simulated", "resolved_questions", "1.0", validator_locator.as_str()), - ("real", "real", "answer_evidence", "observed", "temper://real-usage/accepted-answer"), + ("sim", "simulated", "resolved_questions", evidence.resolved_questions.as_str(), evidence.evidence_locator.as_str()), + ("real", "real", "answer_evidence", evidence.answer_evidence.as_str(), evidence.evidence_locator.as_str()), ("trace", "datadog_observation", "interaction_latency", "captured", "datadog://pending-local-ingestion"), ] { let measurement_id = format!("{campaign_id}-measurement-{ordinal}-{suffix}"); create_entity(client, config, "Measurements", &measurement_id).await?; - post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Measurements", &measurement_id, "Record", json!({"campaign_id": campaign_id, "generation_id": generation_id, "candidate_id": selected_id, "traffic_source_id": format!("{campaign_id}-{}", if source == "real" { "real" } else { "simulated" }), "metric_key": key, "metric_value": value, "source_kind": source, "evidence_locator": locator, "evaluator_app_ref": evaluator_ref, "recorded_at": generated_at_label(), "notes": reason})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Measurements", &measurement_id, "Record", json!({"campaign_id": campaign_id, "generation_id": generation_id, "candidate_id": selected_id, "traffic_source_id": format!("{campaign_id}-{}", if source == "real" { "real" } else { "simulated" }), "metric_key": key, "metric_value": value, "source_kind": source, "evidence_locator": locator, "evaluator_app_ref": evaluator_ref, "recorded_at": generated_at_label(), "notes": evidence.result_summary})).await?; } post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Candidates", &rejected_id, "Eliminate", json!({"selection_reason": "Outperformed by assessed candidate under frozen design."})).await?; - post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Candidates", &selected_id, "Select", json!({"selection_reason": reason})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Candidates", &selected_id, "Select", json!({"selection_reason": evidence.result_summary})).await?; post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Candidates", &selected_id, "Release", json!({})).await?; - post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Generations", &generation_id, "SelectAndRelease", json!({"selected_candidate_id": selected_id, "released_app_ref": winner_ref, "selection_reason": reason})).await?; - post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Campaigns", campaign_id, "RecordRelease", json!({"current_release_ref": winner_ref, "previous_release_ref": parent_ref, "last_release_reason": reason})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Generations", &generation_id, "SelectAndRelease", json!({"selected_candidate_id": selected_id, "released_app_ref": winner_ref, "selection_reason": evidence.result_summary})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Campaigns", campaign_id, "RecordRelease", json!({"current_release_ref": winner_ref, "previous_release_ref": parent_ref, "last_release_reason": evidence.result_summary})).await?; Ok(()) } diff --git a/crates/paw-codex-worker/src/tests.rs b/crates/paw-codex-worker/src/tests.rs index de1753069..047da3a7b 100644 --- a/crates/paw-codex-worker/src/tests.rs +++ b/crates/paw-codex-worker/src/tests.rs @@ -256,6 +256,40 @@ mod tests { assert!(!evolution_candidate_path_allowed("../evaluator/specs/trial.ioa.toml")); } + #[tokio::test] + async fn live_evolution_binds_releases_to_executed_validator_evidence() { + let _guard = ENV_LOCK.lock().await; + let path = unique_temp_dir().join("validator-evidence.json"); + fs::create_dir_all(path.parent().expect("manifest parent")).expect("manifest parent"); + fs::write( + &path, + r#"{"evaluator_ref":"owner/evaluator@aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa","records":[{"generation":"1","candidate_ref":"owner/app@1111111111111111111111111111111111111111","status":"Passed","evidence_locator":"temper://trial/g1/answer","result_summary":"generation one passed","resolved_questions":"1.0","answer_evidence":"observed"},{"generation":"2","candidate_ref":"owner/app@2222222222222222222222222222222222222222","status":"Passed","evidence_locator":"temper://trial/g2/answer","result_summary":"generation two passed","resolved_questions":"1.0","answer_evidence":"observed"}]}"#, + ) + .expect("validator evidence fixture"); + let _evidence = EnvOverride::set( + "EVOLUTION_VALIDATOR_EVIDENCE_PATH", + path.as_os_str().to_os_string(), + ); + let records = load_evolution_validation_evidence( + "owner/evaluator@aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + &[ + ("1", "owner/app@1111111111111111111111111111111111111111"), + ("2", "owner/app@2222222222222222222222222222222222222222"), + ], + true, + ) + .expect("matching executed evidence"); + assert_eq!(records[1].evidence_locator, "temper://trial/g2/answer"); + + let error = load_evolution_validation_evidence( + "owner/evaluator@aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + &[("2", "owner/app@3333333333333333333333333333333333333333")], + true, + ) + .expect_err("unexecuted candidate must not release"); + assert!(format!("{error:#}").contains("validator evidence missing generation 2")); + } + #[test] fn launchd_plist_renders_concrete_worker_environment() { let config = Config { diff --git a/docs/adrs/0052-codex-directed-evolution-brain-provider.md b/docs/adrs/0052-codex-directed-evolution-brain-provider.md index 16b6df574..1355233f1 100644 --- a/docs/adrs/0052-codex-directed-evolution-brain-provider.md +++ b/docs/adrs/0052-codex-directed-evolution-brain-provider.md @@ -26,7 +26,10 @@ state or Evolution Studio. ## Evidence And Release Control The proof mode freezes evaluator-owned `TrialSuite` and `MetricDefinition` -records, records a native `ValidatorRun` for each selected candidate, attaches +records. A live run requires `EVOLUTION_VALIDATOR_EVIDENCE_PATH`, produced by +executing the frozen scenario against the exact pinned candidate refs; a +mismatched or absent record prevents release. The worker records a native +`ValidatorRun` for each validated selected candidate and attaches simulated, real-traffic, and Datadog evidence locators, performs two automatic local releases, then pauses and rolls back. New local Datadog ingestion requires an execution-time `DD_API_KEY`; absent that key the From dc8471a3bf0985b9e15294bbf9114171c2e03a93 Mon Sep 17 00:00:00 2001 From: Sesh Nalla Date: Tue, 26 May 2026 01:21:14 -0400 Subject: [PATCH 4/5] Execute arbitrary directed evolution campaign plans --- crates/paw-codex-worker/src/cli.rs | 3 + .../src/directed_evolution.rs | 287 ++++++++++++++++-- crates/paw-codex-worker/src/main.rs | 3 + crates/paw-codex-worker/src/tests.rs | 19 +- crates/paw-codex-worker/src/worker_types.rs | 1 + ...codex-directed-evolution-brain-provider.md | 16 +- 6 files changed, 296 insertions(+), 33 deletions(-) diff --git a/crates/paw-codex-worker/src/cli.rs b/crates/paw-codex-worker/src/cli.rs index 4bbd7c8d4..823325517 100644 --- a/crates/paw-codex-worker/src/cli.rs +++ b/crates/paw-codex-worker/src/cli.rs @@ -9,6 +9,9 @@ fn parse_worker_command(args: impl IntoIterator) -> WorkerCommand "directed-evolution-demo" | "--directed-evolution-demo" => { return WorkerCommand::DirectedEvolutionDemo; } + "directed-evolution-run" | "--directed-evolution-run" => { + return WorkerCommand::DirectedEvolutionRun; + } "directed-evolution-mutate" | "--directed-evolution-mutate" => { return WorkerCommand::DirectedEvolutionMutate; } diff --git a/crates/paw-codex-worker/src/directed_evolution.rs b/crates/paw-codex-worker/src/directed_evolution.rs index 6e9454cd5..bfa7f3c6d 100644 --- a/crates/paw-codex-worker/src/directed_evolution.rs +++ b/crates/paw-codex-worker/src/directed_evolution.rs @@ -14,8 +14,113 @@ struct EvolutionValidationEvidence { status: String, evidence_locator: String, result_summary: String, - resolved_questions: String, - answer_evidence: String, + measurements: Vec, +} + +#[derive(Debug, Clone, Deserialize)] +struct EvolutionEvidenceMeasurement { + suffix: String, + traffic_source_id: String, + metric_key: String, + metric_value: String, + source_kind: String, + evidence_locator: String, +} + +#[derive(Debug, Clone, Deserialize)] +struct EvolutionCampaignPlan { + campaign_id: String, + name: String, + director_brief: String, + target_app_ref: String, + evaluator_app_ref: String, + brain_provider: String, + automation_mode: String, + traffic_sources: Vec, + selection_design: EvolutionSelectionPlan, + generations: Vec, + #[serde(default)] + capabilities: Vec, + release_control: EvolutionReleaseControlPlan, +} + +#[derive(Debug, Clone, Deserialize)] +struct EvolutionTrafficSourcePlan { + id: String, + name: String, + kind: String, + description: String, +} + +#[derive(Debug, Clone, Deserialize)] +struct EvolutionSelectionPlan { + id: String, + version_label: String, + evaluator_namespace: String, + #[serde(default = "default_trial_suites_entity_set")] + trial_suites_entity_set: String, + #[serde(default = "default_metric_definitions_entity_set")] + metric_definitions_entity_set: String, + #[serde(default = "default_validator_runs_entity_set")] + validator_runs_entity_set: String, + trial_suite: EvolutionTrialSuitePlan, + fitness_model_json: Value, + constraint_definitions_json: Value, + traffic_sources_json: Value, + rationale: String, + proposed_by: String, + approved_by: String, + metrics: Vec, +} + +#[derive(Debug, Clone, Deserialize)] +struct EvolutionTrialSuitePlan { + id: String, + name: String, + description: String, + scenario_manifest_json: Value, + hidden_fixture_locator: String, + authored_by: String, +} + +#[derive(Debug, Clone, Deserialize)] +struct EvolutionMetricPlan { + id: String, + key: String, + description: String, + instrument_kind: String, + instrument_locator: String, + interpretation: String, + hard_constraint: bool, +} + +#[derive(Debug, Clone, Deserialize)] +struct EvolutionGenerationPlan { + ordinal: String, + parent_release_ref: String, + selected_app_ref: String, + #[serde(default = "default_baseline_mutation")] + baseline_mutation_summary: String, + #[serde(default = "default_candidate_mutation")] + selected_mutation_summary: String, +} + +#[derive(Debug, Clone, Deserialize)] +struct EvolutionCapabilityPlan { + id: String, + generation_ordinal: String, + title: String, + observation: String, + evidence_locator: String, + keep: bool, +} + +#[derive(Debug, Clone, Deserialize)] +struct EvolutionReleaseControlPlan { + pause_reason: String, + rollback_current_ref: String, + rollback_previous_ref: String, + rollback_reason: String, } async fn run_directed_evolution_demo(client: &reqwest::Client, config: &Config) -> Result<()> { @@ -93,8 +198,8 @@ async fn run_directed_evolution_demo(client: &reqwest::Client, config: &Config) })).await?; post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Campaigns", &campaign_id, "Start", json!({})).await?; - run_evolution_generation(client, config, &campaign_id, "1", &subject_seed, &generation_one_ref, &design_id, &evaluator_ref, &trial_suite_id, &validation_evidence[0]).await?; - run_evolution_generation(client, config, &campaign_id, "2", &generation_one_ref, &generation_two_ref, &design_id, &evaluator_ref, &trial_suite_id, &validation_evidence[1]).await?; + run_evolution_generation(client, config, &campaign_id, "1", &subject_seed, &generation_one_ref, &design_id, &evaluator_ref, &trial_suite_id, EVALUATOR_NAMESPACE, "ValidatorRuns", "Preserved incumbent for comparison.", "Codex candidate derived from observed usage evidence.", &validation_evidence[0]).await?; + run_evolution_generation(client, config, &campaign_id, "2", &generation_one_ref, &generation_two_ref, &design_id, &evaluator_ref, &trial_suite_id, EVALUATOR_NAMESPACE, "ValidatorRuns", "Preserved incumbent for comparison.", "Codex candidate derived from observed usage evidence.", &validation_evidence[1]).await?; let capability_id = format!("{campaign_id}-capability-evidence"); create_entity(client, config, "EmergentCapabilities", &capability_id).await?; @@ -112,6 +217,86 @@ async fn run_directed_evolution_demo(client: &reqwest::Client, config: &Config) Ok(()) } +async fn run_directed_evolution_run(client: &reqwest::Client, config: &Config) -> Result<()> { + let path = required_env("EVOLUTION_CAMPAIGN_PLAN_PATH")?; + let plan: EvolutionCampaignPlan = serde_json::from_slice( + &fs::read(&path).with_context(|| format!("read evolution campaign plan {path}"))?, + ) + .with_context(|| format!("parse evolution campaign plan {path}"))?; + execute_evolution_campaign_plan(client, config, &plan).await +} + +async fn execute_evolution_campaign_plan(client: &reqwest::Client, config: &Config, plan: &EvolutionCampaignPlan) -> Result<()> { + let require_pinned_refs = evolution_bool_env("EVOLUTION_REQUIRE_PINNED_REFS") + || evolution_bool_env("PAW_EVOLUTION_USE_CODEX"); + require_evolution_ref("target_app_ref", &plan.target_app_ref, require_pinned_refs)?; + require_evolution_ref("evaluator_app_ref", &plan.evaluator_app_ref, require_pinned_refs)?; + if plan.generations.is_empty() { + bail!("evolution campaign plan must contain at least one generation"); + } + for generation in &plan.generations { + require_evolution_ref("parent_release_ref", &generation.parent_release_ref, require_pinned_refs)?; + require_evolution_ref("selected_app_ref", &generation.selected_app_ref, require_pinned_refs)?; + } + let selected_refs: Vec<(&str, &str)> = plan.generations.iter() + .map(|generation| (generation.ordinal.as_str(), generation.selected_app_ref.as_str())) + .collect(); + let evidence = load_evolution_validation_evidence(&plan.evaluator_app_ref, &selected_refs, require_pinned_refs)?; + let rationale = if evolution_bool_env("PAW_EVOLUTION_USE_CODEX") { + evolution_brain_note(config, &plan.campaign_id, &plan.target_app_ref).await? + } else { + plan.selection_design.rationale.clone() + }; + create_entity(client, config, "Campaigns", &plan.campaign_id).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Campaigns", &plan.campaign_id, "Configure", json!({ + "name": plan.name, "director_brief": plan.director_brief, "target_app_ref": plan.target_app_ref, + "brain_provider": plan.brain_provider, "automation_mode": plan.automation_mode + })).await?; + for source in &plan.traffic_sources { + create_entity(client, config, "TrafficSources", &source.id).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "TrafficSources", &source.id, "Configure", json!({ + "campaign_id": plan.campaign_id, "name": source.name, "kind": source.kind, "description": source.description + })).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "TrafficSources", &source.id, "Activate", json!({})).await?; + } + prepare_frozen_evaluator_plan(client, config, plan).await?; + let selection = &plan.selection_design; + create_entity(client, config, "SelectionDesigns", &selection.id).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "SelectionDesigns", &selection.id, "Configure", json!({ + "campaign_id": plan.campaign_id, "version_label": selection.version_label, "evaluator_app_ref": plan.evaluator_app_ref, + "trial_suite_id": selection.trial_suite.id, "fitness_model_json": selection.fitness_model_json.to_string(), + "constraint_definitions_json": selection.constraint_definitions_json.to_string(), "traffic_sources_json": selection.traffic_sources_json.to_string(), + "rationale": rationale, "proposed_by": selection.proposed_by + })).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "SelectionDesigns", &selection.id, "Approve", json!({"approved_by": selection.approved_by})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "SelectionDesigns", &selection.id, "Freeze", json!({"frozen_at": generated_at_label()})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Campaigns", &plan.campaign_id, "ApproveSelection", json!({ + "active_selection_design_id": selection.id, "active_evaluator_ref": plan.evaluator_app_ref + })).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Campaigns", &plan.campaign_id, "Start", json!({})).await?; + for (generation, record) in plan.generations.iter().zip(&evidence) { + run_evolution_generation(client, config, &plan.campaign_id, &generation.ordinal, &generation.parent_release_ref, &generation.selected_app_ref, &selection.id, &plan.evaluator_app_ref, &selection.trial_suite.id, &selection.evaluator_namespace, &selection.validator_runs_entity_set, &generation.baseline_mutation_summary, &generation.selected_mutation_summary, record).await?; + } + for capability in &plan.capabilities { + create_entity(client, config, "EmergentCapabilities", &capability.id).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "EmergentCapabilities", &capability.id, "Configure", json!({ + "campaign_id": plan.campaign_id, "generation_id": format!("{}-generation-{}", plan.campaign_id, capability.generation_ordinal), + "candidate_id": format!("{}-candidate-{}-selected", plan.campaign_id, capability.generation_ordinal), "title": capability.title, + "observation": capability.observation, "evidence_locator": capability.evidence_locator + })).await?; + let action = if capability.keep { "Keep" } else { "Reject" }; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "EmergentCapabilities", &capability.id, action, json!({})).await?; + } + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Campaigns", &plan.campaign_id, "Pause", json!({"pause_reason": plan.release_control.pause_reason})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Campaigns", &plan.campaign_id, "Rollback", json!({ + "current_release_ref": plan.release_control.rollback_current_ref, "previous_release_ref": plan.release_control.rollback_previous_ref, + "last_release_reason": plan.release_control.rollback_reason + })).await?; + info!(campaign_id = %plan.campaign_id, "directed evolution campaign plan completed with automatic release, pause, and rollback"); + println!("Directed evolution campaign completed: {}", plan.campaign_id); + Ok(()) +} + async fn run_directed_evolution_mutation(config: &Config) -> Result<()> { let candidate_dir = PathBuf::from(required_env("EVOLUTION_CANDIDATE_DIR")?); let direction = required_env("EVOLUTION_DIRECTION")?; @@ -120,8 +305,10 @@ async fn run_directed_evolution_mutation(config: &Config) -> Result<()> { if !candidate_dir.join("app.toml").is_file() || !candidate_dir.join("specs").is_dir() { bail!("EVOLUTION_CANDIDATE_DIR must be a Temper-native app bundle with app.toml and specs/"); } + let validator_contract = env::var("EVOLUTION_VALIDATOR_CONTRACT") + .unwrap_or_else(|_| "Preserve every behavior required by the frozen evaluator contract supplied for this campaign; additive behavior is allowed.".to_string()); let prompt = format!( - "You are the Codex v1 mutation brain for directed evolution. Edit the Temper-native app bundle in the current directory for generation {generation}, derived from {parent_ref}. Human direction: {direction}. Produce one small, coherent app improvement grounded in that direction. Only edit app-native files: app.toml, APP.md, specs/, policies/, wasm/, content/, seed-data/, and adrs/. Keep the app installable and preserve the existing Ask/Submit/RecordAnswer/Accept interaction contract so the frozen evaluator can execute it across generations; additive behavior is allowed. Update specs, CSDL, policies, and ADRs together when behavior changes. Do not edit evaluator files, create external crates, run git commands, or invent fitness results." + "You are the Codex v1 mutation brain for directed evolution. Edit the Temper-native app bundle in the current directory for generation {generation}, derived from {parent_ref}. Human direction: {direction}. Frozen evaluator contract: {validator_contract}. Produce one small, coherent app improvement grounded in that direction. Only edit app-native files: app.toml, APP.md, specs/, policies/, wasm/, content/, seed-data/, and adrs/. Keep the app installable and update specs, CSDL, policies, and ADRs together when behavior changes. Do not edit evaluator files, create external crates, run git commands, or invent fitness results." ); let output = run_codex_exec_command(config, &candidate_dir, prompt, "generate directed evolution candidate").await?; if !output.status.success() { @@ -158,6 +345,26 @@ fn evolution_candidate_path_allowed(path: &str) -> bool { .any(|prefix| path.starts_with(prefix)) } +fn default_baseline_mutation() -> String { + "Retained incumbent candidate for comparison.".to_string() +} + +fn default_candidate_mutation() -> String { + "Candidate proposed by the configured evolution brain.".to_string() +} + +fn default_trial_suites_entity_set() -> String { + "TrialSuites".to_string() +} + +fn default_metric_definitions_entity_set() -> String { + "MetricDefinitions".to_string() +} + +fn default_validator_runs_entity_set() -> String { + "ValidatorRuns".to_string() +} + fn load_evolution_validation_evidence( evaluator_ref: &str, selected_refs: &[(&str, &str)], @@ -177,8 +384,14 @@ fn load_evolution_validation_evidence( status: "Passed".to_string(), evidence_locator: format!("temper://fixture/generation-{generation}"), result_summary: "Fixture-only protocol evidence.".to_string(), - resolved_questions: "1.0".to_string(), - answer_evidence: "fixture".to_string(), + measurements: vec![EvolutionEvidenceMeasurement { + suffix: "fixture".to_string(), + traffic_source_id: "fixture".to_string(), + metric_key: "fixture_result".to_string(), + metric_value: "passed".to_string(), + source_kind: "fixture".to_string(), + evidence_locator: format!("temper://fixture/generation-{generation}"), + }], }) .collect()); } @@ -197,11 +410,8 @@ fn load_evolution_validation_evidence( .iter() .find(|record| record.generation == *generation && record.candidate_ref == *candidate_ref) .with_context(|| format!("validator evidence missing generation {generation} candidate {candidate_ref}"))?; - if evidence.status != "Passed" - || evidence.evidence_locator.trim().is_empty() - || evidence.resolved_questions.trim().is_empty() - || evidence.answer_evidence.trim().is_empty() - { + if evidence.status != "Passed" || evidence.evidence_locator.trim().is_empty() || evidence.measurements.is_empty() + || evidence.measurements.iter().any(|measurement| measurement.metric_key.trim().is_empty() || measurement.metric_value.trim().is_empty() || measurement.source_kind.trim().is_empty() || measurement.evidence_locator.trim().is_empty()) { bail!("generation {generation} has no passing executed validator evidence"); } selected.push(evidence.clone()); @@ -209,6 +419,27 @@ fn load_evolution_validation_evidence( Ok(selected) } +async fn prepare_frozen_evaluator_plan(client: &reqwest::Client, config: &Config, plan: &EvolutionCampaignPlan) -> Result<()> { + let suite = &plan.selection_design.trial_suite; + let selection = &plan.selection_design; + create_entity(client, config, &selection.trial_suites_entity_set, &suite.id).await?; + post_protocol_action(client, config, &selection.evaluator_namespace, &selection.trial_suites_entity_set, &suite.id, "Configure", json!({ + "name": suite.name, "description": suite.description, "subject_app_ref": plan.target_app_ref, + "scenario_manifest_json": suite.scenario_manifest_json.to_string(), "hidden_fixture_locator": suite.hidden_fixture_locator, + "authored_by": suite.authored_by + })).await?; + post_protocol_action(client, config, &selection.evaluator_namespace, &selection.trial_suites_entity_set, &suite.id, "Freeze", json!({"frozen_at": generated_at_label()})).await?; + for metric in &selection.metrics { + create_entity(client, config, &selection.metric_definitions_entity_set, &metric.id).await?; + post_protocol_action(client, config, &selection.evaluator_namespace, &selection.metric_definitions_entity_set, &metric.id, "Configure", json!({ + "trial_suite_id": suite.id, "key": metric.key, "description": metric.description, "instrument_kind": metric.instrument_kind, + "instrument_locator": metric.instrument_locator, "interpretation": metric.interpretation, "hard_constraint": metric.hard_constraint + })).await?; + post_protocol_action(client, config, &selection.evaluator_namespace, &selection.metric_definitions_entity_set, &metric.id, "Freeze", json!({"frozen_at": generated_at_label()})).await?; + } + Ok(()) +} + async fn prepare_frozen_evaluator( client: &reqwest::Client, config: &Config, @@ -260,43 +491,45 @@ fn evolution_bool_env(key: &str) -> bool { fn evolution_ref(key: &str, smoke_default: &str, require_pinned: bool) -> Result { let value = env::var(key).unwrap_or_else(|_| smoke_default.to_string()); - if !require_pinned { return Ok(value); } + require_evolution_ref(key, &value, require_pinned)?; + Ok(value) +} + +fn require_evolution_ref(label: &str, value: &str, require_pinned: bool) -> Result<()> { + if !require_pinned { return Ok(()); } value.rsplit_once('@').map(|(_, hash)| hash) .filter(|hash| hash.len() == 40 && hash.chars().all(|character| character.is_ascii_hexdigit())) - .with_context(|| format!("{key} must be an immutable Genesis ref owner/app@<40-hex-commit> for a live Codex evolution run"))?; - Ok(value) + .with_context(|| format!("{label} must be an immutable Genesis ref owner/app@<40-hex-commit> for a live Codex evolution run"))?; + Ok(()) } #[allow(clippy::too_many_arguments)] -async fn run_evolution_generation(client: &reqwest::Client, config: &Config, campaign_id: &str, ordinal: &str, parent_ref: &str, winner_ref: &str, design_id: &str, evaluator_ref: &str, trial_suite_id: &str, evidence: &EvolutionValidationEvidence) -> Result<()> { +async fn run_evolution_generation(client: &reqwest::Client, config: &Config, campaign_id: &str, ordinal: &str, parent_ref: &str, winner_ref: &str, design_id: &str, evaluator_ref: &str, trial_suite_id: &str, evaluator_namespace: &str, validator_runs_entity_set: &str, baseline_mutation: &str, selected_mutation: &str, evidence: &EvolutionValidationEvidence) -> Result<()> { let generation_id = format!("{campaign_id}-generation-{ordinal}"); let selected_id = format!("{campaign_id}-candidate-{ordinal}-selected"); let rejected_id = format!("{campaign_id}-candidate-{ordinal}-baseline"); create_entity(client, config, "Generations", &generation_id).await?; post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Generations", &generation_id, "Configure", json!({"campaign_id": campaign_id, "ordinal": ordinal, "parent_release_ref": parent_ref, "selection_design_id": design_id, "evaluator_app_ref": evaluator_ref})).await?; post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Generations", &generation_id, "Begin", json!({})).await?; - for (candidate_id, app_ref, mutation) in [(&rejected_id, parent_ref, "Preserved incumbent for comparison."), (&selected_id, winner_ref, "Codex candidate derived from observed usage evidence.")] { + for (candidate_id, app_ref, mutation) in [(&rejected_id, parent_ref, baseline_mutation), (&selected_id, winner_ref, selected_mutation)] { create_entity(client, config, "Candidates", candidate_id).await?; post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Candidates", candidate_id, "Configure", json!({"campaign_id": campaign_id, "generation_id": generation_id, "app_ref": app_ref, "parent_app_ref": parent_ref, "mutation_summary": mutation, "brain_run_id": format!("codex-{ordinal}")})).await?; post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Candidates", candidate_id, "StartTrials", json!({})).await?; post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Candidates", candidate_id, "Assess", json!({"assessment_json": format!(r#"{{"evidence":"validator-run","candidate":"{}","generation":"{}"}}"#, candidate_id, ordinal)})).await?; } let validator_id = format!("{campaign_id}-validator-{ordinal}-selected"); - create_entity(client, config, "ValidatorRuns", &validator_id).await?; - post_protocol_action(client, config, EVALUATOR_NAMESPACE, "ValidatorRuns", &validator_id, "Configure", json!({ + create_entity(client, config, validator_runs_entity_set, &validator_id).await?; + post_protocol_action(client, config, evaluator_namespace, validator_runs_entity_set, &validator_id, "Configure", json!({ "trial_suite_id": trial_suite_id, "candidate_id": selected_id, "scenario_id": format!("generation-{ordinal}-mixed-traffic"), "validator_kind": "native_trial" })).await?; - post_protocol_action(client, config, EVALUATOR_NAMESPACE, "ValidatorRuns", &validator_id, "Pass", json!({ + post_protocol_action(client, config, evaluator_namespace, validator_runs_entity_set, &validator_id, "Pass", json!({ "evidence_locator": evidence.evidence_locator, "result_summary": evidence.result_summary })).await?; - for (suffix, source, key, value, locator) in [ - ("sim", "simulated", "resolved_questions", evidence.resolved_questions.as_str(), evidence.evidence_locator.as_str()), - ("real", "real", "answer_evidence", evidence.answer_evidence.as_str(), evidence.evidence_locator.as_str()), - ("trace", "datadog_observation", "interaction_latency", "captured", "datadog://pending-local-ingestion"), - ] { - let measurement_id = format!("{campaign_id}-measurement-{ordinal}-{suffix}"); + for measurement in &evidence.measurements { + let measurement_id = format!("{campaign_id}-measurement-{ordinal}-{}", measurement.suffix); + let traffic_source_id = measurement.traffic_source_id.replace("{campaign_id}", campaign_id); create_entity(client, config, "Measurements", &measurement_id).await?; - post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Measurements", &measurement_id, "Record", json!({"campaign_id": campaign_id, "generation_id": generation_id, "candidate_id": selected_id, "traffic_source_id": format!("{campaign_id}-{}", if source == "real" { "real" } else { "simulated" }), "metric_key": key, "metric_value": value, "source_kind": source, "evidence_locator": locator, "evaluator_app_ref": evaluator_ref, "recorded_at": generated_at_label(), "notes": evidence.result_summary})).await?; + post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Measurements", &measurement_id, "Record", json!({"campaign_id": campaign_id, "generation_id": generation_id, "candidate_id": selected_id, "traffic_source_id": traffic_source_id, "metric_key": measurement.metric_key, "metric_value": measurement.metric_value, "source_kind": measurement.source_kind, "evidence_locator": measurement.evidence_locator, "evaluator_app_ref": evaluator_ref, "recorded_at": generated_at_label(), "notes": evidence.result_summary})).await?; } post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Candidates", &rejected_id, "Eliminate", json!({"selection_reason": "Outperformed by assessed candidate under frozen design."})).await?; post_protocol_action(client, config, EVOLUTION_NAMESPACE, "Candidates", &selected_id, "Select", json!({"selection_reason": evidence.result_summary})).await?; diff --git a/crates/paw-codex-worker/src/main.rs b/crates/paw-codex-worker/src/main.rs index c9299e606..1c68f91a1 100644 --- a/crates/paw-codex-worker/src/main.rs +++ b/crates/paw-codex-worker/src/main.rs @@ -50,6 +50,9 @@ async fn main() -> Result<()> { if command == WorkerCommand::DirectedEvolutionDemo { return run_directed_evolution_demo(&client, &config).await; } + if command == WorkerCommand::DirectedEvolutionRun { + return run_directed_evolution_run(&client, &config).await; + } if command == WorkerCommand::DirectedEvolutionMutate { return run_directed_evolution_mutation(&config).await; } diff --git a/crates/paw-codex-worker/src/tests.rs b/crates/paw-codex-worker/src/tests.rs index 047da3a7b..3cf67990b 100644 --- a/crates/paw-codex-worker/src/tests.rs +++ b/crates/paw-codex-worker/src/tests.rs @@ -230,6 +230,10 @@ mod tests { parse_worker_command(vec!["directed-evolution-demo".to_string()]), WorkerCommand::DirectedEvolutionDemo ); + assert_eq!( + parse_worker_command(vec!["directed-evolution-run".to_string()]), + WorkerCommand::DirectedEvolutionRun + ); assert_eq!( parse_worker_command(vec!["directed-evolution-mutate".to_string()]), WorkerCommand::DirectedEvolutionMutate @@ -256,6 +260,17 @@ mod tests { assert!(!evolution_candidate_path_allowed("../evaluator/specs/trial.ioa.toml")); } + #[test] + fn generic_evolution_plan_allows_subject_defined_metrics_and_traffic() { + let plan: EvolutionCampaignPlan = serde_json::from_str( + r#"{"campaign_id":"campaign-support","name":"Support Inbox","director_brief":"Improve resolution.","target_app_ref":"owner/support@1111111111111111111111111111111111111111","evaluator_app_ref":"owner/support-eval@2222222222222222222222222222222222222222","brain_provider":"codex","automation_mode":"automatic_release","traffic_sources":[{"id":"ticket-stream","name":"tickets","kind":"real","description":"incoming support tickets"}],"selection_design":{"id":"support-selection","version_label":"v1","evaluator_namespace":"Acme.SupportEvaluation","trial_suite":{"id":"support-suite","name":"Triage","description":"Resolve urgent cases.","scenario_manifest_json":[{"id":"urgent-ticket"}],"hidden_fixture_locator":"temper://fixture","authored_by":"codex"},"fitness_model_json":{"comparison":"preference","signals":["resolution_quality"]},"constraint_definitions_json":[],"traffic_sources_json":["tickets"],"rationale":"Prefer resolved cases.","proposed_by":"codex","approved_by":"human","metrics":[{"id":"resolution-metric","key":"resolution_quality","description":"quality","instrument_kind":"native","instrument_locator":"temper://quality","interpretation":"higher is better","hard_constraint":false}]},"generations":[{"ordinal":"1","parent_release_ref":"owner/support@1111111111111111111111111111111111111111","selected_app_ref":"owner/support@3333333333333333333333333333333333333333"}],"release_control":{"pause_reason":"inspect","rollback_current_ref":"owner/support@1111111111111111111111111111111111111111","rollback_previous_ref":"owner/support@3333333333333333333333333333333333333333","rollback_reason":"rollback"}}"#, + ) + .expect("generic campaign plan should parse"); + assert_eq!(plan.selection_design.metrics[0].key, "resolution_quality"); + assert_eq!(plan.selection_design.evaluator_namespace, "Acme.SupportEvaluation"); + assert_eq!(plan.traffic_sources[0].name, "tickets"); + } + #[tokio::test] async fn live_evolution_binds_releases_to_executed_validator_evidence() { let _guard = ENV_LOCK.lock().await; @@ -263,7 +278,7 @@ mod tests { fs::create_dir_all(path.parent().expect("manifest parent")).expect("manifest parent"); fs::write( &path, - r#"{"evaluator_ref":"owner/evaluator@aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa","records":[{"generation":"1","candidate_ref":"owner/app@1111111111111111111111111111111111111111","status":"Passed","evidence_locator":"temper://trial/g1/answer","result_summary":"generation one passed","resolved_questions":"1.0","answer_evidence":"observed"},{"generation":"2","candidate_ref":"owner/app@2222222222222222222222222222222222222222","status":"Passed","evidence_locator":"temper://trial/g2/answer","result_summary":"generation two passed","resolved_questions":"1.0","answer_evidence":"observed"}]}"#, + r#"{"evaluator_ref":"owner/evaluator@aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa","records":[{"generation":"1","candidate_ref":"owner/app@1111111111111111111111111111111111111111","status":"Passed","evidence_locator":"temper://trial/g1/validator","result_summary":"generation one passed","measurements":[{"suffix":"quality","traffic_source_id":"simulated","metric_key":"quality","metric_value":"0.8","source_kind":"simulated","evidence_locator":"temper://trial/g1/validator"}]},{"generation":"2","candidate_ref":"owner/app@2222222222222222222222222222222222222222","status":"Passed","evidence_locator":"temper://trial/g2/validator","result_summary":"generation two passed","measurements":[{"suffix":"retention","traffic_source_id":"real","metric_key":"workflow_completion","metric_value":"0.92","source_kind":"real","evidence_locator":"temper://trial/g2/validator"}]}]}"#, ) .expect("validator evidence fixture"); let _evidence = EnvOverride::set( @@ -279,7 +294,7 @@ mod tests { true, ) .expect("matching executed evidence"); - assert_eq!(records[1].evidence_locator, "temper://trial/g2/answer"); + assert_eq!(records[1].measurements[0].metric_key, "workflow_completion"); let error = load_evolution_validation_evidence( "owner/evaluator@aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", diff --git a/crates/paw-codex-worker/src/worker_types.rs b/crates/paw-codex-worker/src/worker_types.rs index 9ff94b8f9..640e8ab5d 100644 --- a/crates/paw-codex-worker/src/worker_types.rs +++ b/crates/paw-codex-worker/src/worker_types.rs @@ -29,6 +29,7 @@ enum WorkerCommand { Doctor, LaunchdPlist, DirectedEvolutionDemo, + DirectedEvolutionRun, DirectedEvolutionMutate, } diff --git a/docs/adrs/0052-codex-directed-evolution-brain-provider.md b/docs/adrs/0052-codex-directed-evolution-brain-provider.md index 1355233f1..28d70b7a9 100644 --- a/docs/adrs/0052-codex-directed-evolution-brain-provider.md +++ b/docs/adrs/0052-codex-directed-evolution-brain-provider.md @@ -6,13 +6,21 @@ Accepted. ## Decision -V1 runs the directed-evolution brain through `paw-codex-worker`. The new -`directed-evolution-demo` mode drives the native Genesis protocol and can call -Codex for a selection-design rationale when `PAW_EVOLUTION_USE_CODEX=1`. +V1 runs the directed-evolution brain through `paw-codex-worker`. The +`directed-evolution-run` mode consumes an `EVOLUTION_CAMPAIGN_PLAN_PATH` +manifest so arbitrary Temper-native subjects can supply their own traffic, +trial suite, metrics, capability decisions, generations and release controls. +The plan also declares its evaluator namespace and entity-set names, so the +runner does not depend on the Agent Answers evaluator namespace. +The `directed-evolution-demo` mode remains an Agent Answers convenience entry +point and can call Codex for a selection-design rationale when +`PAW_EVOLUTION_USE_CODEX=1`. Deterministic smoke mode exercises the protocol without an external model call. The `directed-evolution-mutate` mode asks Codex to edit a candidate workspace, then rejects any change outside the Temper-native subject app directories -before Genesis publishes or installs that candidate. +before Genesis publishes or installs that candidate. Its frozen evaluator +compatibility contract is campaign input, rather than a built-in dependency on +the Agent Answers interaction model. Live Codex mode requires the seed and both selected candidate versions to be immutable Genesis commit refs (`owner/app@hash`); it does not accept illustrative candidate labels as releases. From 8366d42250309854a66602d3130906d1a2b64f82 Mon Sep 17 00:00:00 2001 From: Sesh Nalla Date: Tue, 26 May 2026 08:55:54 -0400 Subject: [PATCH 5/5] Fix directed evolution CI checks --- .../src/directed_evolution.rs | 2 +- .../src/directed_evolution_tests.rs | 64 ++++++++++++++++++ crates/paw-codex-worker/src/tests.rs | 66 +------------------ 3 files changed, 66 insertions(+), 66 deletions(-) create mode 100644 crates/paw-codex-worker/src/directed_evolution_tests.rs diff --git a/crates/paw-codex-worker/src/directed_evolution.rs b/crates/paw-codex-worker/src/directed_evolution.rs index bfa7f3c6d..364a99939 100644 --- a/crates/paw-codex-worker/src/directed_evolution.rs +++ b/crates/paw-codex-worker/src/directed_evolution.rs @@ -153,7 +153,7 @@ async fn run_directed_evolution_demo(client: &reqwest::Client, config: &Config) let now = generated_at_label(); let validation_evidence = load_evolution_validation_evidence( &evaluator_ref, - &[(&"1", &generation_one_ref), (&"2", &generation_two_ref)], + &[("1", &generation_one_ref), ("2", &generation_two_ref)], require_pinned_refs, )?; diff --git a/crates/paw-codex-worker/src/directed_evolution_tests.rs b/crates/paw-codex-worker/src/directed_evolution_tests.rs new file mode 100644 index 000000000..14a83dc0b --- /dev/null +++ b/crates/paw-codex-worker/src/directed_evolution_tests.rs @@ -0,0 +1,64 @@ +#[test] +fn live_evolution_requires_immutable_genesis_refs() { + assert_eq!( + evolution_ref("MISSING_REF", "demo/agent-answers@seed", false).expect("smoke ref"), + "demo/agent-answers@seed" + ); + let error = evolution_ref("MISSING_REF", "demo/agent-answers@seed", true) + .expect_err("live evolution must reject a label ref"); + assert!(format!("{error:#}").contains("immutable Genesis ref")); +} + +#[test] +fn evolution_candidate_changes_are_limited_to_native_bundle_files() { + assert!(evolution_candidate_path_allowed("specs/answer.ioa.toml")); + assert!(evolution_candidate_path_allowed("wasm/validator/src/lib.rs")); + assert!(evolution_candidate_path_allowed("adrs/0002-evidence.md")); + assert!(!evolution_candidate_path_allowed("crates/random-helper/src/lib.rs")); + assert!(!evolution_candidate_path_allowed("../evaluator/specs/trial.ioa.toml")); +} + +#[test] +fn generic_evolution_plan_allows_subject_defined_metrics_and_traffic() { + let plan: EvolutionCampaignPlan = serde_json::from_str( + r#"{"campaign_id":"campaign-support","name":"Support Inbox","director_brief":"Improve resolution.","target_app_ref":"owner/support@1111111111111111111111111111111111111111","evaluator_app_ref":"owner/support-eval@2222222222222222222222222222222222222222","brain_provider":"codex","automation_mode":"automatic_release","traffic_sources":[{"id":"ticket-stream","name":"tickets","kind":"real","description":"incoming support tickets"}],"selection_design":{"id":"support-selection","version_label":"v1","evaluator_namespace":"Acme.SupportEvaluation","trial_suite":{"id":"support-suite","name":"Triage","description":"Resolve urgent cases.","scenario_manifest_json":[{"id":"urgent-ticket"}],"hidden_fixture_locator":"temper://fixture","authored_by":"codex"},"fitness_model_json":{"comparison":"preference","signals":["resolution_quality"]},"constraint_definitions_json":[],"traffic_sources_json":["tickets"],"rationale":"Prefer resolved cases.","proposed_by":"codex","approved_by":"human","metrics":[{"id":"resolution-metric","key":"resolution_quality","description":"quality","instrument_kind":"native","instrument_locator":"temper://quality","interpretation":"higher is better","hard_constraint":false}]},"generations":[{"ordinal":"1","parent_release_ref":"owner/support@1111111111111111111111111111111111111111","selected_app_ref":"owner/support@3333333333333333333333333333333333333333"}],"release_control":{"pause_reason":"inspect","rollback_current_ref":"owner/support@1111111111111111111111111111111111111111","rollback_previous_ref":"owner/support@3333333333333333333333333333333333333333","rollback_reason":"rollback"}}"#, + ) + .expect("generic campaign plan should parse"); + assert_eq!(plan.selection_design.metrics[0].key, "resolution_quality"); + assert_eq!(plan.selection_design.evaluator_namespace, "Acme.SupportEvaluation"); + assert_eq!(plan.traffic_sources[0].name, "tickets"); +} + +#[tokio::test] +async fn live_evolution_binds_releases_to_executed_validator_evidence() { + let _guard = ENV_LOCK.lock().await; + let path = unique_temp_dir().join("validator-evidence.json"); + fs::create_dir_all(path.parent().expect("manifest parent")).expect("manifest parent"); + fs::write( + &path, + r#"{"evaluator_ref":"owner/evaluator@aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa","records":[{"generation":"1","candidate_ref":"owner/app@1111111111111111111111111111111111111111","status":"Passed","evidence_locator":"temper://trial/g1/validator","result_summary":"generation one passed","measurements":[{"suffix":"quality","traffic_source_id":"simulated","metric_key":"quality","metric_value":"0.8","source_kind":"simulated","evidence_locator":"temper://trial/g1/validator"}]},{"generation":"2","candidate_ref":"owner/app@2222222222222222222222222222222222222222","status":"Passed","evidence_locator":"temper://trial/g2/validator","result_summary":"generation two passed","measurements":[{"suffix":"retention","traffic_source_id":"real","metric_key":"workflow_completion","metric_value":"0.92","source_kind":"real","evidence_locator":"temper://trial/g2/validator"}]}]}"#, + ) + .expect("validator evidence fixture"); + let _evidence = EnvOverride::set( + "EVOLUTION_VALIDATOR_EVIDENCE_PATH", + path.as_os_str().to_os_string(), + ); + let records = load_evolution_validation_evidence( + "owner/evaluator@aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + &[ + ("1", "owner/app@1111111111111111111111111111111111111111"), + ("2", "owner/app@2222222222222222222222222222222222222222"), + ], + true, + ) + .expect("matching executed evidence"); + assert_eq!(records[1].measurements[0].metric_key, "workflow_completion"); + + let error = load_evolution_validation_evidence( + "owner/evaluator@aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + &[("2", "owner/app@3333333333333333333333333333333333333333")], + true, + ) + .expect_err("unexecuted candidate must not release"); + assert!(format!("{error:#}").contains("validator evidence missing generation 2")); +} diff --git a/crates/paw-codex-worker/src/tests.rs b/crates/paw-codex-worker/src/tests.rs index 3cf67990b..b3dad9a73 100644 --- a/crates/paw-codex-worker/src/tests.rs +++ b/crates/paw-codex-worker/src/tests.rs @@ -16,6 +16,7 @@ mod tests { include!("codex_plan_tests.rs"); include!("pull_request_tests.rs"); include!("worker_http_tests.rs"); + include!("directed_evolution_tests.rs"); static ENV_LOCK: Mutex<()> = Mutex::const_new(()); @@ -240,71 +241,6 @@ mod tests { ); } - #[test] - fn live_evolution_requires_immutable_genesis_refs() { - assert_eq!( - evolution_ref("MISSING_REF", "demo/agent-answers@seed", false).expect("smoke ref"), - "demo/agent-answers@seed" - ); - let error = evolution_ref("MISSING_REF", "demo/agent-answers@seed", true) - .expect_err("live evolution must reject a label ref"); - assert!(format!("{error:#}").contains("immutable Genesis ref")); - } - - #[test] - fn evolution_candidate_changes_are_limited_to_native_bundle_files() { - assert!(evolution_candidate_path_allowed("specs/answer.ioa.toml")); - assert!(evolution_candidate_path_allowed("wasm/validator/src/lib.rs")); - assert!(evolution_candidate_path_allowed("adrs/0002-evidence.md")); - assert!(!evolution_candidate_path_allowed("crates/random-helper/src/lib.rs")); - assert!(!evolution_candidate_path_allowed("../evaluator/specs/trial.ioa.toml")); - } - - #[test] - fn generic_evolution_plan_allows_subject_defined_metrics_and_traffic() { - let plan: EvolutionCampaignPlan = serde_json::from_str( - r#"{"campaign_id":"campaign-support","name":"Support Inbox","director_brief":"Improve resolution.","target_app_ref":"owner/support@1111111111111111111111111111111111111111","evaluator_app_ref":"owner/support-eval@2222222222222222222222222222222222222222","brain_provider":"codex","automation_mode":"automatic_release","traffic_sources":[{"id":"ticket-stream","name":"tickets","kind":"real","description":"incoming support tickets"}],"selection_design":{"id":"support-selection","version_label":"v1","evaluator_namespace":"Acme.SupportEvaluation","trial_suite":{"id":"support-suite","name":"Triage","description":"Resolve urgent cases.","scenario_manifest_json":[{"id":"urgent-ticket"}],"hidden_fixture_locator":"temper://fixture","authored_by":"codex"},"fitness_model_json":{"comparison":"preference","signals":["resolution_quality"]},"constraint_definitions_json":[],"traffic_sources_json":["tickets"],"rationale":"Prefer resolved cases.","proposed_by":"codex","approved_by":"human","metrics":[{"id":"resolution-metric","key":"resolution_quality","description":"quality","instrument_kind":"native","instrument_locator":"temper://quality","interpretation":"higher is better","hard_constraint":false}]},"generations":[{"ordinal":"1","parent_release_ref":"owner/support@1111111111111111111111111111111111111111","selected_app_ref":"owner/support@3333333333333333333333333333333333333333"}],"release_control":{"pause_reason":"inspect","rollback_current_ref":"owner/support@1111111111111111111111111111111111111111","rollback_previous_ref":"owner/support@3333333333333333333333333333333333333333","rollback_reason":"rollback"}}"#, - ) - .expect("generic campaign plan should parse"); - assert_eq!(plan.selection_design.metrics[0].key, "resolution_quality"); - assert_eq!(plan.selection_design.evaluator_namespace, "Acme.SupportEvaluation"); - assert_eq!(plan.traffic_sources[0].name, "tickets"); - } - - #[tokio::test] - async fn live_evolution_binds_releases_to_executed_validator_evidence() { - let _guard = ENV_LOCK.lock().await; - let path = unique_temp_dir().join("validator-evidence.json"); - fs::create_dir_all(path.parent().expect("manifest parent")).expect("manifest parent"); - fs::write( - &path, - r#"{"evaluator_ref":"owner/evaluator@aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa","records":[{"generation":"1","candidate_ref":"owner/app@1111111111111111111111111111111111111111","status":"Passed","evidence_locator":"temper://trial/g1/validator","result_summary":"generation one passed","measurements":[{"suffix":"quality","traffic_source_id":"simulated","metric_key":"quality","metric_value":"0.8","source_kind":"simulated","evidence_locator":"temper://trial/g1/validator"}]},{"generation":"2","candidate_ref":"owner/app@2222222222222222222222222222222222222222","status":"Passed","evidence_locator":"temper://trial/g2/validator","result_summary":"generation two passed","measurements":[{"suffix":"retention","traffic_source_id":"real","metric_key":"workflow_completion","metric_value":"0.92","source_kind":"real","evidence_locator":"temper://trial/g2/validator"}]}]}"#, - ) - .expect("validator evidence fixture"); - let _evidence = EnvOverride::set( - "EVOLUTION_VALIDATOR_EVIDENCE_PATH", - path.as_os_str().to_os_string(), - ); - let records = load_evolution_validation_evidence( - "owner/evaluator@aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", - &[ - ("1", "owner/app@1111111111111111111111111111111111111111"), - ("2", "owner/app@2222222222222222222222222222222222222222"), - ], - true, - ) - .expect("matching executed evidence"); - assert_eq!(records[1].measurements[0].metric_key, "workflow_completion"); - - let error = load_evolution_validation_evidence( - "owner/evaluator@aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", - &[("2", "owner/app@3333333333333333333333333333333333333333")], - true, - ) - .expect_err("unexecuted candidate must not release"); - assert!(format!("{error:#}").contains("validator evidence missing generation 2")); - } - #[test] fn launchd_plist_renders_concrete_worker_environment() { let config = Config {