Files
waggle-os/benchmarks/results/manifest-v4-preregistration.yaml
Oleg Maslov 0c3e2ead3b
Some checks failed
Installer Smoke / installer-smoke (push) Has been cancelled
moving
2026-09-02 10:10:29 +02:00

458 lines
19 KiB
YAML
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# Manifest v4 — Task 2.5 Stage 3 N=400 Pre-Registration (machine-readable twin)
# Canonical markdown surface: manifest-v4-preregistration.md
# Any change to this file requires new PM-ratified decision doc + manifest v5.
# SHA-256 of this YAML's bytes is recorded in the anchor commit message.
manifest_version: v4.0.0-preregistration
manifest_type: stage_3_n400_preregistration
locked_date: 2026-04-24
authority: PM (Marko Markovic) — Option 1 (N=400 direct) ratified 2026-04-24 on Stage 2-Retry Gate C PARTIAL PASS exit
sprint: 12
task: 2.5
stage: 3
branch: feature/c3-v3-wrapper
code_freeze_head: 373516c2784807da8536dbc0c194c54f4e4cd4be
code_freeze_head_short: 373516c
supersedes: stage_2_retry_n20_gate_c
inherits_from: bench_spec_lock_v1_2026_04_22
# ── Field 7 slots (preregistration.ts PreregistrationManifestPayload) ───────
manifest_path: benchmarks/results/manifest-v4-preregistration.yaml
manifest_locked_at: 2026-04-24T00:00:00Z
dataset:
name: locomo
source_url: https://raw.githubusercontent.com/snap-research/locomo/main/data/locomo10.json
raw_archive_path: benchmarks/data/locomo10.json
raw_archive_sha256: 79fa87e90f04081343b8c8debecb80a9a6842b76a7aa537dc9fdf651ea698ff4
raw_archive_bytes: 2805274
canonical_path: benchmarks/data/locomo/locomo-1540.jsonl
canonical_sha256: 39e415e2f3a0fa1bd3cb1804a58d0b440b50d3070b2100698437e4ec402a5b24
canonical_instance_count: 1531
paper_total_claim: 1540
paper_reference: "Maharana et al., ACL-2024 — Evaluating Very Long-Term Conversational Memory of LLM Agents"
category_distribution:
single_hop: 841
multi_hop: 281
temporal: 320
open_ended: 89
# ── Primary hypothesis (directional, confirmatory) ──────────────────────────
primary_hypothesis:
name: memory_lift_retrieval_vs_no_context
direction: one_sided_positive
statement: "retrieval_judge_accuracy no-context_judge_accuracy ≥ 5pp"
test: fisher_exact_one_sided
alpha_threshold: 0.10
effect_size_threshold_pp: 5
justification_ex_ante:
- gate_b_dry_run_conv_scope_20_of_20_vs_whole_corpus_8_of_20_leak_2026_04_24
- gate_c_monotonicity_no_context_0_10_lt_retrieval_0_35_lt_agentic_0_40_lt_oracle_0_55
# ── Secondary endpoints (ex-ante, non-blocking, descriptive) ────────────────
secondary_endpoints:
S1_monotonicity_no_context_leq_retrieval:
direction: one_sided_positive
threshold_pp: 0
test: fisher_exact_one_sided
alpha: 0.20
S2_monotonicity_retrieval_leq_agentic:
direction: one_sided_positive
threshold_pp: 0
test: fisher_exact_one_sided
alpha: 0.20
S3_monotonicity_agentic_leq_oracle_context:
direction: one_sided_positive
threshold_pp: 0
test: fisher_exact_one_sided
alpha: 0.20
S4_agentic_lift_over_retrieval:
direction: descriptive
threshold_pp: 0
report: [point_estimate, wilson_95_ci]
S5_abstain_penalty_oracle_minus_full_context:
direction: descriptive_expected_positive
report: [point_estimate]
rationale: "SYSTEM_EVOLVED strict-abstain prompt is expected to reduce full-context accuracy below oracle-context on instances where the oracle answer is ambiguous; diagnostic only, no gating"
# ── Sample design ───────────────────────────────────────────────────────────
sample:
cells:
- no-context
- oracle-context
- full-context
- retrieval
- agentic
n_per_cell: 400
total_evaluations: 2000
instance_selection_seed: 42
instance_selection_method: "shuffle-then-take-first-N, deterministic given seed"
matched_pairs: true
concurrency: 2
# ── Cells semantics (frozen at HEAD 373516c) ────────────────────────────────
cells_semantics:
no_context:
system_prompt: SYSTEM_BASELINE
user_prompt: "Question: {question}"
memory_injection: none
added_at: stage_2_retry_1_1_2026_04_24
oracle_context:
system_prompt: SYSTEM_BASELINE
user_prompt: "Context: {instance.context}\\n\\nQuestion: {instance.question}"
memory_injection: oracle_fed_by_locomo
harness_alias: raw
note: "PM-facing alias for harness `raw` (oracle-fed on LoCoMo; not actually zero-memory)"
full_context:
system_prompt: SYSTEM_EVOLVED
user_prompt: "# Recalled Memories\\n- [memory:synth] {instance.context}\\n\\nQuestion: {instance.question}"
memory_injection: oracle_fed_plus_evolved_abstain
retrieval:
system_prompt: SYSTEM_BASELINE
substrate: waggle_core_hybrid_search
scope: conversation_scoped_via_gopId
top_k_default: 20
top_k_upper_clamp: 50
agentic:
system_prompt: SYSTEM_AGENTIC_softened_stage2_retry
system_prompt_sha256: 6facae6decc44a6404290514accb4f7cb364081b32d02847a20f8e871633e328
system_prompt_bytes: 1467
tool_allowlist:
- search_memory
tool_binding: "search_memory is bound to instance.conversation_id via makeSearchMemoryTool(substrate, 20, boundToGopId); non-overridable by agent"
max_turns: 3
timeout_ms: 180000
forced_answer_fallback:
enabled: true
system_prompt: SYSTEM_AGENTIC_FORCED_FALLBACK
trigger: "resp.content.trim() === '' && capturedToolResults.length > 0 on runAgentLoop exit"
gate_c_firing_rate: 0
# ── Model stack ─────────────────────────────────────────────────────────────
subject_model: qwen3.6-35b-a3b-via-dashscope-direct
subject_fallback_1: qwen3.6-35b-a3b-via-openrouter
subject_fallback_2: NOT_AVAILABLE
subject_route_table:
primary:
alias: qwen3.6-35b-a3b-via-dashscope-direct
litellm_model: qwen3.6-35b-a3b-via-dashscope-direct
upstream_route: "LiteLLM local alias -> openai/qwen3.6-35b-a3b @ https://dashscope-intl.aliyuncs.com/compatible-mode/v1"
provider: alibaba
thinking: on
max_tokens: 16000
reasoning_shape: dashscope-native
price_per_million_input_usd: 0.20
price_per_million_output_usd: 0.80
context_window: 262144
pinning_surface: floating_alias
pinning_surface_carve_out_reason: "DashScope-intl does not expose immutable model snapshots; floating alias mandated by B3 addendum § 5"
note: "TRUE Qwen 3.6-35B-A3B (not the 3.5 regress of the OpenRouter bridge)"
fallback_1:
alias: qwen3.6-35b-a3b-via-openrouter
litellm_model: qwen3.6-35b-a3b-via-openrouter
upstream_route: "LiteLLM -> OpenRouter bridge (openrouter/qwen/qwen3.5-35b-a3b)"
provider: alibaba_via_openrouter
thinking: on
max_tokens: 64000
reasoning_shape: openrouter-unified
pinning_surface: floating_alias
note: "Known 3.5 regress via OpenRouter bridge; only used on primary network failure"
trigger_condition: fetch_error_on_primary
fallback_2:
alias: NOT_AVAILABLE
note: "No third fallback; halt if fallback_1 also fails"
judge_ensemble:
primary:
- judge_role: primary
slot: primary_judge_1
model_id: claude-opus-4-7
provider: anthropic
litellm_model: claude-opus-4-7
pinning_surface: anthropic_immutable
pinning_surface_carve_out_reason: null
price_per_million_input_usd: 15.00
price_per_million_output_usd: 75.00
- judge_role: primary
slot: primary_judge_2
model_id: gpt-5.4
provider: openai_via_openrouter
litellm_model: gpt-5.4
pinning_surface: floating_alias
pinning_surface_carve_out_reason: "Direct OpenAI Chat Completions via LiteLLM local alias `gpt-5.4` -> `openai/gpt-5.4`; OpenAI does not expose immutable model snapshots for gpt-5.x; B3 addendum § 5"
price_per_million_input_usd: 10.00
price_per_million_output_usd: 30.00
- judge_role: primary
slot: primary_judge_3
model_id: gemini-3.1-pro
provider: google_via_openrouter
litellm_model: gemini-3.1-pro
pinning_surface: floating_alias
pinning_surface_carve_out_reason: "Direct Google AI Studio via LiteLLM local alias `gemini-3.1-pro` -> `gemini/gemini-3.1-pro-preview`; no stable alias (preview-only); B3 addendum § 5"
price_per_million_input_usd: 3.50
price_per_million_output_usd: 10.50
tiebreak:
judge_role: reserve
model_id: grok-4.20
provider: xai_via_openrouter
litellm_model: openrouter/x-ai/grok-4.20
pinning_surface: floating_alias
pinning_surface_carve_out_reason: "xAI does not expose immutable model snapshots through OpenRouter routing layer; floating alias mandated by B3 addendum § 5"
price_per_million_input_usd: 5.00
price_per_million_output_usd: 15.00
trigger: three_way_split_1_1_1
path_enum: quadri-vendor
defensive_2_2_path: pm-escalation
consistency_constraint: same_physical_judge_models_as_stage_1_stage_1_5_stage_2_stage_2_retry
vote_policy: majority_with_grok_reserve_on_1_1_1_split
snapshot_drift_policy: halt_and_manifest_flag
# Convenience flat list for preregistration.ts emitter compatibility
judge_primary:
id: claude-opus-4-7
judge_secondary:
id: gpt-5.4
judge_tie_breaker:
id: gemini-3.1-pro
# ── Substrate ───────────────────────────────────────────────────────────────
substrate:
implementation: "@waggle/core::HybridSearch (RRF-fused FTS5 + vec0)"
ingest_keying: "gop_id = conversation_id"
scope_filter:
parameter: gopId
source_location: packages/core/src/mind/search.ts:14
field_name: SearchOptions.gopId
non_overridable: true
benchmark_binding: instance.conversation_id
top_k_default: 20
top_k_upper_clamp: 50
embedder:
factory: createOllamaEmbedder
base_url: http://localhost:11434
model: nomic-embed-text
dims: 1024
cost: zero_local_inference
ingest_batch_size: 200
code_reference: benchmarks/harness/src/substrate.ts
# ── κ monitoring (inherited from Bench-Spec LOCK v1) ────────────────────────
kappa_monitoring:
baseline_reference: sprint_10_task_2_2_kappa_0_7458
compute: fleiss_kappa_on_pre_tiebreak_vote_matrix
thresholds:
pass_no_flag_kappa_min: 0.65
pass_with_flag_kappa_range: [0.60, 0.65]
halt_kappa_max: 0.60
halt_drop_from_baseline_max_pp: 10
halt_protocol: preserve_partial_jsonl_write_halted_session_ping_notify_pm
# ── Confidence intervals (inherited from Bench-Spec LOCK v1) ────────────────
confidence_intervals:
primary:
method: wilson_score_95
description: "Frequentist binomial CI on instance-level binary verdicts."
secondary:
method: cluster_bootstrap_95
iterations: 10000
seed: 42
cluster_unit: conversation_id
resample_mode: cluster_level_with_replacement
quantiles: [2.5, 97.5]
# ── Failure taxonomy (inherited from Bench-Spec LOCK v1) ────────────────────
failure_taxonomy:
version: v1
categories:
- {code: F1, name: contradicts_ground_truth}
- {code: F2, name: partial_answer}
- {code: F3, name: off_topic}
- {code: F4, name: refusal}
- {code: F5, name: tool_use_error, scope: tool_permitted_cells_only}
- {code: F6, name: format_violation}
special:
null_correct:
description: judge_majority_verdict_correct_no_f_code
f_other:
description: failure_outside_f1_f6_taxonomy
mandatory_rationale_min_words: 10
rate_threshold_for_taxonomy_review_pct: 10
# ── Stopping rules (ex-ante, no interim looks) ──────────────────────────────
stopping_rules:
budget_hard_halt_usd: 28.00
budget_cap_usd: 30.00
streak_halt: "3 consecutive fetch failures on same model -> halt (streak-tracker.ts)"
pre_cell_health_check: "GET /health/liveliness + POST /v1/chat/completions ping per model -> halt before cell on any 5xx/fetch-error (health-check.ts)"
runner_lock: "concurrent_runners: FORBIDDEN (runner-lock.ts); contention -> halt"
deviation_from_preregistration: "any change to manifest v4 §1-§9 during run -> immediate halt + PM raise"
no_interim_looks: true
mid_run_amendment_policy: halt_restart_required
# ── Post-hoc exclusion policy: NONE ─────────────────────────────────────────
post_hoc_exclusion:
policy: none
evaluator_loss_handling:
included_in_denominator: true
reported_separately: true
denominator_formula: "correct + incorrect + evaluator_loss"
subject_failures_classified_as: failure_mode_per_f1_f6_taxonomy
instance_whitelist: none
instance_blacklist: none
# ── Budget ──────────────────────────────────────────────────────────────────
budget:
cap_usd: 30.00
hard_halt_usd: 28.00
expected_burn_usd: 23.00
variance_ceiling_usd: 28.00
breakdown_expected:
subject_qwen_dashscope_direct_usd: 2.50
judge_triple_opus_gpt5_gemini_usd: 20.00
embedding_ollama_local_usd: 0.00
tie_break_grok_reserve_usd: 0.50
scale_from_stage_2_retry_gate_c: "Gate C $1.1637 / 100 evals × 20 = $23.27"
# ── Target sample + CLI invocation ──────────────────────────────────────────
target_N: 400
target_cells:
- no-context
- oracle-context
- full-context
- retrieval
- agentic
target_total_evaluations: 2000
target_concurrency: 2
cli_invocation_template: >
npx tsx scripts/run-mini-locomo.ts
--manifest benchmarks/results/manifest-v4-preregistration.yaml
--subject qwen3.6-35b-a3b-via-dashscope-direct
--subject-fallback-1 qwen3.6-35b-a3b-via-openrouter
--judge-ensemble claude-opus-4-7,gpt-5.4,gemini-3.1-pro
--v3-cells --N 400 --parallel-concurrency 2 --seed 42
# ── Code freeze assertions ──────────────────────────────────────────────────
code_freeze:
head: 373516c2784807da8536dbc0c194c54f4e4cd4be
branch: feature/c3-v3-wrapper
frozen_paths:
- benchmarks/harness/src/cells.ts
- benchmarks/harness/src/substrate.ts
- benchmarks/harness/src/judge-client.ts
- benchmarks/harness/src/judge-runner.ts
- benchmarks/harness/src/health-check.ts
- benchmarks/harness/src/streak-tracker.ts
- benchmarks/harness/src/runner-lock.ts
- benchmarks/harness/src/runner.ts
- benchmarks/harness/config/models.json
- packages/agent/src/agent-loop.ts
- packages/agent/src/tools.ts
- packages/core/src/mind/search.ts
- packages/core/src/mind/frames.ts
- packages/core/src/mind/sessions.ts
- packages/core/src/mind/db.ts
- litellm-config.yaml
permitted_delta_during_run:
- "new JSONL files emitted to benchmarks/results/ by the N=400 run"
- "no other file modifications"
# ── Deviation policy ────────────────────────────────────────────────────────
deviation_policy:
on_detection:
- immediate_halt
- pm_raise
- re_preregister_new_manifest_v5_if_accepted
consistency_with_bench_spec_lock_v1: preregistration.mid_run_amendment_policy == halt_restart_required
# ── PM gates ────────────────────────────────────────────────────────────────
pm_gates:
gate_p_pre_run:
trigger: "anchor commit of manifest v4 md + yaml on feature/c3-v3-wrapper"
action: "CC-1 halts; awaits PM GO on pre-registration content before N=400 kick"
pm_checks:
- "pre-registration content matches Stage 2-Retry Gate C ratifications"
- "§1-§10 locked unambiguously for Gate D adjudication"
outcomes:
- GO: "§1.2 N=400 kickoff authorized"
- REVISE: "Gate P loop"
gate_d_post_run:
trigger: "N=400 run exit (clean completion or halted per stopping_rules)"
action: "CC-1 writes Gate D exit report at PM-Waggle-OS/sessions/2026-04-24-task25-stage3-n400-complete.md; halts"
pm_checks:
- "primary endpoint pass/fail (Fisher one-sided p < 0.10 on retrieval - no-context ≥ 5pp)"
- "secondary endpoint summary (monotonicity chain, agentic lift, abstain penalty)"
- "deviation count (expected: 0)"
- "budget usage vs $28 hard halt / $30 cap"
- "evaluator-loss count"
outcomes:
- compose_sota_claim_authority_pm
- publish_gate
- further_scope_matched_mem0_or_multi_model
cc1_self_advance: forbidden_at_both_gates
# ── Scope boundaries ────────────────────────────────────────────────────────
scope_boundaries:
claimable_at_gate_d:
- memory_lift_magnitude_and_significance_conv_scope_qwen_harness_head_373516c
- per_cell_judge_accuracy_wilson_95
- monotonicity_chain_observation_5_cell
- conv_scope_fair_comparison_methodology
- agentic_discipline_search_rate_turns_unknown_fallback
not_claimable_at_gate_d:
- direct_comparability_to_mem0_91_6_different_scope_and_memory_layer
- multi_model_generalization_stage_3_is_qwen_only
- production_waggle_orchestrator_performance
reserved_for_pm_at_gate_d:
- public_claim_phrasing_venue
- matched_scope_mem0_co_run_stage_4
- publication_timing
cc1_does_not_compose_public_sota_claim: true
# ── Related artefacts ───────────────────────────────────────────────────────
related:
bench_spec_lock_v1_parent: PM-Waggle-OS/decisions/2026-04-22-bench-spec-locked.manifest.yaml
stage_2_retry_gate_c_exit: PM-Waggle-OS/sessions/2026-04-24-task25-stage2-retry-complete.md
stage_3_brief: PM-Waggle-OS/briefs/2026-04-24-cc-task25-stage3-n400-kickoff.md
stage_2_retry_commit: 373516c
rollback_tag: checkpoint/pre-self-evolution-2026-04-14
canonical_md_surface: benchmarks/results/manifest-v4-preregistration.md
# ── Validation gates inherited from Bench-Spec LOCK v1 ──────────────────────
validation_gates:
before_n400_kickoff:
- anchor_commit_sha_recorded
- manifest_yaml_sha256_recorded_in_commit_message
- pm_gate_p_ratification_received
- health_check_green_on_subject_plus_3_judges
- no_concurrent_runner
at_gate_d_exit:
- all_2000_evals_accounted_in_denominators
- evaluator_loss_reported_separately
- primary_fisher_one_sided_computed
- secondary_endpoints_reported
- budget_reconciled
- deviation_count_reported
- code_freeze_reverified_head_373516c