feat: unify task-first native model runtime #380
Closed
nsaspy
wants to merge 16 commits from
prolog-rlm-v1 into main
pull from: prolog-rlm-v1
merge into: nsaspy:main
nsaspy:main
nsaspy:feat/prolog-rlm-language
nsaspy:codex/claude-api-provider
nsaspy:codex/symbolic-experts
nsaspy:codex/openai-api-provider
nsaspy:feat/zai-provider-protocols
nsaspy:hardening/project-semantic-20260911
nsaspy:hardening/ws7-nix-ci-pinning
nsaspy:chore/track-prolog
nsaspy:rage/98-semantic-project-knowledge
nsaspy:fix/339-agent-zero-skill-graph
nsaspy:ci/tree-sitter-runner-labels
nsaspy:issue-293
nsaspy:research/text-streaming
nsaspy:docs/agent-editor-handoff
nsaspy:icon-add
nsaspy:hydra/fleet-jobs-20260904
nsaspy:rage/355-d6-11-plan-native-dispatch
nsaspy:cd/nix-installer-action-url
nsaspy:issue-355
nsaspy:rage/97-query-capture-apis
nsaspy:research/expert-direct-tool-projection
nsaspy:rage/96-versioned-syntax-facts
nsaspy:fix/direct-deadline-typing
nsaspy:dogfood/auto-dig-native-tools
nsaspy:feature/rage-feature-freeze
nsaspy:fix/323-native-batch-cardinality
nsaspy:fix/328-direct-user-namespace-text-string
nsaspy:fix/transient-provider-retry-2026-09-01
nsaspy:fix/openrouter-dotted-tool-result-name
nsaspy:rage/325-native-call-isolation
nsaspy:fix/316-original-batch-effect-isolation
nsaspy:fix/313-per-call-preflight
nsaspy:fix/312-peek-schema-contract
nsaspy:docs/agents-worktree-rule
nsaspy:fix/child-native-capability-narrowing
nsaspy:test/paid-lane-glm-53-flash
nsaspy:docs/spec-seeded-symbolic-plans
nsaspy:rage/290-spec-plan-authority
nsaspy:fix/304-context-peek-selector-contract
nsaspy:fix/298-native-any-json-schema
nsaspy:rage/288-spec-plan-graph-executor
nsaspy:feat/spec-plan-flow-api
nsaspy:gpt-5-6-sol-high/questions-for-rlm-prolog-and-lambda-rlm
nsaspy:rage/277-planner-protocol-context
nsaspy:research-approval/rlm-research-026-task-deadlines-20260827135838
nsaspy:research-approval/rlm-research-025-lem-ui-20260827135751
nsaspy:rage/183-live-operator-behavior
nsaspy:agent/127-agentprolog-config
nsaspy:recovery/132-agentprolog-monorepo
nsaspy:rage/223-durable-context-mount-recovery
nsaspy:rage/223-constraint-benchmark-recovery
nsaspy:research-approval/rlm-research-011-managed-context-tool-discovery-20260827054019
nsaspy:feat/research-approval-schema
nsaspy:agent/rrlm-control-plane-research
nsaspy:docs/219-adrrd-review
nsaspy:salvage/231-runtime-status
nsaspy:salvage/231-runtime-status-run
nsaspy:tmp
nsaspy:tmp2
nsaspy:tmp3
nsaspy:rage/245-planner-structural-retry
nsaspy:rage/168-skill-selection-eval
nsaspy:rage/250-skill-catalog-graph
nsaspy:rage/56-result-acceptance
nsaspy:rage/172-parent-resume-replan
nsaspy:rage/175-deadline-policy-recovery
nsaspy:rage/257-provider-tool-choice-normalization
nsaspy:agent/tool-result-projection-presets
nsaspy:fix/234-numeric-schema-bounds
nsaspy:feature/231-rlm-cli-reference-harness
nsaspy:feature/223-durable-context-mounts
nsaspy:feature/223-real-constraint-benchmark
nsaspy:feature/223-real-constraint-benchmark-clean
nsaspy:feature/223-real-constraint-benchmark-final
nsaspy:feature/223-real-constraint-benchmark-impl
nsaspy:feature/223-real-constraint-benchmark-now
nsaspy:feature/223-real-constraint-benchmark-tdd
nsaspy:feature/223-real-constraint-benchmark-work
nsaspy:rage/176-root-planner-tool-projection
nsaspy:rage/172-typed-delegation-policy
nsaspy:rage/175-subagent-deadline-policy
nsaspy:rage/206-prompt-command-runtime
nsaspy:rage/203-subagent-skill-role-provenance
nsaspy:rage/200-permanent-rlm-context
nsaspy:fix/190-cli-help-success
nsaspy:archive/pr-132-agentprolog-config-20260827
nsaspy:feature/117-prolog-skill-activation-linear
nsaspy:feature/117-prolog-skill-activation
nsaspy:fix/194-completion-budget-usage
nsaspy:fix/191-capability-filtered-tool-schemas
nsaspy:fix/185-reasoning-effort
nsaspy:ci/report-workflow-failures-20260825
nsaspy:cleanup/186-remove-legacy-harnesses
nsaspy:fix/185-reasoning-routing
nsaspy:agent/evolution-async-evaluator
nsaspy:rage/181-binding-replay-race
nsaspy:agent/124-deepseek-harness-prolog
nsaspy:rage/144-fallback-closure
nsaspy:rage/164-conversation-metadata
nsaspy:agent/subagent-supervised-call-conformance
nsaspy:fix/160-context-adapter-closed-data
nsaspy:codex/move-agent-zero-adaptor
nsaspy:codex/sol-high-integration
nsaspy:fix/151-plunit-gate
nsaspy:fix/165-evolution-closed-data
nsaspy:fix/162-async-control-exceptions
nsaspy:fix/158-prompt-compiler-closed-dicts
nsaspy:fix/151-plunit-main-ownership
nsaspy:fix/156-registry-destroy-hooks
nsaspy:fix/154-anonymous-dict-canonicalization
nsaspy:fix/flake-lock-reproducibility
nsaspy:agent/issue-142-evolution-kernel
nsaspy:agent/141-flake-runtime-package
nsaspy:agent/rlm-subagent-runtime
nsaspy:agent/144-subagent-fallback-a
nsaspy:agent/107-bound-adapter-metadata
nsaspy:validation/clean-pack-install
nsaspy:fix/45-authoritative-nested-model-events
nsaspy:fix/46-router-safe-live-streaming
nsaspy:agent/prompt-context-compiler
nsaspy:agent/42-canonical-recursive-fingerprints
nsaspy:agent/136-static-load-errors-fail-ci
nsaspy:agent/44-completion-error-usage
nsaspy:fix/67-loader-registry-cleanup
nsaspy:agent/opentui-solid-reference-client-current
nsaspy:agent/95-project-source-registry
nsaspy:feature/issue-117-prolog-skill-compiler
nsaspy:backlog/roadmap-86-merged
nsaspy:agent/opentui-solid-reference-client
nsaspy:79-tool-effect-boundary
nsaspy:agent/prolog-agent-ui-research
nsaspy:agent/conversation-cold-context
nsaspy:agent/94-tree-sitter-ffi
nsaspy:agent/conversation-warm-context
nsaspy:111-opentui-solid-reference-client
nsaspy:109-prolog-agent-ui-v1
nsaspy:agent/conversation-runtime
nsaspy:agent/spec-mode-language
nsaspy:agent/spec-verify-foundation
nsaspy:agent/prompt-compiler-research
nsaspy:reconcile-backlog-postmerge
nsaspy:reconcile-backlog-issues
nsaspy:agent/prolog-agent-roadmap
nsaspy:feature/issue-84-effect-store-migration
nsaspy:fix/issue-80-effect-substrate-adversarial-hardening
nsaspy:feature/issue-57-effect-identity
nsaspy:agent/mcp-declaration-security
nsaspy:agent/external-tool-category-boundary
nsaspy:feature/issue-53-authority-pending-async
nsaspy:feature/issue-54-agent-graph-canonical-async
nsaspy:feature/issue-54-tools-mcp-canonical-async
nsaspy:feature/issue-54-async-canonical-runtime
nsaspy:agent/dual-sync-async-runtime
nsaspy:agent/reconcile-todo-status
nsaspy:feature/issue-20-deep-recursion-experiments
nsaspy:feature/issue-19-cli-demo-trace
nsaspy:feature/issue-18-benchmark-conformance
nsaspy:feature/issue-17-adaptive-recursion
nsaspy:feature/issue-16-durable-artifacts
nsaspy:feature/issue-15-mcp-2026-dual-version
nsaspy:feature/issue-14-mcp-2025-11-25
nsaspy:hotfix/live-tool-native-openrouter
nsaspy:feature/issue-13-chain-runtime
nsaspy:fix/stable-live-repair-gate
nsaspy:fix/live-repair-strategy-parser
nsaspy:feature/issue-12-durable-graph
nsaspy:feature/issue-11-agent-supervision
nsaspy:feature/issue-10-structured-outcomes-repair
nsaspy:feature/issue-9-rlm-completion
nsaspy:feature/issue-8-capability-tools
nsaspy:feature/issue-7-typed-plan-runtime
nsaspy:feature/issue-6-context-store
nsaspy:fix/openrouter-reasoning-response
nsaspy:fix/live-openrouter-smoke-stability
nsaspy:feature/issue-5-openrouter-provider
nsaspy:feature/issue-4-swi-bootstrap
nsaspy:agent/agentic-harness-research
nsaspy:agent/prolog-rlm-foundation
No reviewers
Labels
Clear labels
bug
Something isn't working
documentation
Improvements or additions to documentation
duplicate
This issue or pull request already exists
enhancement
New feature or request
good first issue
Good for newcomers
help wanted
Extra attention is needed
invalid
This doesn't seem right
question
Further information is requested
wontfix
This will not be worked on
No labels
bug
documentation
duplicate
enhancement
good first issue
help wanted
invalid
question
wontfix
Milestone
Clear milestone
No items
No milestone
Projects
Clear projects
No items
No project
Assignees
Clear assignees
No assignees
1 participant
Notifications
Due date
The due date is invalid or out of range. Please use the format "yyyy-mm-dd".
No due date set.
Dependencies
No dependencies set.
Reference
nsaspy/prolog-rlm!380
Loading…
Add table
Add a link
Reference in a new issue
No description provided.
Delete branch "prolog-rlm-v1"
Deleting a branch is permanent. Although the deleted branch may continue to exist for a short time before it actually gets removed, it CANNOT be undone in most cases. Continue?
Summary
all_toolscache profileRuntime invariants
Prolog continues to own capabilities, schema validation, authority, durable effects, budgets, cancellation, context lifetime, accounting, and verification. Provider call IDs are correlation only. Registered operations re-enter the canonical tool/effect runtime, and potentially large results stay behind per-call opaque contexts.
Cost, compiler, and cache policy
Direct mode defaults to
prompt_compile_mode(compiled), matching root completion. The canonical prompt compiler uses the real query, capability set, and context budget to select provider-visible registered tools and skills once per direct session; that exact selected projection is then reused on continuation turns within the session.prompt_compile_mode(all_tools)remains an explicit compatibility/cache profile for workloads where a stable warm inventory is measurably cheaper. Hosts should optimize provider-reported cached tokens, total prompt tokens, and actual cost together rather than maximizing hit percentage alone.The credentialed cache gate deliberately selects the explicit
all_toolsprofile, constructs ten fresh sessions, and requires provider-reported cache hits on at least 80 percent of the nine warm requests. Draft updates skip this paid suite; it runs when the PR becomes ready or is manually dispatched.Review fixes in the consolidated head
Verification
Complete local gate at
0c9c2d509435971ead2a0804dd9730f5229139d2:Exact-head gate at
8eddb58b30cbdcb35283f0bd4cb5c332400bd20e:Exact-head gate at
fac93b782508f47e9a6aa3b0b884549830497f36(typed-plan native model steps):rlm_completion/rlm_direct/rlm_planloads: passPaid evidence and current boundary
Earlier 40k managed-conversation evidence passed locally with two HTTP 200 generations, exact context-search to model to final dataflow, two model calls, 7422 provider-native tokens, and runtime cost 0.007575160 USD.
The automatic paid run on superseded head
f7d93e6is not a passing gate. Its direct native context and registered-tool cases passed with HTTP 200 responses, but the typed-plan root planner lane exhausted three attempts withmissing_field(op)and the 40k managed-conversation lane also failed. The corrected head intentionally has no paid claim while this PR remains draft.First-class typed-plan integration
Typed plans remain a first-class, fully supported execution strategy with typed context, tool, model, recursive, parallel, retry, checkpoint, and final operations. The
modeloperation now runs through the same provider-native session executor as direct mode whenever the runtime supplies the canonicalrlm_direct_model_step/10handler — root completion and direct-modetyped_plan_executealways do — while standaloneplan_run/5without a handler keeps its one-raw-call compatibility profile andllm_query/3stays exactly one raw call. The plan reserves one step and one model call, then charges actual continuation iterations, model calls, tool calls, context operations, and observation bytes atomically against the shared plan budget; the final response and every provider response are recorded soplan_usage/2and execution errors account for all spent calls, tokens, and cost. Model-suppliedmessages,tools,tool_choice, and streaming controls are rejected before provider dispatch; runtime-selected schemas remain authoritative. The native path is canonical, not legacy.Non-goals
No product-specific Auto-Dig or coding-agent runtime, ambient shell/filesystem/network authority, new scheduler, authority engine, effect journal, verifier, persistence layer, or provider-specific core cache database.
Refs #277
Refs #279
The root model may now answer directly through the strict {"mode":"direct","answer":"<nonempty final text>"} envelope and finish in exactly one model call with plan:none, empty transitions and bindings, zero recursion, and truthful usage/trajectory data. A valid typed plan still wins, so context, tool, model, and recursive plan semantics are unchanged. Fail-closed boundary: envelopes with unapproved fields, non-direct modes, or empty/nontext answers fail explicitly as invalid_root_decision; prose, malformed plans, and native provider tool calls are never accepted as either root-decision form. The root prompt, the mandatory rlm-operate skill, and planner repair diagnostics state both forms and prefer direct completion when runtime operations add no value; repair diagnostics remain bounded and never echo rejected provider output. The deep-experiment report now records the actual root decision and claims plan_parsed/plan_validated only when a plan actually executed. The live planner-context acceptance test uses neutral repository fixtures (README and completion/tools runtime docs) instead of SPEC/RAGE workflow records. Fixes the paid harness_guided_depth_2 failure where the forced-planner protocol left no textual final model output, and removes the second model call ordinary questions previously paid just to echo the goal.openrouter_provider/2 now carries app_title('prolog-rlm') and app_referer('https://github.com/lost-rob0t/prolog-rlm'); the OpenAI-compatible transport sends them as X-OpenRouter-Title and HTTP-Referer. Attribution is descriptive identity only: absent keys send no headers, invalid values fail closed as configuration_error before network IO, and downstream provider terms override both. Deterministic coverage uses a local HTTP server capturing the real request headers (5 tests). Adds scripts/openrouter_completion.sh for per-generation debugging and a repo-local debug-io skill. Live verification: generation gen-1787858510-MAIPUiC1ViAP9FPPwWXf shows app_id 4845152 with origin github.com/lost-rob0t/prolog-rlm.Pull request closed