diff --git a/README.md b/README.md index b725dbc0..d683df7e 100644 --- a/README.md +++ b/README.md @@ -201,15 +201,9 @@ checking its launcher and workspace boundary. ## Where this fits -Imp is one of four repositories that together run persistent agents with -AT Protocol accounts. Imp is the library: typed language-model programs, an -MCP client (`Imp.MCP`), and the ACP server side (`Imp.ACP`). Dwell hosts -residents on Imp and owns their capabilities and grants. Kite exposes one -AT Protocol account as MCP tools and delivers its notifications to Dwell. -Haven is the person's app: an ACP client to residents and to any other agent. -`ex_mcp` (our fork) is the one MCP and ACP implementation all four use. - -Dependency direction: Haven → ex_mcp; Kite → ex_mcp; Dwell → Imp → ex_mcp. -Imp is never a service and never depends on the other three. Haven does not -compile against Imp; it launches an Imp program as an external ACP process. -Dwell inherits Imp's ex_mcp revision, so a bump here is a bump for Dwell. +Imp is a library: typed language-model programs, an MCP client (`Imp.MCP`), and +the ACP server side (`Imp.ACP`). It depends on `deepfates/ex_mcp`, a fork of +`ex_mcp`, which is the one MCP and ACP implementation Imp uses. Ordinary Imp +startup opens no protocol endpoint, and Imp is never a service. A host +application owns product lifetimes and decides when to launch an Imp program as +an external ACP process. diff --git a/benchmarks/authorities.json b/benchmarks/authorities.json index 35e6506c..a2f27534 100644 --- a/benchmarks/authorities.json +++ b/benchmarks/authorities.json @@ -2,12 +2,8 @@ "schema_version": 1, "purpose": "Audit ledger of upstream authorities for algorithm and benchmark families named by Imp claim surfaces. This inventory is not proof of conformance, parity, benchmark validity, or effectiveness.", "generated_from": [ - "benchmarks/claims.json", - "docs/internal/UPSTREAM_SURFACE_MAP.md", - "docs/internal/COVERAGE_MATRIX.md", - "docs/internal/PARITY_VALIDATION_PROGRAM.md", - "docs/internal/INSTRUCTION_OPTIMIZER_FIDELITY.md", - "docs/internal/RESEARCH_LANDSCAPE.md" + "docs/differentials/INSTRUCTION_OPTIMIZER_FIDELITY.md", + "docs/differentials/RESEARCH_LANDSCAPE.md" ], "status_vocabulary": { "upstream_repository": ["release_and_commit_pinned", "commit_pinned", "gap", "not_applicable"], @@ -213,7 +209,7 @@ "primary_authority": {"status": "pinned", "locator": "https://arxiv.org/abs/2310.03714v1", "revision": "v1", "title": "DSPy: Compiling Declarative Language Model Calls into Self-Improving Pipelines"}, "upstream_tests": {"status": "present", "references": ["tests/primitives/test_module.py#sha256=05cf969f1cfd7f111d742122b1601b9d5ca9264ec82eb9c891d87b06be1ea603", "tests/primitives/test_example.py#sha256=cfde2da81cf6b85a21a1fc4c54ac4d8accc472c40b0f8c05cdef85f5d9b159e1", "tests/predict/test_predict.py#sha256=fc1dad9b813a0fd742c1459856372816de722eaff2d89512bdcd191f7ecfaf12", "tests/signatures/test_signature.py#sha256=566ea8d2879f68c5c8694076b4f96bd3f2334fd0055084ad8501de8a922d4efc"]}, "dataset_protocol": {"status": "not_applicable", "references": [], "immutable_digests": []}, - "local_differential": {"status": "present", "artifacts": ["test/fixtures/golden_trace/cases.json", "benchmarks/results/golden-trace-parity-*.json"]}, + "local_differential": {"status": "present", "artifacts": ["test/fixtures/golden_trace/cases.json"]}, "notes": "Normalized semantic parity is narrower than byte-identical Python prompts." }, { @@ -228,7 +224,7 @@ "primary_authority": {"status": "no_primary_authority", "locator": null, "revision": null, "title": null}, "upstream_tests": {"status": "present", "references": ["tests/clients/test_lm.py#sha256=f0d72801f06a62dcd1e5d41ba7fc3811875739f45819d0e6e046cd49049716b5", "tests/clients/test_cache.py#sha256=4eb029925ae27ce523b5f5f4cdc9a9c13be11dcc856698e1481b211a272cf3b3", "tests/utils/test_usage_tracker.py#sha256=ec6a52eb03b3898e0f863aa04bcee2f99a9c945b54d018e510c6380e7cb50fd3"]}, "dataset_protocol": {"status": "not_applicable", "references": [], "immutable_digests": []}, - "local_differential": {"status": "partial", "artifacts": ["benchmarks/results/dsex-dspy-parity-campaign-*.json", "benchmarks/results/live-matched-model-matrix-*.json"]}, + "local_differential": {"status": "gap", "artifacts": []}, "notes": "The existing 3.3.0b1 differential identity remains immutable for retained artifacts. The separately pinned DSPy 3.3.1 stable inventory owns the normalized LMRequest/LMResponse/stream semantic disposition; Flex remains explicitly experimental and is mapped through the Optimize Anything opportunity rather than relabeled as parity." }, { @@ -243,7 +239,7 @@ "primary_authority": {"status": "no_primary_authority", "locator": null, "revision": null, "title": null}, "upstream_tests": {"status": "present", "references": ["tests/adapters/test_chat_adapter.py#sha256=5475deb388ea90efafbb963c6a4d4f1f7725f068ee03ee054865cc537bac897b", "tests/adapters/test_json_adapter.py#sha256=ac4d6b7025a0923f45b1cde3ef69200db1a5f5a62b1f72639bcd9966dc29e742", "tests/signatures/test_adapter_image.py#sha256=e3c73e6bccae5575f9ea6da819322f1f3dcfe9d163cbd9ea34b6b949d19eee7b", "tests/adapters/test_audio.py#sha256=46a38bd1996015507d5a61709cf444f9f136a924c21d2f00fd33066e5c0b5647"]}, "dataset_protocol": {"status": "pinned", "references": ["benchmarks/data/multimodal/openai-responses-manifest.json", "six preregistered source-controlled synthetic image/PDF exact-answer rows", "gpt-4.1-mini-2025-04-14 OpenAI Responses typed image/input_file protocol"], "immutable_digests": ["sha256:8a8ff54d7eb8ed47b43ca1bbb3c23f7717b5f0df3c6cdb4b32e7c7174cdd61e9", "sha256:6033f665e8e696f4ae53de6d15e047352a1a2328ab871b55f6d33180b7f6f8ff", "sha256:d6eb339a975408ee49740b1e6e0538be38e80e0dc12185ad4f4fcf78fe7c565e", "sha256:11374319c1f851869f26639f376979ae990c5bd03997687b7fdd5550e4e97f5d", "sha256:d68555405fad87c0451804703bce0f83b202505f1a93cf00175559c2ad471236", "sha256:f90f699ef1afe26daaec324e112867bc64d016c694e6079ac0139161b8fc5321"]}, - "local_differential": {"status": "partial", "artifacts": ["test/fixtures/golden_trace/cases.json", "benchmarks/results/overhead-parity-*.json", "benchmarks/evidence/admitted/multimodal_live/02d3c35797723e6ce4a7c544a3f602579771430276d6838beb14bcfe091e391e.json"]}, + "local_differential": {"status": "partial", "artifacts": ["test/fixtures/golden_trace/cases.json"]}, "notes": "Encoding contracts and the pinned six-row live image/PDF campaign establish the scoped typed-delivery claim only; broad multimodal quality and audio remain unsupported." }, { @@ -257,7 +253,7 @@ "upstream_repository": {"status": "commit_pinned", "repository": "https://github.com/stanfordnlp/dspy", "version": "3.3.0b1", "git_ref": "b2829b7ae3b6e276ac6a8bef66a7ec519dbc923f", "commit": "b2829b7ae3b6e276ac6a8bef66a7ec519dbc923f", "source_paths": ["dspy/adapters/types/tool.py", "dspy/utils/mcp.py", "dspy/predict/react.py", "dspy/predict/react_v2.py", "dspy/predict/code_act.py", "dspy/predict/program_of_thought.py"], "source_manifest": {"path": "benchmarks/authority_sources/dspy-3.3.0b1-b2829b7.json", "sha256": "2ed4fc1ddd75663349aa32d06f9000be6b8211de02ec88908ae8d7cb0c6820af", "file_count": 27}}, "primary_authority": {"status": "no_primary_authority", "locator": null, "revision": null, "title": null}, "upstream_tests": {"status": "present", "references": ["tests/predict/test_react.py#sha256=2880800214b46cd913a65ebec392ce44423d371e1cefdf7a7caa5a927ef38d54", "tests/predict/test_code_act.py#sha256=8da89223fd6a640fd021d01a8e97a376d5e95a97735525a58a1c3e4e07fcc1f3", "tests/predict/test_program_of_thought.py#sha256=4914f0401e38287fe6125b6f028fac9d762f88503c78e08fc06ee334fd6240ba"]}, - "dataset_protocol": {"status": "pinned", "references": ["docs/internal/PARITY_VALIDATION_PROGRAM.md#lane-4-rag-tools-agents-and-production-semantics", "benchmarks/config/bfcl-adapted-differential-v1.json", "test/fixtures/benchmarks/bfcl-adapted-v1.json", "benchmarks/authority_sources/bfcl-protocol-6ea5797.json", "benchmarks/config/rag-tool-failure-differential-v1.json"], "immutable_digests": ["sha256:4074845ebd6852b6e4bceadf1ce69df70630a0e14fcacc27b23d7ed0b153a1b6", "sha256:0e0cafb74a488053c8daad2e9227cc673887af27ead49e8aa14e1359a0e04eb4", "sha256:58cff5c0351fc647ce0a7f752d85894e66b5ffa7ec6cbf6b5f9dc28f27bcb6cf", "sha256:f665b136916893ceb21941604a68bd4d05639b2098120724436162e80d09be7a"]}, + "dataset_protocol": {"status": "pinned", "references": ["benchmarks/config/bfcl-adapted-differential-v1.json", "test/fixtures/benchmarks/bfcl-adapted-v1.json", "benchmarks/authority_sources/bfcl-protocol-6ea5797.json", "benchmarks/config/rag-tool-failure-differential-v1.json"], "immutable_digests": ["sha256:4074845ebd6852b6e4bceadf1ce69df70630a0e14fcacc27b23d7ed0b153a1b6", "sha256:0e0cafb74a488053c8daad2e9227cc673887af27ead49e8aa14e1359a0e04eb4", "sha256:58cff5c0351fc647ce0a7f752d85894e66b5ffa7ec6cbf6b5f9dc28f27bcb6cf", "sha256:f665b136916893ceb21941604a68bd4d05639b2098120724436162e80d09be7a"]}, "local_differential": {"status": "present", "artifacts": ["mix imp.benchmark.bfcl_adapted", "mix imp.benchmark.rag_tool_failure_differential", "test/fixtures/golden_trace/cases.json"]}, "notes": "The family baseline remains DSPy 3.3.0b1. The checked-in CC0 BFCL-shaped fixture is C1/T1 fixture-scorer agreement only. A deliberately narrower C2 compatibility differential separately pins DSPy 3.2.1 because Imp's compared ReAct mode targets that release; it verifies the clean tag/commit and all 296 manifest files before executing an identical six-scenario queued-action failure schedule. This dual revision is explicit and does not replace the family baseline. The differential covers retry observations, injected timeout errors, fixture idempotency, unknown/failing tools, and exact terminal traces, but not model tool selection, retrieval quality, wall-clock timeout behavior, or official BFCL performance." }, @@ -273,7 +269,7 @@ "primary_authority": {"status": "pinned", "locator": "https://arxiv.org/abs/2512.24601v3", "revision": "v3", "title": "Recursive Language Models"}, "upstream_tests": {"status": "present", "references": ["tests/predict/test_rlm.py#sha256=af69b32ea5a2c40d95ca148f52a8e7df028b6592096830d7d0527f5903fc282d"]}, "dataset_protocol": {"status": "partial", "references": ["test/fixtures/rlm_contract_cases.json", "test/fixtures/benchmarks/hotpotqa-small.jsonl"], "immutable_digests": []}, - "local_differential": {"status": "partial", "artifacts": ["benchmarks/results/rlm-contract-*.json", "benchmarks/results/rlm-benchmark-parity-*.json"]}, + "local_differential": {"status": "gap", "artifacts": []}, "notes": "T0 replay and T1 contracts are not T3 long-context effectiveness or paper reproduction." }, { @@ -288,7 +284,7 @@ "primary_authority": {"status": "pinned", "locator": "https://arxiv.org/abs/2312.13382v2", "revision": "v2", "title": "DSPy Assertions: Computational Constraints for Self-Refining Language Model Pipelines"}, "upstream_tests": {"status": "present", "references": ["tests/evaluate/test_evaluate.py#sha256=04a87b5f0efe668a4d7a476afa7cc956a5fd86417b3a1548e3cc934ee5c0858a", "tests/evaluate/test_metrics.py#sha256=0096234f1c1f202dd807de4769dfaf7d2987b678eaa81bba1c2cf4aed60637ef", "tests/evaluate/test_auto_evaluation.py#sha256=2ee33255d3de75dce4dbb41ce2cfe6abd277644903203f8a4370d87b78af1467", "tests/predict/test_refine.py#sha256=13d061e2c5a2650b2c0eaf52db202b55e30382b98885bda528ae0686c010a509"]}, "dataset_protocol": {"status": "not_applicable", "references": [], "immutable_digests": []}, - "local_differential": {"status": "present", "artifacts": ["benchmarks/config/auto-evaluation-differential-v1.json", "benchmarks/evidence/admitted/auto_evaluation_contract/9f7210133747e00ea2f0a187a90babb3ba8d41d558a410735f8185e8013b63c0.json", "test/auto_evaluation_contract_test.exs", "test/auto_evaluation_fidelity_test.exs"]}, + "local_differential": {"status": "present", "artifacts": ["benchmarks/config/auto-evaluation-differential-v1.json", "test/auto_evaluation_contract_test.exs", "test/auto_evaluation_fidelity_test.exs"]}, "notes": "The clean provider-free auto-evaluation differential covers the deterministic DSPy 3.2.1 source contract without claiming judge quality. Matched natural-data quality remains open." }, { @@ -302,8 +298,8 @@ "upstream_repository": {"status": "release_and_commit_pinned", "repository": "https://github.com/stanfordnlp/dspy", "version": "3.2.1", "git_ref": "refs/tags/3.2.1", "commit": "29448ae12756abdd14bd8796c819247ebb83673c", "source_paths": ["dspy/teleprompt/bootstrap.py", "dspy/teleprompt/random_search.py", "dspy/teleprompt/knn_fewshot.py"], "source_manifest": {"path": "benchmarks/authority_sources/dspy-3.2.1-29448ae.json", "sha256": "c12bec921bb00be7a19fbf98c4de4b9e17e2d90405cf9e30fe1ed7d6b3463f3a", "file_count": 296}}, "primary_authority": {"status": "no_primary_authority", "locator": null, "revision": null, "title": null}, "upstream_tests": {"status": "present", "references": ["tests/teleprompt/test_bootstrap.py#sha256=54092d2c983a810ae33932ef507d22a7496f5b69510a0e6f31d1ef54527b4cc2", "tests/teleprompt/test_knn_fewshot.py#sha256=a1e1f25611a17a845ed24cc5e915d8c5ac9a10fe48e74806721ef713cf6b3ee7", "tests/teleprompt/test_random_search.py#sha256=3301e2f1544e5271cf8a31990c6ff61bd138dc85401a4b1bc3722083ff98f058"]}, - "dataset_protocol": {"status": "protocol_defined", "references": ["docs/internal/PARITY_VALIDATION_PROGRAM.md#lane-3-optimizer-lift-parity"], "immutable_digests": []}, - "local_differential": {"status": "partial", "artifacts": ["benchmarks/results/optimizer-lift-parity-*.json"]}, + "dataset_protocol": {"status": "gap", "references": [], "immutable_digests": []}, + "local_differential": {"status": "gap", "artifacts": []}, "notes": "Provider-free lift rows do not establish matched control-flow or T3 effectiveness." }, { @@ -317,8 +313,8 @@ "upstream_repository": {"status": "release_and_commit_pinned", "repository": "https://github.com/stanfordnlp/dspy", "version": "3.2.1", "git_ref": "refs/tags/3.2.1", "commit": "29448ae12756abdd14bd8796c819247ebb83673c", "source_paths": ["dspy/teleprompt/copro_optimizer.py", "dspy/teleprompt/infer_rules.py"], "source_manifest": {"path": "benchmarks/authority_sources/dspy-3.2.1-29448ae.json", "sha256": "c12bec921bb00be7a19fbf98c4de4b9e17e2d90405cf9e30fe1ed7d6b3463f3a", "file_count": 296}}, "primary_authority": {"status": "no_primary_authority", "locator": null, "revision": null, "title": null}, "upstream_tests": {"status": "partial", "references": ["tests/teleprompt/test_copro_optimizer.py#sha256=a9561ce73c056f60ed3ad9fd51c59da27d7acb6d271a16868f94d339eff0db47", "No dedicated upstream tests were found for InferRules or the Imp-native InstructionSearch/SignatureOptimizer surfaces."]}, - "dataset_protocol": {"status": "protocol_defined", "references": ["docs/internal/PARITY_VALIDATION_PROGRAM.md#lane-3-optimizer-lift-parity"], "immutable_digests": []}, - "local_differential": {"status": "partial", "artifacts": ["test/infer_rules_upstream_differential_test.exs", "benchmarks/results/optimizer-lift-parity-*.json"]}, + "dataset_protocol": {"status": "gap", "references": [], "immutable_digests": []}, + "local_differential": {"status": "partial", "artifacts": ["test/infer_rules_upstream_differential_test.exs"]}, "notes": "The provider-free InferRules test executes the exact pinned DSPy 3.2.1 implementation and covers example formatting, rule appending, implicit train/validation splitting, multi-predictor traversal, candidate scoring, the drop-one-example context recovery schedule, and the upstream mutable-signature aliasing defect. Imp deliberately isolates candidates, retains a baseline, records exhausted proposal failures, and reuses its sequential logical-call rollout ID while DSPy draws a new random ID per retry. This is controlled loop evidence rather than whole-optimizer parity or effectiveness. InstructionSearch remains an Imp-native deviation where no stable sidecar equivalent exists." }, { @@ -332,7 +328,7 @@ "upstream_repository": {"status": "commit_pinned", "repository": "https://github.com/stanfordnlp/dspy", "version": "3.3.0b1", "git_ref": "refs/tags/3.3.0b1", "commit": "b2829b7ae3b6e276ac6a8bef66a7ec519dbc923f", "source_paths": ["dspy/teleprompt/mipro_optimizer_v2.py", "dspy/teleprompt/utils.py", "dspy/teleprompt/bootstrap.py", "dspy/propose/grounded_proposer.py"], "source_manifest": {"path": "benchmarks/authority_sources/dspy-3.3.0b1-b2829b7.json", "sha256": "2ed4fc1ddd75663349aa32d06f9000be6b8211de02ec88908ae8d7cb0c6820af", "file_count": 27}}, "primary_authority": {"status": "pinned", "locator": "https://arxiv.org/abs/2406.11695v2", "revision": "v2", "title": "MIPROv2"}, "upstream_tests": {"status": "absent", "references": ["No dedicated MIPROv2 tests in the audited DSPy 3.3.0b1 tree; adjacent bootstrap and proposer tests exist."]}, - "dataset_protocol": {"status": "partial", "references": ["MIPROv2 seven-program benchmark lineage", "docs/internal/PARITY_VALIDATION_PROGRAM.md#lane-3-optimizer-lift-parity"], "immutable_digests": []}, + "dataset_protocol": {"status": "partial", "references": ["MIPROv2 seven-program benchmark lineage"], "immutable_digests": []}, "local_differential": {"status": "partial", "artifacts": ["tmp/instruction-optimizer-contract/instruction-optimizer-contract-*.json"]}, "upstream_source_hashes": ["6bf7632836d3a54ab0da3f38a8f1963813472312e9c0e3f2ff19b4377af407f3", "218c38c25dde75aab9b1d452a15c75687c2e1842d7157dcc6c695f5adbcaf182", "0a588f11f09a358a5306540cc42401d905073c9452e54d32348b13d12bbb1255", "c9900b74c0997410f915f2a470d39dcd9d55c1fa8b9cdf35799915ec0b1617e3"], "notes": "The pinned T1 structural task is implemented but does not count until a fresh artifact passes; exact sampler sequences and T3 effectiveness remain explicit gaps." @@ -348,7 +344,7 @@ "upstream_repository": {"status": "commit_pinned", "repository": "https://github.com/stanfordnlp/dspy", "version": "3.3.0b1", "git_ref": "refs/tags/3.3.0b1", "commit": "b2829b7ae3b6e276ac6a8bef66a7ec519dbc923f", "source_paths": ["dspy/teleprompt/simba.py", "dspy/teleprompt/simba_utils.py"], "source_manifest": {"path": "benchmarks/authority_sources/dspy-3.3.0b1-b2829b7.json", "sha256": "2ed4fc1ddd75663349aa32d06f9000be6b8211de02ec88908ae8d7cb0c6820af", "file_count": 27}}, "primary_authority": {"status": "no_primary_authority", "locator": null, "revision": null, "title": null}, "upstream_tests": {"status": "absent", "references": ["No dedicated SIMBA tests in the audited DSPy 3.3.0b1 tree."]}, - "dataset_protocol": {"status": "protocol_defined", "references": ["docs/internal/PARITY_VALIDATION_PROGRAM.md#lane-3-optimizer-lift-parity"], "immutable_digests": []}, + "dataset_protocol": {"status": "gap", "references": [], "immutable_digests": []}, "local_differential": {"status": "partial", "artifacts": ["tmp/instruction-optimizer-contract/instruction-optimizer-contract-*.json"]}, "upstream_source_hashes": ["4de72e1d0cb1cd30a180569c21973c41fa272c3ebb82a365e3f307986ab67a55", "ed745647ffcfcf4090e5d5b5489cd0b13ebfff1d38a22559563f4f606b31fb2c"], "notes": "No primary SIMBA paper is claimed; released implementation and DSPy documentation are the identified authorities." @@ -364,10 +360,10 @@ "upstream_repository": {"status": "release_and_commit_pinned", "repository": "https://github.com/gepa-ai/gepa", "version": "0.1.4", "git_ref": "refs/tags/v0.1.4", "commit": "8b0ce6cd99a234f6b74daf37558a2ac0ce18f975", "source_paths": ["src/gepa"], "source_manifest": {"path": "benchmarks/authority_sources/gepa-0.1.4-8b0ce6c.json", "sha256": "443ea361efc8c7b94589627d750325617b2174e4cb0988650f070ca45f2d5908", "file_count": 114}}, "primary_authority": {"status": "pinned", "locator": "https://arxiv.org/abs/2507.19457v2", "revision": "v2", "title": "GEPA"}, "upstream_tests": {"status": "partial", "references": ["https://github.com/gepa-ai/gepa/tree/8b0ce6cd99a234f6b74daf37558a2ac0ce18f975/tests", "The historical v0.1.1 provider-free helper fixtures remain executable through scripts/gepa_v011_contract.py."]}, - "dataset_protocol": {"status": "protocol_defined", "references": ["docs/internal/PARITY_VALIDATION_PROGRAM.md#lane-3-optimizer-lift-parity"], "immutable_digests": []}, - "local_differential": {"status": "partial", "artifacts": ["benchmarks/evidence/admitted/gepa_contract/3f188ccdc6e3ad7cd1b9f00f9096e62c3024097d6de654b90364712477ef8cc7.json", "tmp/gepa-v014-contract/gepa-v014-contract-*.json", "benchmarks/results/gepa-replication-*.json", "benchmarks/results/optimizer-lift-parity-*.json"]}, + "dataset_protocol": {"status": "gap", "references": [], "immutable_digests": []}, + "local_differential": {"status": "partial", "artifacts": ["tmp/gepa-v014-contract/gepa-v014-contract-*.json"]}, "upstream_source_hashes": ["ba361b477de74c20eb813b277b0fb85b6898ca534e09c8e878604fb1c8980c53", "5ee9ccfdf31e2d4d1262793c569e44ef7b39659a3e971e4f3dc7d656d69a1d85", "9ad128c981c7344ba0e89d053c2fe33e98a2d74d830679d620cb7cd0d7b1820c", "60aca7024e31a3e273a01187a6329f381f297a77ec7b6add4b9c90b4d64e9b6c", "cd0a3254927e399d0cae4a212076f7577161027b3c4ff19d03c3d2150408ee5a", "248cc6eb125eeddaa98f90b7780db2754ec0444a6143aeb1f97ff5660cf39568", "d33475e411a38353f34272b12b0b2a7af24bbbeca2c2e4fe6c204fa476e87fdb"], - "notes": "GEPA v0.1.4 is the current implementation authority and selected T1 contract. The admitted provider-free differential covers 15 structural cases only; the v0.1.1 receipt remains historical. Operational execution, paper reproduction, effectiveness, full optimizer parity, and exact cross-runtime RNG sequences remain explicit gaps; the full six-family campaign is incomplete." + "notes": "GEPA v0.1.4 is the current implementation authority and selected T1 contract. The admitted provider-free differential covers 15 structural cases only, and its receipt is unpublished history; the v0.1.1 receipt remains historical. Operational execution, paper reproduction, effectiveness, full optimizer parity, and exact cross-runtime RNG sequences remain explicit gaps; the full six-family campaign is incomplete." }, { "id": "family.optimizer_avatar_actor", @@ -413,9 +409,9 @@ "primary_authority": {"status": "pinned", "locator": "https://arxiv.org/abs/2407.10930v2", "revision": "v2", "title": "Fine-Tuning and Prompt Optimization: Two Great Steps that Work Better Together"}, "upstream_tests": {"status": "partial", "references": ["tests/teleprompt/test_bootstrap_finetune.py#sha256=2e96b40f24ec9633139e6a6ed19f0ffc4add95fb82d88779a1b54b705927f4f3"]}, "dataset_protocol": {"status": "partial", "references": ["provider-compatible local training protocol", "benchmarks/data/provider-training-banking77-v1.json"], "immutable_digests": ["sha256:b84958ebf577bc5d57f2c6cf4a6033d7aafcb3a5cf91a79aa826b4345ebb1f3f"]}, - "local_differential": {"status": "partial", "artifacts": ["benchmarks/evidence/admitted/local_mlx/7016478544971aba539f522905ec40f41a29380a1b09291ef7cca91cb7d4567d.json"]}, + "local_differential": {"status": "gap", "artifacts": []}, "upstream_source_hashes": ["d3d3411e8f00d36fc7967290753af66bbe31b01038363e8685427cc58af5963f"], - "notes": "The admitted clean Imp local-MLX artifact proves scoped held-out local weight effectiveness and portable rebinding, not DSPy semantic conformance or paid-provider effectiveness. Imp intentionally corrects DSPy 3.2.1's pred_ind shadowing bug and adds bounded atomic provider lifecycle handling." + "notes": "The admitted clean Imp local-MLX artifact, run in July 2026 and unpublished, proved scoped held-out local weight effectiveness and portable rebinding, not DSPy semantic conformance or paid-provider effectiveness; no artifact is retained in this repository. Imp intentionally corrects DSPy 3.2.1's pred_ind shadowing bug and adds bounded atomic provider lifecycle handling." }, { "id": "family.optimizer_mmgrpo", @@ -493,8 +489,8 @@ "primary_authority": {"status": "pinned", "locator": "https://arxiv.org/abs/2605.19633v1", "revision": "v1", "title": "optimize_anything: A Universal API for Optimizing any Text Parameter"}, "upstream_tests": {"status": "partial", "references": ["https://github.com/gepa-ai/gepa/tree/8b0ce6cd99a234f6b74daf37558a2ac0ce18f975/tests", "https://github.com/gepa-ai/optimize-anything-artifact/tree/58cdf89d856f2fbc174991b89076eccdcf68e4ca"]}, "dataset_protocol": {"status": "protocol_defined", "references": ["https://github.com/gepa-ai/optimize-anything-artifact/tree/58cdf89d856f2fbc174991b89076eccdcf68e4ca", "benchmarks/config/optimize-anything-upstream-differential-v1.json", "SWE-bench/SWE-bench_Verified@91aa3ed51b709be6457e12d00300a6a596d4c6a3"], "immutable_digests": ["sha256:43ed5a3d1d98da36472c1ade65ddd2085d7b4ff694fcaf6a023a07c5c1f32f21"]}, - "local_differential": {"status": "partial", "artifacts": ["benchmarks/evidence/admitted/optimize_anything/080f41578d725c8841d7484f6953cba419626b36f08043408c1027622ede4653.json", "benchmarks/config/optimize-anything-upstream-differential-v1.json", "mix imp.benchmark.optimize_anything_upstream_differential", "mix benchmark.optimize_anything.check", "mix imp.benchmark.optimize_anything --live"]}, - "notes": "Released algorithm authority is GEPA v0.1.4, and the public Optimize Anything artifact is independently pinned at 58cdf89d856f2fbc174991b89076eccdcf68e4ca. The historical v0.1.1 contract remains pinned for exact regression comparison. The admitted nine-run artifact is a scoped Imp-native effectiveness lane, not an upstream comparison. The matched three-domain upstream differential is implemented but remains partial until its hostile-admission findings are repaired and the complete protocol is rerun." + "local_differential": {"status": "partial", "artifacts": ["benchmarks/config/optimize-anything-upstream-differential-v1.json", "mix imp.benchmark.optimize_anything_upstream_differential", "mix benchmark.optimize_anything.check", "mix imp.benchmark.optimize_anything --live"]}, + "notes": "Released algorithm authority is GEPA v0.1.4, and the public Optimize Anything artifact is independently pinned at 58cdf89d856f2fbc174991b89076eccdcf68e4ca. The historical v0.1.1 contract remains pinned for exact regression comparison. The admitted nine-run artifact is a scoped Imp-native effectiveness lane, not an upstream comparison, and the artifact itself is unpublished history. The matched three-domain upstream differential is implemented but remains partial until its hostile-admission findings are repaired and the complete protocol is rerun." }, { "id": "family.retrieval_rag_data", @@ -522,8 +518,8 @@ "upstream_repository": {"status": "release_and_commit_pinned", "repository": "https://github.com/stanfordnlp/dspy", "version": "3.2.1", "git_ref": "refs/tags/3.2.1", "commit": "29448ae12756abdd14bd8796c819247ebb83673c", "source_paths": ["dspy/utils", "dspy/streaming", "dspy/clients/cache.py", "dspy/utils/inspect_history.py", "dspy/utils/callback.py"], "source_manifest": {"path": "benchmarks/authority_sources/dspy-3.2.1-29448ae.json", "sha256": "c12bec921bb00be7a19fbf98c4de4b9e17e2d90405cf9e30fe1ed7d6b3463f3a", "file_count": 296}}, "primary_authority": {"status": "no_primary_authority", "locator": null, "revision": null, "title": null}, "upstream_tests": {"status": "present", "references": ["tests/streaming/test_streaming.py#sha256=89fb9c83a3bb17ed2c210bdbc2cfa4fb60d101818d5453380276e5d577fabd90", "tests/utils/test_asyncify.py#sha256=1ae0eeb96768e50f3a27d770b3caaaa1759b87f0a4492d484fd1e3ce03ab098d", "tests/utils/test_parallelizer.py#sha256=939889a2f561a8615e31cd63f7455c63655e09490501808e979d3c49c728b245"]}, - "dataset_protocol": {"status": "protocol_defined", "references": ["docs/internal/PARITY_VALIDATION_PROGRAM.md#lane-5-provider-free-performance"], "immutable_digests": []}, - "local_differential": {"status": "present", "artifacts": ["benchmarks/results/overhead-parity-*.json", "test/fixtures/golden_trace/cases.json"]}, + "dataset_protocol": {"status": "gap", "references": [], "immutable_digests": []}, + "local_differential": {"status": "present", "artifacts": ["test/fixtures/golden_trace/cases.json"]}, "notes": "Only named provider-free paths may support performance claims." }, { @@ -553,7 +549,7 @@ "primary_authority": {"status": "pinned", "locator": "https://arxiv.org/abs/2110.14168v2", "revision": "v2", "title": "Training Verifiers to Solve Math Word Problems"}, "upstream_tests": {"status": "not_applicable", "references": []}, "dataset_protocol": {"status": "pinned", "references": ["openai/gsm8k@740312add88f781978c0658806c59bc2815b9866 main/test, all 1319 rows", "benchmarks/data/gsm8k-test-0-1319.manifest.json"], "immutable_digests": ["sha256:42b46ba97ea0aa7aac2fa145439ab5c73d535a81f1a417194643bb1ebbae26ab"]}, - "local_differential": {"status": "partial", "artifacts": ["benchmarks/results/dsex-dspy-parity-campaign-*.json"]}, + "local_differential": {"status": "gap", "artifacts": []}, "notes": "The normalized test split is bound independently to its Hugging Face dataset revision and full-file digest; the original repository and paper are pinned separately." }, { @@ -568,7 +564,7 @@ "primary_authority": {"status": "pinned", "locator": "https://arxiv.org/abs/1809.09600v1", "revision": "v1", "title": "HotpotQA: A Dataset for Diverse, Explainable Multi-hop Question Answering"}, "upstream_tests": {"status": "not_applicable", "references": []}, "dataset_protocol": {"status": "pinned", "references": ["hotpotqa/hotpot_qa@1908d6afbbead072334abe2965f91bd2709910ab", "GEPA artifact Random(0) train/dev/test construction", "benchmarks/data/hotpotqa-validation-0-7405.manifest.json"], "immutable_digests": ["sha256:14e5499a2e1bd552a754024d59a8838b12fe65147e35cb18ea93b1b0707b074a", "sha256:bfc87f7d260635544e28e86e884de5cec9c5a9ca190c226fca5c42979402e22b", "sha256:694006eb031b34f4bc00b04ba8430e5fc40a84d4a3f3f958ccdca0d11278994c", "sha256:16b3ec8a4054676352a0cdc4627be1f2a1cd7938c4d67ad6651f94869171e37d"]}, - "local_differential": {"status": "partial", "artifacts": ["benchmarks/results/dsex-dspy-parity-campaign-*.json", "benchmarks/results/gepa-replication-*.json"]}, + "local_differential": {"status": "gap", "artifacts": []}, "notes": "General HotPotQA parity and GEPA HotpotQABench replication are distinct protocols grouped under one dataset lineage." }, { @@ -583,7 +579,7 @@ "primary_authority": {"status": "no_primary_authority", "locator": null, "revision": null, "title": null}, "upstream_tests": {"status": "not_applicable", "references": []}, "dataset_protocol": {"status": "partial", "references": ["Color/classification smoke", "local classification and instruction-following rows"], "immutable_digests": []}, - "local_differential": {"status": "partial", "artifacts": ["benchmarks/results/dsex-dspy-parity-campaign-*.json", "benchmarks/results/optimizer-lift-parity-*.json"]}, + "local_differential": {"status": "gap", "artifacts": []}, "notes": "This is an explicitly Imp-owned smoke fixture, not an upstream benchmark or scientific-effectiveness authority; it remains partial until its generated rows are materialized and hashed." }, { diff --git a/benchmarks/config/local-mlx-lifecycle-banking77-v2-deployment-continuation.json b/benchmarks/config/local-mlx-lifecycle-banking77-v2-deployment-continuation.json index 0e726d57..e62a6bfc 100644 --- a/benchmarks/config/local-mlx-lifecycle-banking77-v2-deployment-continuation.json +++ b/benchmarks/config/local-mlx-lifecycle-banking77-v2-deployment-continuation.json @@ -3,6 +3,7 @@ "campaign_id": "local-mlx-sft-lifecycle-banking77-v2-deployment-continuation", "status": "draft_for_principal_review_not_authorized", "output_artifact": "benchmarks/results/local-mlx-lifecycle-banking77-v2-deployment-continuation.json", + "note": "Paths under benchmarks/results/ are run write targets; that directory is not tracked, so any predecessor artifact named here is unpublished history recorded for provenance, not a file in this repository.", "purpose": "Complete only the deployment, base-versus-fused evaluation, persistence, fresh-process reload, identity, and cleanup measurements that stopped V2 left unobserved. This draft does not authorize a server launch.", "closed_predecessor": { "campaign_id": "local-mlx-sft-lifecycle-banking77-v2", diff --git a/benchmarks/config/local-mlx-lifecycle-banking77-v2.json b/benchmarks/config/local-mlx-lifecycle-banking77-v2.json index cea7fd72..b077198e 100644 --- a/benchmarks/config/local-mlx-lifecycle-banking77-v2.json +++ b/benchmarks/config/local-mlx-lifecycle-banking77-v2.json @@ -3,6 +3,7 @@ "campaign_id": "local-mlx-sft-lifecycle-banking77-v2", "status": "draft_for_principal_review_not_authorized", "output_artifact": "benchmarks/results/local-mlx-lifecycle-banking77-v2.json", + "note": "Paths under benchmarks/results/ are run write targets; that directory is not tracked, so any predecessor artifact named here is unpublished history recorded for provenance, not a file in this repository.", "purpose": "Run the already-approved one-model Banking77 base-versus-fused lifecycle probe with the JSONL that current public Imp actually renders. This is a draft only: it does not authorize training, inference, or a server.", "closed_predecessor": { "campaign_id": "local-mlx-sft-lifecycle-banking77-v1", diff --git a/benchmarks/config/support-ticket-lift-openrouter-free-v2.json b/benchmarks/config/support-ticket-lift-openrouter-free-v2.json index 43d55e29..652addf0 100644 --- a/benchmarks/config/support-ticket-lift-openrouter-free-v2.json +++ b/benchmarks/config/support-ticket-lift-openrouter-free-v2.json @@ -4,6 +4,7 @@ "status": "preregistered_not_run", "authorization": "not_launched", "output_artifact": "benchmarks/results/support-ticket-lift-openrouter-free-v2-20260725.json", + "note": "Paths under benchmarks/results/ are run write targets; that directory is not tracked, so any predecessor artifact named here is unpublished history recorded for provenance, not a file in this repository.", "purpose": "Repeat the stopped v1 preflight with enough bounded output room to remove the observed 64-token ceiling ambiguity. This is a measurement repair, not score-directed tuning and not C3 optimizer-effectiveness evidence.", "predecessor": { "campaign": "support-ticket-baseline-vs-labeled-few-shot", diff --git a/benchmarks/config/support-ticket-lift-openrouter-free-v3.json b/benchmarks/config/support-ticket-lift-openrouter-free-v3.json index 05f01a48..f2f5e77c 100644 --- a/benchmarks/config/support-ticket-lift-openrouter-free-v3.json +++ b/benchmarks/config/support-ticket-lift-openrouter-free-v3.json @@ -4,6 +4,7 @@ "status": "preregistered_not_run", "authorization": "not_launched", "output_artifact": "benchmarks/results/support-ticket-lift-openrouter-free-v3-20260725.json", + "note": "Paths under benchmarks/results/ are run write targets; that directory is not tracked, so any predecessor artifact named here is unpublished history recorded for provenance, not a file in this repository.", "purpose": "Measure the already-frozen support-ticket baseline versus LabeledFewShot design on the one exact-free route selected by an independent synthetic typed-format canary. This preregistration is based only on schema completion and guarded route operation, not benchmark scores, and does not authorize launch.", "closed_predecessors": [ { diff --git a/decisions.md b/decisions.md index 92b4460b..1c290a13 100644 --- a/decisions.md +++ b/decisions.md @@ -14,11 +14,11 @@ not necessarily when it was made. | 2026-09-13 | An ACP tool kind is derived from the MCP `ToolAnnotations` the server declares, not from a host table keyed by tool name; a tool that declares no hint gets `nil` rather than a guess, and an explicit `:tool_kinds` entry still wins. | `lib/imp/acp/tool_kind.ex` moduledoc, `test/acp_options_tool_kinds_test.exs`. A name table goes stale the moment a server publishes a tool it does not name, and that is exactly when a permission mode that would have asked does not ask. | In force. | Does not retire. | | 2026-09-13 | On `session/load` and `session/resume` the session is installed with the `_meta` it was created with, not the `_meta` the request carried; both are passed to the program factory as `:meta` and `:requested_meta`. | `lib/imp/acp.ex` moduledoc, `test/acp_imp_acp_test.exs`. A session's history and transcript belong to the configuration that produced them, so a resume that silently adopts a different one replays one configuration's transcript as another's; only the factory knows which of its own `_meta` keys are identity-bearing, so it is given both and decides whether to refuse. | In force. | Does not retire. | | 2026-09-13 | `Imp.ACP` (the ACP session/program adapter) is part of Imp; the separate `imp_acp` package is retired with no compatibility shim and no second implementation. | Retired `imp_acp` README, `AGENTS.md`. One protocol stack, one owner of the ExMCP dependency. | In force, and released in `v0.4.0`. | The "unreleased" caveat retired at `v0.4.0`. The absorption itself does not retire. | -| 2026-09-13 | Absorption renames: `IMP_ACP_PATH` becomes `IMP_PATH` (copied workspace example), `IMP_ACP_ROOT` becomes `IMP_ROOT` (Haven preset), launcher is `examples/workspace_agent/scripts/workspace-agent-acp`. Module names `Imp.ACP`, `Imp.ACP.Host`, `Imp.ACP.MCP` and the `deepfates.com/imp-acp` wire metadata namespace keep their names. | Retired `imp_acp` README. Saved executable and environment references are updated explicitly, not by rewriting contact history. | In force. | Does not retire. | +| 2026-09-13 | Absorption renames: `IMP_ACP_PATH` becomes `IMP_PATH` (copied workspace example), `IMP_ACP_ROOT` becomes `IMP_ROOT` (host application preset), launcher is `examples/workspace_agent/scripts/workspace-agent-acp`. Module names `Imp.ACP`, `Imp.ACP.Host`, `Imp.ACP.MCP` and the `deepfates.com/imp-acp` wire metadata namespace keep their names. | Retired `imp_acp` README. Saved executable and environment references are updated explicitly, not by rewriting contact history. | In force. | Does not retire. | | 2026-09-13 | The workspace agent's session store keeps the `imp_acp/workspace_agent/sessions` directory name. | `examples/workspace_agent/README.md`. Preserves saved sessions; it is a storage location, not a dependency on the retired package. | In force. | Saved sessions are migrated, or the owner accepts losing restart continuity for them. | | 2026-09-13 | Ordinary Imp boot starts no protocol listener, subprocess or remote connection and does not start the ExMCP application. Protocol entry points (`Imp.ACP.*`, non-empty `Imp.MCP.connect/2`) start it explicitly. Releases using them declare `applications: [ex_mcp: :load]`. | `mix.exs` dependency comment, `docs/PRODUCTION_OPERATIONS.md` "Protocol runtime in releases", commit `728f8c77`. Prediction and optimizer processes must not open listeners or acquire protocol boot output. | In force. | Does not retire while Imp is a library inside a host application. | -| 2026-09-13 (pin dated 2026-09-15) | ExMCP is the `deepfates/ex_mcp` fork at `7285330b490476cc153dd60fb9adac9cd39d4a94`, the one ref shared by kite, haven, dwell and imp. Do not move it independently of the other consumers. | `mix.exs` `ex_mcp_dependency/0`; the fork's `FORK.md` records each patch (byte-safe stdio frames, owner-bound subprocess cleanup, ACP delivery barriers, per-connection HTTP trust) with its failure. Workshop standing decision "one ref, not four". | In force. | Each patch is upstreamed or made unnecessary and the dependency becomes a released Hex version; until then the ref moves for all four repositories in one coordinated change. | -| 2026-09-13 | `mix.exs` resolves ExMCP three ways: a bundled `vendor/ex_mcp` if present, else `EX_MCP_PATH`, else the GitHub pin. | `mix.exs`. No reason is recorded beside the conditional; the bundled path is presumably for the source package and the env override for fork development (*unverified*, inferred from `test/acp_imp_acp_test.exs` and the workspace agent test passing `EX_MCP_PATH` through). | In force, reason unrecorded. Conflicts with the workshop rule that a new path deletes the old one; owner to rule. | The owner decides whether two of the three paths go, or records why all three stay. | +| 2026-09-13 (pin dated 2026-09-15) | ExMCP is the `deepfates/ex_mcp` fork at `7285330b490476cc153dd60fb9adac9cd39d4a94`, the one ref shared with the other consumers of the same fork. Do not move it independently of them. | `mix.exs` `ex_mcp_dependency/0`; the fork's `FORK.md` records each patch (byte-safe stdio frames, owner-bound subprocess cleanup, ACP delivery barriers, per-connection HTTP trust) with its failure. Owner ruling, 2026-09-11: one ref, not one per consumer. | In force. | Each patch is upstreamed or made unnecessary and the dependency becomes a released Hex version; until then the ref moves for every consumer of the fork in one coordinated change. | +| 2026-09-13 | `mix.exs` resolves ExMCP three ways: a bundled `vendor/ex_mcp` if present, else `EX_MCP_PATH`, else the GitHub pin. | `mix.exs`. No reason is recorded beside the conditional; the bundled path is presumably for the source package and the env override for fork development (*unverified*, inferred from `test/acp_imp_acp_test.exs` and the workspace agent test passing `EX_MCP_PATH` through). | In force, reason unrecorded. Conflicts with the standing rule (2026-09-11) that a new path deletes the old one; owner to rule. | The owner decides whether two of the three paths go, or records why all three stay. | | 2026-09-13 | Known interoperability limit, not a feature: when an HTTP MCP server selects protocol version `2025-03-26`, the pinned fork's `notifications/initialized` can still carry the client's `2025-11-25` header, and a strict older server may reject the session. Compatibility with strict older HTTP servers is not claimed. | `docs/PRODUCTION_OPERATIONS.md`, commit `902a5546`. Inherited from the fork's connection manager. | Open defect, documented. | The fork settles the HTTP version before sending that notification and a test against a strict older server passes. | | 2026-09-13 | Imported MCP tool calls disable generic transport retries, use ExMCP's `:safe_only` broken-stream policy, and never repeat an ambiguous write. The retired Imp-specific retry, backoff and Retry-After options are rejected, not ignored, and are not reimplemented above ExMCP. | `docs/PRODUCTION_OPERATIONS.md`. A timeout does not say whether the call ran; replay needs a real server idempotency contract, which is the application's to establish. | In force. | Does not retire. | | 2026-09-13 | Tool errors return `{:error, {:mcp_tool_error, envelope}}` with the original content, codes and operation identifiers, so refusal, authorization refusal and indeterminate effect stay distinguishable. The old text-only error tuple is gone. | `docs/PRODUCTION_OPERATIONS.md`. A boundary declares its failure classes. | In force. | Does not retire. | @@ -34,8 +34,8 @@ not necessarily when it was made. | 2026-07-17 | Evidence-infrastructure tests (`:evidence_infrastructure`) are excluded from the default `mix test`. | `CONTRIBUTING.md`. They need full git history, pinned DSPy Python environments and sometimes provider credentials, none of which a fresh clone has. | In force. | Does not retire. | | 2026-07-10 | ReqLLM is the provider transport boundary; upstream algorithm names keep upstream semantics; deliberate Elixir-native alternatives get a distinct contract and rationale; fixtures and symbol presence never become parity claims. | `CONTRIBUTING.md` "Design Standard". | In force. | Does not retire. | | 2026-07-07 | `mix` aliases (`check`, `package.check`, ...) exist only in a source checkout (detected by `test/package_contract_test.exs` being present); the shipped package carries runtime sources only. | `mix.exs` `aliases/0` and its comment (detection entered in commit `c193433f`). Local benchmark and evidence control files stay out of the consumer's dependency tree. | In force. | Does not retire. | -| 2026-08-22 | `CONTRIBUTING.md` says "Tickets record unfinished work, not product truth", and `.tickets/` holds 57 tracked files. | `CONTRIBUTING.md`; `git ls-files .tickets`. | **Conflicts** with the workshop standing decision (2026-09-11) that no ticket file, log or diary belongs in a repository. Not acted on here; owner to rule. | The owner deletes `.tickets/` (and the sentence) or reopens the workshop decision. | -| 2026-09-11 (workshop) | No OAuth for now, neither MCP's nor ATProto's. | Workshop `AGENTS.md`. Imp carries `Imp.ACP.DemoMCPOAuthPlug` and a demo MCP HTTP server mix task used only by its own tests to exercise ExMCP's PKCE client; ordinary use has no OAuth path. | In force. | A multi-tenant reason appears; that is Haven's question, not Imp's. | -| 2026-09-11 (workshop) | Nothing leaves these repositories: no issues, pull requests or messages to upstream maintainers (`azmaveth/ex_mcp` included). Fixes live in the fork with their reason beside them. | Workshop `AGENTS.md`, the fork's `FORK.md`. | In force. | The owner says otherwise. | -| 2026-09-11 (workshop) | Tests falsify: revert the change and watch the named test fail before believing it. The owner's machine is a consumer, not a privileged insider: whatever a stranger must do to run this, the owner does the same. | Workshop `AGENTS.md`. The `stranger` CI job runs `deps.get` and `mix check` from a cold checkout for the same reason. | In force. | Does not retire. | -| 2026-09-11 (workshop) | Work is a pull request on a topic branch; working notes go in the pull request or nowhere. Adopting a new path deletes the old one in the same change. | Workshop `AGENTS.md`. | In force. | Does not retire. | +| 2026-08-22 | No ticket file, log or diary belongs in this repository; unfinished work is a pull request on a topic branch. | `CONTRIBUTING.md` "Unfinished work is a pull request on a topic branch; there is no ticket file in this repository." An earlier `.tickets/` directory and the `CONTRIBUTING.md` sentence that justified it were both removed. | Resolved: `git ls-files .tickets` is empty. | The owner reopens it. | +| 2026-09-11 | No OAuth for now, neither MCP's nor ATProto's. | Owner ruling, 2026-09-11. Imp carries `Imp.ACP.DemoMCPOAuthPlug` and a demo MCP HTTP server mix task used only by its own tests to exercise ExMCP's PKCE client; ordinary use has no OAuth path. | In force. | A multi-tenant reason appears; that is a host application's question, not Imp's. | +| 2026-09-11 | Outside contributions are welcome through GitHub issues and pull requests on this repository. Patches to the `deepfates/ex_mcp` fork are upstreamed to `azmaveth/ex_mcp` only by the owner; until then a fix lives in the fork with its reason beside it. | Owner ruling, 2026-09-11, amended 2026-09-17 when the repository went public; the fork's `FORK.md` records each patch with its failure. | In force. | The owner says otherwise. | +| 2026-09-11 | Tests falsify: revert the change and watch the named test fail before believing it. The owner's machine is a consumer, not a privileged insider: whatever a stranger must do to run this, the owner does the same. | Owner ruling, 2026-09-11. The `stranger.check` CI job in `.github/workflows/ci.yml` runs `deps.get` and `mix check` from a cold checkout for the same reason. | In force. | Does not retire. | +| 2026-09-11 | Work is a pull request on a topic branch; working notes go in the pull request or nowhere. Adopting a new path deletes the old one in the same change. | Owner ruling, 2026-09-11; `CONTRIBUTING.md` "Unfinished work is a pull request on a topic branch". | In force. | Does not retire. | diff --git a/docs/differentials/COMBEE_FIDELITY.md b/docs/differentials/COMBEE_FIDELITY.md index 2e8ea2b4..85ac6b65 100644 --- a/docs/differentials/COMBEE_FIDELITY.md +++ b/docs/differentials/COMBEE_FIDELITY.md @@ -217,10 +217,9 @@ on 2026-07-13 retained 8/8, 4/8, and 6/8 records respectively with 4, 1, and 3 calls. Live mode is gated by `COMBEE_PREFLIGHT_MODE=live` and -`COMBEE_LIVE_PROVIDER=1`. The bounded run used pinned -`openai:gpt-4.1-mini-2025-04-14` and is stored at -`benchmarks/results/gepa-combee-preflight-live-20260713T231824Z.json`. All arms -retained 8/8 correct answers. Naive large-batch took 2.62 seconds, 1 call, 755 +`COMBEE_LIVE_PROVIDER=1`. The bounded run on 2026-07-13 used pinned +`openai:gpt-4.1-mini-2025-04-14`; its result JSON is unpublished history and is +not retained in this repository. All arms retained 8/8 correct answers. Naive large-batch took 2.62 seconds, 1 call, 755 tokens, and $0.000450; ComBee took 4.25 seconds, 3 calls, 1,909 tokens, and $0.001228; small-batch took 6.75 seconds, 4 calls, 1,266 tokens, and $0.000841. Usage and cost came from ReqLLM telemetry. diff --git a/docs/differentials/CONFIDENCE_CALIBRATION.md b/docs/differentials/CONFIDENCE_CALIBRATION.md index da48041d..26b52c84 100644 --- a/docs/differentials/CONFIDENCE_CALIBRATION.md +++ b/docs/differentials/CONFIDENCE_CALIBRATION.md @@ -100,8 +100,8 @@ evidence gates and remains negative. ## July 13 Evidence -`benchmarks/results/confidence-calibration-live-20260713T225422Z.json` is the -fresh authoritative run. All gates passed: +The 2026-07-13 live run was the fresh authoritative run; its result JSON is +unpublished history and is not retained in this repository. All gates passed: - calibration outcomes: 142 correct, 58 incorrect; - calibration occupancy: 7 occupied bins, 6 supported bins; @@ -121,10 +121,9 @@ those source sets are disjoint rather than paired. The earlier July 13 support-routing fixture was all correct and used a one-bin mapping. Its zero post-calibration Brier score and ECE are not evidence of -learned calibration. The retained -`benchmarks/results/confidence-calibration-assessment-20260713.json` records the -rejection and source-run digests; it is superseded as calibration evidence, not -rewritten as a successful result. +learned calibration. The 2026-07-13 assessment that recorded the rejection and +the source-run digests is unpublished history; that observation is superseded as +calibration evidence, not rewritten as a successful result. This remains a narrow operational probe. It does not establish calibration on other datasets, prompts, providers, model versions, or deployment populations. diff --git a/docs/differentials/MULTIMODAL_FIDELITY.md b/docs/differentials/MULTIMODAL_FIDELITY.md index 2345ea3c..18024342 100644 --- a/docs/differentials/MULTIMODAL_FIDELITY.md +++ b/docs/differentials/MULTIMODAL_FIDELITY.md @@ -25,9 +25,10 @@ It records six dispatches in this run, zero resumed rows, six durable rows, and All rows have distinct ReqLLM request IDs, OpenAI request IDs, and OpenAI Responses IDs. The claim gate has no rejections. -`benchmarks/results/multimodal-quality-live-20260713T215119Z.json` is -invalidated and removed. Its checkpoint schema and pre-dispatch shape evidence -did not exclude forged minimal rows, so its result must not be used as proof. +The 2026-07-13 multimodal quality live run is invalidated and removed, and its +result JSON is unpublished history. Its checkpoint schema and pre-dispatch shape +evidence did not exclude forged minimal rows, so its result must not be used as +proof. ## Manifest Contract diff --git a/docs/differentials/PLAYBOOK_OPTIMIZER.md b/docs/differentials/PLAYBOOK_OPTIMIZER.md index 7a7dbfaa..972a191e 100644 --- a/docs/differentials/PLAYBOOK_OPTIMIZER.md +++ b/docs/differentials/PLAYBOOK_OPTIMIZER.md @@ -87,10 +87,10 @@ without converting a failed lift into a success claim. ### Release Evidence -The committed seed-61 campaign in -`benchmarks/results/playbook-equation-balancer-live-v11.json` binds to commit -`37f58dc6e3110606b5b08d01124b2cfb1ef52566` and records SHA-256 identities for -all implementation sources. Its source/group-disjoint results were: +The seed-61 campaign bound to commit +`37f58dc6e3110606b5b08d01124b2cfb1ef52566` and recorded SHA-256 identities for +all implementation sources. Its result JSON is unpublished history and is not +retained in this repository. Its source/group-disjoint results were: | Split | Baseline | Candidate | Solver calls | |---|---:|---:|---:| diff --git a/docs/differentials/RLM_FIDELITY.md b/docs/differentials/RLM_FIDELITY.md index 2552b0ab..52c4e4a3 100644 --- a/docs/differentials/RLM_FIDELITY.md +++ b/docs/differentials/RLM_FIDELITY.md @@ -252,10 +252,9 @@ explicit campaign identity/version; no ambiguous intent was silently reused. The resolved `v7` plan selected the same single frozen row, direct, `simple_retrieval`, and RLM on both runtimes: exactly six jobs and zero provider -calls during `--plan`. The completed artifact is -`benchmarks/results/rlm-preflight/rlm-benchmark-parity-20260713T233641Z.json`; -its manifest, checkpoint, and audit use the `v7` campaign identity in the same -directory. +calls during `--plan`. The campaign completed on 2026-07-13; its artifact, +manifest, checkpoint, and audit all carried the `v7` campaign identity and are +unpublished history, not retained in this repository. | Runtime | Approach | Score | Calls | Input | Output | USD | Latency ms | Cost authority | | --- | --- | ---: | ---: | ---: | ---: | ---: | ---: | --- | @@ -310,8 +309,8 @@ version-pinned Python dependencies, both setup-script hashes, a complete DSPy source-tree verifier, and a dated pricing authority. This remains one-row end-to-end and output-contract evidence only, not aggregate quality, reliability, semantic superiority, paper parity, or T3 evidence. The exact -command and limitations are recorded in -`benchmarks/results/rlm-anthropic-pilot/README.md`; the artifact SHA-256 is +command and limitations were recorded beside the pilot artifact, which is +unpublished history; the artifact SHA-256 is `b6b9757696c424568633cd80f84f626fb2093db51176b8d79e279b477ddfcaff`. ## Mechanical T3 Gate diff --git a/examples/matched_instruction_optimizers_trec/README.md b/examples/matched_instruction_optimizers_trec/README.md index 619ef943..15184c4b 100644 --- a/examples/matched_instruction_optimizers_trec/README.md +++ b/examples/matched_instruction_optimizers_trec/README.md @@ -20,7 +20,7 @@ messages, models, splits, budgets, or parsing after results. ## Completed outcome The sealed treatment completed on 2026-07-27 under the owner's `$100` aggregate -workshop provider ceiling. Imp GEPA improved its mean untouched accuracy over +provider spend ceiling. Imp GEPA improved its mean untouched accuracy over its own baseline by `+0.4000` (source-ID-clustered 95% interval `[0.2958, 0.5042]`, Holm-adjusted `p = 0.00020`). Its mean difference from pinned DSPy GEPA was `-0.0083`, with 95% interval `[-0.0458, 0.0292]`, clearing @@ -29,13 +29,12 @@ own baseline by `+0.1458`, with interval `[0.0458, 0.2458]` and Holm-adjusted `p = 0.00270`. The frozen headline therefore passed with GEPA as the declared winner. -The committed compact result is -`benchmarks/evidence/archive/matched_experiments/trec/matched-instruction-optimizers-trec-20260726.json`. -It binds -the full retained Imp, upstream, and aggregate artifacts by SHA-256; those raw -artifacts remain local because they contain 181 MB of per-call evidence. +The sealed campaign artifact, `matched-instruction-optimizers-trec-20260726.json`, +is unpublished history: neither it nor the full retained Imp, upstream, and +aggregate artifacts it binds by SHA-256 are in this repository, because those +raw artifacts hold 181 MB of per-call evidence. The treatment used 6,491 calls and `$3.13862325` in provider-reported cost. -Adding the conservative pre-treatment workshop bound yields at most +Adding the conservative pre-treatment bound yields at most `$6.221975` against the `$100` ceiling. The compact scored-row inputs are committed separately from the private raw @@ -47,9 +46,9 @@ noninferiority decision with one provider-free command: mix run --no-start \ examples/matched_instruction_optimizers_trec/recompute_compact.exs -- \ examples/matched_instruction_optimizers_trec/contract.json \ - benchmarks/evidence/archive/matched_experiments/trec/imp-scored-rows.json \ - benchmarks/evidence/archive/matched_experiments/trec/upstream-scored-rows.json \ - benchmarks/evidence/archive/matched_experiments/trec/aggregate-recomputed.json + examples/matched_instruction_optimizers_trec/data/imp-scored-rows.json \ + examples/matched_instruction_optimizers_trec/data/upstream-scored-rows.json \ + examples/matched_instruction_optimizers_trec/data/aggregate-recomputed.json ``` This verifies the statistics asserted by the compact rows. It does not @@ -70,7 +69,7 @@ task and 48 optimizer transports after selection and untouched evaluation. Across both runtimes and three seeds, the revised outer safety envelope is 7,140 task calls plus 342 optimizer calls. Its conservative reservation is -$59.10912. With the current conservative workshop aggregate of $3.08335175, +$59.10912. With the current conservative spend aggregate of $3.08335175, the combined worst case is $62.19247175 and fits the owner's `$100` ceiling. The semantic stopping rule is unchanged; the outer cap is operational only, and firing it makes the treatment inconclusive rather than scoring a truncated diff --git a/examples/matched_instruction_optimizers_trec/contract.exs b/examples/matched_instruction_optimizers_trec/contract.exs index 5a7fba49..1ae48018 100644 --- a/examples/matched_instruction_optimizers_trec/contract.exs +++ b/examples/matched_instruction_optimizers_trec/contract.exs @@ -54,7 +54,7 @@ defmodule MatchedInstructionOptimizersTREC.Contract do require!( manifest["launch_status"] in [ "blocked_pending_gepa_execution_and_fail_closed_preflight", - "blocked_workshop_spend_ceiling_after_gepa_legal_envelope", + "blocked_spend_ceiling_after_gepa_legal_envelope", "sealed" ], "launch_status drift" diff --git a/examples/matched_instruction_optimizers_trec/paired_coordinator_test.py b/examples/matched_instruction_optimizers_trec/paired_coordinator_test.py index df52800c..676605c4 100644 --- a/examples/matched_instruction_optimizers_trec/paired_coordinator_test.py +++ b/examples/matched_instruction_optimizers_trec/paired_coordinator_test.py @@ -27,17 +27,17 @@ def test_preflight_subprocesses_never_receive_provider_authority(self) -> None: with mock.patch.dict(os.environ, {"OPENROUTER_API_KEY": "secret"}): self.assertNotIn("OPENROUTER_API_KEY", paired.preflight_environment()) - def test_revised_legal_envelope_fits_the_authorized_workshop_spend_ceiling(self) -> None: + def test_revised_legal_envelope_fits_the_authorized_spend_ceiling(self) -> None: manifest = json.loads((HERE / "contract.json").read_text()) self.assertEqual(paired.worst_case_usd(manifest), paired.Decimal("59.10912000")) - self.assertEqual(paired.WORKSHOP_CEILING, paired.Decimal("100.00")) + self.assertEqual(paired.SPEND_CEILING, paired.Decimal("100.00")) self.assertEqual( paired.PRIOR_SPEND_BOUND + paired.worst_case_usd(manifest), paired.Decimal("62.19247175"), ) self.assertLessEqual( paired.PRIOR_SPEND_BOUND + paired.worst_case_usd(manifest), - paired.WORKSHOP_CEILING, + paired.SPEND_CEILING, ) def test_graceful_stop_signal_gets_a_bounded_rescue_window(self) -> None: diff --git a/examples/matched_instruction_optimizers_trec/run_paired.py b/examples/matched_instruction_optimizers_trec/run_paired.py index e5c63a02..ad566d14 100644 --- a/examples/matched_instruction_optimizers_trec/run_paired.py +++ b/examples/matched_instruction_optimizers_trec/run_paired.py @@ -25,7 +25,7 @@ GEPA_ROOT = ROOT / "tmp" / "gepa-v0.1.4" UPSTREAM_PYTHON = ROOT / "tmp" / "dspy-parity-venv" / "bin" / "python" PRIOR_SPEND_BOUND = Decimal("3.08335175") -WORKSHOP_CEILING = Decimal("100.00") +SPEND_CEILING = Decimal("100.00") PREFLIGHT_PREFIX = "PAIRED_PREFLIGHT_JSON=" @@ -195,7 +195,7 @@ def preflight() -> dict[str, Any]: maximum = worst_case_usd(manifest) require(maximum == Decimal("59.10912000"), f"sealed maximum spend drift: {maximum}") - require(PRIOR_SPEND_BOUND + maximum <= WORKSHOP_CEILING, "workshop spend ceiling would be exceeded") + require(PRIOR_SPEND_BOUND + maximum <= SPEND_CEILING, "spend ceiling would be exceeded") imp = preflight_imp(manifest_sha) upstream = preflight_upstream(manifest) @@ -223,7 +223,7 @@ def preflight() -> dict[str, Any]: "source_commit": git(ROOT, "rev-parse", "HEAD"), "manifest_sha256": manifest_sha, "prior_spend_bound": str(PRIOR_SPEND_BOUND), - "workshop_ceiling": str(WORKSHOP_CEILING), + "spend_ceiling": str(SPEND_CEILING), "treatment_maximum": str(maximum), "combined_maximum": str(PRIOR_SPEND_BOUND + maximum), "imp": imp, diff --git a/lib/imp/acp/local.ex b/lib/imp/acp/local.ex index 65296036..739c7e69 100644 --- a/lib/imp/acp/local.ex +++ b/lib/imp/acp/local.ex @@ -2,7 +2,7 @@ defmodule Imp.ACP.Local do @moduledoc """ Local ACP attachment to a long-running application over a private UNIX socket. - Each accepted connection gets its own ordinary `Imp.ACP` adapter. A resident + Each accepted connection gets its own ordinary `Imp.ACP` adapter. A host application supplies a program factory that attaches to its own runtime; closing the connection closes the adapter, not that independently owned runtime. `relay/2` exposes this socket as standard ACP stdio to an ordinary client. diff --git a/lib/imp/mcp/oauth.ex b/lib/imp/mcp/oauth.ex index f93729d8..06704bac 100644 --- a/lib/imp/mcp/oauth.ex +++ b/lib/imp/mcp/oauth.ex @@ -2,8 +2,9 @@ defmodule Imp.MCP.OAuth do @moduledoc """ Browser-authorized OAuth credentials for remote HTTP MCP servers. - A host that runs Imp programs — a Dwell resident, Haven, a script — can - declare an MCP server whose Authorization header it does not know yet. A + A host that runs Imp programs — a long-running application, a desktop + client, a script — can declare an MCP server whose Authorization header it + does not know yet. A person authorizes once in a browser on the machine that runs the host, the resulting grant is written to disk encrypted, and a short-lived `Authorization` header is materialized only when a connection is built. diff --git a/mix.exs b/mix.exs index a1bef226..76abf6b4 100644 --- a/mix.exs +++ b/mix.exs @@ -122,7 +122,7 @@ defmodule Imp.MixProject do ] end - # Shared constellation reference. This fork carries byte-safe stdio, + # Shared fork reference. This fork carries byte-safe stdio, # caller-owned request/subprocess cleanup, ACP delivery barriers, and # per-connection HTTP trust propagation. See its FORK.md for each failure # and retirement condition; do not move this ref independently of consumers. diff --git a/test/mcp_oauth_test.exs b/test/mcp_oauth_test.exs index 351f2c84..74b6c3fa 100644 --- a/test/mcp_oauth_test.exs +++ b/test/mcp_oauth_test.exs @@ -710,7 +710,7 @@ defmodule Imp.MCPOAuthTest do defp fake_state, do: Agent.get(Imp.MCPOAuthTest.FakeState, & &1) # The host-owned redirect path: no loopback listener, the host hands the - # callback parameters to complete/2 itself. This is Haven's shape. + # callback parameters to complete/2 itself. This is a desktop client's shape. defp authorize_without_browser(store, resource_url) do assert {:ok, pending} = OAuth.begin(store, resource_url, diff --git a/test/react_v2_test.exs b/test/react_v2_test.exs index 4d71d46e..3415e059 100644 --- a/test/react_v2_test.exs +++ b/test/react_v2_test.exs @@ -318,7 +318,7 @@ defmodule ReActV2Test do end test "a forced submit on an OpenRouter client with a configured effort still runs" do - # The live failure: a resident's client carried an effort, ReAct's forced + # The live failure: a host's client carried an effort, ReAct's forced # submit named `reasoning_effort: nil` for that one call, and the client # refused the two as a collision. One option, and nil means none. {:ok, state} = Agent.start_link(fn -> :initial end)