Close local release-quality gaps and reconcile workplan status
Assistant: codex Assistant-Model: gpt-6-astra Assistant-Session: 01a0e332-3365-77c0-8491-084e9ea33ac1
This commit is contained in:
parent
37436bb562
commit
7cd633986e
56 changed files with 462 additions and 237 deletions
|
|
@ -2,7 +2,7 @@
|
||||||
# Custodian Brief — llm-connect
|
# Custodian Brief — llm-connect
|
||||||
|
|
||||||
**Domain:** agents
|
**Domain:** agents
|
||||||
**Last synced:** 2026-09-09 19:47 UTC
|
**Last synced:** 2026-09-27 14:15 UTC
|
||||||
**State Hub:** http://127.0.0.1:8000 *(adjust if running on a remote machine)*
|
**State Hub:** http://127.0.0.1:8000 *(adjust if running on a remote machine)*
|
||||||
|
|
||||||
## Active Workstreams
|
## Active Workstreams
|
||||||
|
|
@ -12,7 +12,7 @@ Progress: 3/4 done | workplan_id: `d396a090-c2ed-5042-aaea-b75fd1d3a471`
|
||||||
|
|
||||||
**Open tasks:**
|
**Open tasks:**
|
||||||
- ! Integrate the admitted owner route and prove production confinement `98a38d75`
|
- ! Integrate the admitted owner route and prove production confinement `98a38d75`
|
||||||
*(wait: Requires trusted owner hosting and actual lease/token delivery, provider custody and direct-route denial under HFACT T03/T04; accepted tariff/FX and G0 remain HFACT T01.)*
|
*(wait: Local Unix hosting, worker lease/token lifecycle and bwrap confinement proved; requires admitted credential-to-owner bootstrap, matched protected artifact and Railiance custody/placement under HFACT T03/T04; live tariff/FX and G0 remain HFACT T01.)*
|
||||||
|
|
||||||
---
|
---
|
||||||
## MCP Orientation (when available)
|
## MCP Orientation (when available)
|
||||||
|
|
|
||||||
4
Makefile
4
Makefile
|
|
@ -1,7 +1,7 @@
|
||||||
-include .env
|
-include .env
|
||||||
export
|
export
|
||||||
|
|
||||||
.PHONY: install test lint typecheck start help
|
.PHONY: install test lint typecheck check start help
|
||||||
|
|
||||||
UV ?= $(shell command -v uv 2>/dev/null || if [ -x "$$HOME/.local/bin/uv" ]; then printf "%s" "$$HOME/.local/bin/uv"; else printf "%s" "uv"; fi)
|
UV ?= $(shell command -v uv 2>/dev/null || if [ -x "$$HOME/.local/bin/uv" ]; then printf "%s" "$$HOME/.local/bin/uv"; else printf "%s" "uv"; fi)
|
||||||
|
|
||||||
|
|
@ -23,6 +23,8 @@ lint: ## Run ruff linter
|
||||||
typecheck: ## Run mypy type checker
|
typecheck: ## Run mypy type checker
|
||||||
$(UV) run mypy llm_connect
|
$(UV) run mypy llm_connect
|
||||||
|
|
||||||
|
check: lint typecheck test ## Run all release quality checks
|
||||||
|
|
||||||
start: ## Start llm-connect HTTP server (reads .env and LLM_CONNECT_* overrides)
|
start: ## Start llm-connect HTTP server (reads .env and LLM_CONNECT_* overrides)
|
||||||
$(UV) run python -m llm_connect.server \
|
$(UV) run python -m llm_connect.server \
|
||||||
--host $(LLM_CONNECT_HOST) \
|
--host $(LLM_CONNECT_HOST) \
|
||||||
|
|
|
||||||
|
|
@ -12,11 +12,11 @@
|
||||||
| workplan | LLM-WP-0006 | finished | — | workplans/LLM-WP-0006-activity-core-always-on-endpoint.md |
|
| workplan | LLM-WP-0006 | finished | — | workplans/LLM-WP-0006-activity-core-always-on-endpoint.md |
|
||||||
| workplan | LLM-WP-0007 | finished | — | workplans/LLM-WP-0007-kimi-k3-default-and-spend-reporting.md |
|
| workplan | LLM-WP-0007 | finished | — | workplans/LLM-WP-0007-kimi-k3-default-and-spend-reporting.md |
|
||||||
| workplan | LLM-WP-0008 | finished | — | workplans/LLM-WP-0008-provider-account-balance-cli.md |
|
| workplan | LLM-WP-0008 | finished | — | workplans/LLM-WP-0008-provider-account-balance-cli.md |
|
||||||
| workplan | LLM-WP-0009 | active | — | workplans/LLM-WP-0009-owner-metered-messages-transport.md |
|
| workplan | LLM-WP-0009 | blocked | — | workplans/LLM-WP-0009-owner-metered-messages-transport.md |
|
||||||
| workplan | LLM-WP-0001 | completed | — | workplans/llm-connect-WP-0001-foundation-gaaf-baseline.md |
|
| workplan | LLM-WP-0001 | finished | — | workplans/llm-connect-WP-0001-foundation-gaaf-baseline.md |
|
||||||
| workplan | LLM-WP-0002 | completed | — | workplans/llm-connect-WP-0002-core-extensions.md |
|
| workplan | LLM-WP-0002 | finished | — | workplans/llm-connect-WP-0002-core-extensions.md |
|
||||||
| workplan | LLM-WP-0003 | completed | — | workplans/llm-connect-WP-0003-functional-extensions.md |
|
| workplan | LLM-WP-0003 | finished | — | workplans/llm-connect-WP-0003-functional-extensions.md |
|
||||||
| workplan | LLM-WP-0004 | completed | — | workplans/llm-connect-WP-0004-adaptive-cost-quality-routing.md |
|
| workplan | LLM-WP-0004 | finished | — | workplans/llm-connect-WP-0004-adaptive-cost-quality-routing.md |
|
||||||
| workplan | LLM-WP-0005 | finished | — | workplans/llm-connect-WP-0005-cost-model-and-problem-class-estimators.md |
|
| workplan | LLM-WP-0005 | finished | — | workplans/llm-connect-WP-0005-cost-model-and-problem-class-estimators.md |
|
||||||
| task | LLM-WP-ADHOC-2026-06-02-T01 | done | — | workplans/ADHOC-2026-06-02.md |
|
| task | LLM-WP-ADHOC-2026-06-02-T01 | done | — | workplans/ADHOC-2026-06-02.md |
|
||||||
| task | LLM-WP-ADHOC-2026-06-02-T02 | done | — | workplans/ADHOC-2026-06-02.md |
|
| task | LLM-WP-ADHOC-2026-06-02-T02 | done | — | workplans/ADHOC-2026-06-02.md |
|
||||||
|
|
|
||||||
|
|
@ -8,12 +8,14 @@ REINAH-WP-0003-T05/T06.
|
||||||
## Contract
|
## Contract
|
||||||
|
|
||||||
The owner constructs immutable `MessagesPolicy` with an exact model, tariff
|
The owner constructs immutable `MessagesPolicy` with an exact model, tariff
|
||||||
reference, maximum admitted context/output, integer micro-USD per-token upper
|
reference, input liability context/output ceiling, integer micro-USD per-token upper
|
||||||
rates, explicit beta allowlist, body limit and request timeout. There is no
|
rates, explicit beta allowlist, body limit and request timeout. There is no
|
||||||
built-in live price, FX source, token estimate or default beta grant. Input rates
|
built-in live price, FX source, token estimate or default beta grant. Input rates
|
||||||
must conservatively cover input, both cache-write lifetimes, cache reads and all
|
must conservatively cover input, both cache-write lifetimes, cache reads and all
|
||||||
accepted multipliers. Accepted provider limits and tariff validity remain an
|
accepted multipliers. Accepted provider limits and tariff validity remain an
|
||||||
operator policy responsibility; a fixture policy cannot establish them.
|
operator policy responsibility; a fixture policy cannot establish them. The input
|
||||||
|
context is a reservation assumption, not an enforced token cap: it must cover
|
||||||
|
the provider's maximum accepted context for the admitted model/features.
|
||||||
|
|
||||||
`RequestMeter.reserve_request(token, policy_sha256, liability_microusd)` must
|
`RequestMeter.reserve_request(token, policy_sha256, liability_microusd)` must
|
||||||
atomically check authority and remaining parent capacity, persist the hold and
|
atomically check authority and remaining parent capacity, persist the hold and
|
||||||
|
|
@ -57,14 +59,18 @@ the parent already binds worker, definition, project, target, grant and runtime
|
||||||
digests. Route tokens are random, stored only as hashes, cannot be rebound or
|
digests. Route tokens are random, stored only as hashes, cannot be rebound or
|
||||||
renewed by the workload, and revoke on parent terminal observation.
|
renewed by the workload, and revoke on parent terminal observation.
|
||||||
|
|
||||||
The next integration must host the listener in the trusted owner boundary, keep
|
Rein's `MessagesOwner` now hosts the private listener, binds the accepted queue
|
||||||
provider credentials and ledger inaccessible to the sandbox, deliver only its
|
lease, delivers only the run token/base URL, and revokes on lease loss or exit.
|
||||||
run token/base URL, bind actual lease loss to route revocation, and prove direct
|
Real local bwrap tests prove owner/workload state separation and direct-route
|
||||||
provider and alternate-route denial. This module does not install a listener,
|
denial using a fake provider. The factory's 2026-09-10 placement receipt also
|
||||||
configure a sandbox, resolve credentials, or promote a profile. The installed
|
records synthetic checks of a pinned protected artifact on Railiance. The
|
||||||
CLI fixture demonstrates transport/ledger behavior in a fake-provider namespace;
|
2026-09-27 return additionally records the approved corrected owner/Secrets Engine
|
||||||
it does **not** prove secret or network separation between real owner/workload
|
installation and a synthetic two-request tool/result proof on runtime `b6e4e8a4`,
|
||||||
processes. LLM-WP-0009-T03 retains this owner integration return.
|
profile `harness.agent-dev-local@1.1.1`. The inactive EUR 10 proposal and remaining
|
||||||
|
native approvals are recorded in LLM-WP-0009-T03; no paid proof is claimed.
|
||||||
|
This module does not resolve credentials or promote a profile. Native credential
|
||||||
|
delivery, service/profile admission and recovery, accepted live provider policy,
|
||||||
|
and admitted model/queue evidence remain open under LLM-WP-0009-T03.
|
||||||
|
|
||||||
## Verification and primary protocol references
|
## Verification and primary protocol references
|
||||||
|
|
||||||
|
|
@ -91,6 +97,6 @@ At most 16 request handlers are active; idle header reads time out. TCP host/por
|
||||||
selection cannot coexist with Unix mode. The protocol and durable meter interface
|
selection cannot coexist with Unix mode. The protocol and durable meter interface
|
||||||
are unchanged. Rein's MessagesOwner supplies the private listener, accepted lease
|
are unchanged. Rein's MessagesOwner supplies the private listener, accepted lease
|
||||||
and cancellation hooks; sand-boxer mounts only the socket and enforces sole routing.
|
and cancellation hooks; sand-boxer mounts only the socket and enforces sole routing.
|
||||||
The provider key and ledger remain outside the workload. This source/library path
|
The provider key and ledger remain outside the workload. The source/library path
|
||||||
is tested with a fake provider; accepted custody, protected bootstrap/artifact and
|
and installed bootstrap/artifact were tested with a fake provider; accepted
|
||||||
Railiance placement are still required by LLM-WP-0009-T03.
|
native custody and service admission remain required by LLM-WP-0009-T03.
|
||||||
|
|
|
||||||
92
docs/evidence/2026-09-27-request-admission-quality.json
Normal file
92
docs/evidence/2026-09-27-request-admission-quality.json
Normal file
|
|
@ -0,0 +1,92 @@
|
||||||
|
{
|
||||||
|
"date": "2026-09-27",
|
||||||
|
"base_revision": "37436bb562a3f20abc3b88defc5bae6508c32e7e",
|
||||||
|
"scope": "Source release-quality checks and repository loose-end review; no installed artifact or production acceptance",
|
||||||
|
"before": {
|
||||||
|
"ruff_errors": 177,
|
||||||
|
"mypy_errors": 36
|
||||||
|
},
|
||||||
|
"after": {
|
||||||
|
"command": "make check",
|
||||||
|
"exit_code": 0,
|
||||||
|
"ruff_errors": 0,
|
||||||
|
"mypy_errors": 0,
|
||||||
|
"typechecked_source_files": 35,
|
||||||
|
"tests_passed": 264,
|
||||||
|
"tests_skipped": 0
|
||||||
|
},
|
||||||
|
"changes_sha256": {
|
||||||
|
"Makefile": "8c2dc0248a53808782ab9258755fe15971a6061ae0b1c2a2f3c0cd57ad7daf11",
|
||||||
|
"contracts/functional/messages-admission.md": "e1385f99f7929f18594fbf9cc90ed9b87448cbe3b7fae6388e8546c5ba55bfd8",
|
||||||
|
"examples/adaptive_routing_fixture_batch.py": "f23c58ff662bb38c852e6c70b6ee31f8aa0331ec9311e07821b81bf21e5c1c55",
|
||||||
|
"llm_connect/__init__.py": "e921043aa3fbc3a59bbce637dc38abe9b0e1ab30493bb0b27d8e0c9463b23a13",
|
||||||
|
"llm_connect/_diagnostics.py": "51ce22556ab2c7d7b4c7d7ffd4741d6e3c231d073366d9cbeb0b89b7ad2c85ae",
|
||||||
|
"llm_connect/_http.py": "df309e5dbcb4c163bfe34314e37c5ab8f1c54a4ab55da4102cd143c2940a2356",
|
||||||
|
"llm_connect/_payload.py": "a598e73d5dc325eda01510ced71c3ee02fb98f296e79189401b998d07bade7e3",
|
||||||
|
"llm_connect/adapter.py": "7ba82b35306526cc58deebd3b4e5f3ec13d177d0794312f734d652ff9033cf5d",
|
||||||
|
"llm_connect/balance.py": "2ddc3e5f60bd5e685c59baa191ed8e6e9d5730fe509ffe96a1b1d38c0e865ca7",
|
||||||
|
"llm_connect/claude_code.py": "5fcf33936d007d892d9eefb4720559e36b6d010a5d6b3830b5273fd6266c5b26",
|
||||||
|
"llm_connect/config.py": "2da9511caa44cc0fa9c8f0797d8b28c9b964f0d06d48f63b317aaf3586dbe835",
|
||||||
|
"llm_connect/embedding_cache.py": "00e7b2c891c1666c7799a70270416819382bed5d5e7776384fa10c906b52f08c",
|
||||||
|
"llm_connect/embedding_factory.py": "ca510a1148d63a4adde76b19a41478b795175be70c2d66afc4d8381b01303849",
|
||||||
|
"llm_connect/embedding_openai.py": "9a03832ca703d90cc222df300c7d485da406200f8d3c35446e210a4a53dddaf2",
|
||||||
|
"llm_connect/exceptions.py": "c459934f1877ae1f44ac50c5cf2630a2b35e4a3525b131b2a0bfd08ac05357f8",
|
||||||
|
"llm_connect/factory.py": "3cd9f74ce49cb076c71496018b3a8525de98d5e9951513a7e4ade27e26273a80",
|
||||||
|
"llm_connect/fx.py": "455f32b28c01371f48102dccc391b0c8536c0ba4f6a18357dc4b22620336b0ac",
|
||||||
|
"llm_connect/gemini.py": "1a5f7f2c69879a95946c6a74b8a0ef05d5fdb070656d973d7a9d513050203b8f",
|
||||||
|
"llm_connect/grading.py": "eb1ec18b6442d45c043da41aae74732e510281ff90c045c3088841d288703380",
|
||||||
|
"llm_connect/models.py": "edc10e34a6f3f45b9d5fccb06c0b66752381bbb360976aa3446ce862d9082eab",
|
||||||
|
"llm_connect/openai.py": "1793730584d4d6f5da3be1de5aca9a04af03548734a688576c997c9336699256",
|
||||||
|
"llm_connect/openrouter.py": "6b2e19dd3ab87ab028295d1e36688fff3b4e18055620f1eace3ca66af1ae6aff",
|
||||||
|
"llm_connect/problem_classes.py": "4a046f3ca544a40dd712f3c9b0d1ecd9eb2e087cc6348ecf468c420a6089001a",
|
||||||
|
"llm_connect/profiles.py": "ab3f6af59f28861c05ec320321e636f08a8eac140c726c36c7677c9985f8055d",
|
||||||
|
"llm_connect/quality.py": "7ff616287d788e86ae8a0ed069d1a7e0b89e319b7f42d476941e15c91ba72ccc",
|
||||||
|
"llm_connect/rates.py": "71186fb944065707c0f873cd0923fd2b24125a8013851d54e2610fcdbf0c08c4",
|
||||||
|
"llm_connect/replay.py": "c0d7d33d95167c09d940c04e01969ea6670436831f723b66b96f41bad612ded0",
|
||||||
|
"llm_connect/routing.py": "35e28e2e06b637b506ddd4f37dd79fad60fe785c412a4843e3117417168acc3f",
|
||||||
|
"llm_connect/server.py": "33ccff82b52b9181d40629e90500be5ce75d4dd08afe5113d75d9d6b9cf83579",
|
||||||
|
"llm_connect/shadowing.py": "c9f649bda3a4306d506554befd702e70fb3f32761502273cc338a147e60238e1",
|
||||||
|
"llm_connect/toml_config.py": "5d14c35e6e6aca6006eed60c9caa63d1dbd9632fb8d9fc6d3e62c99dcb75c965",
|
||||||
|
"llm_connect/usage.py": "ba1c93d11dc98b55c9f87090c4255319f42b413bb7d36426fb46d2ad773f7b37",
|
||||||
|
"pyproject.toml": "e430482f51667b4de47cf23311fb9a8d6e85673b58a659b31d63547798b6684d",
|
||||||
|
"tests/conftest.py": "b93789db1467f2c60cb98ce638f17d78070678159eab19a652932e58c0099006",
|
||||||
|
"tests/test_activity_core_smoke.py": "10d2b607c98114dca4d5143bb487673d15a6a3fd6449a3a0ace8f4d07e94c6d8",
|
||||||
|
"tests/test_adapter.py": "cbbac15d76fedc580287e39e78f69c7e91f534d051c238cc668ae0f65a321023",
|
||||||
|
"tests/test_async.py": "a6a96bcf4e0396484aa3dd81f33aed282d0ce44f58f5e0da18ee5446f01e4fb5",
|
||||||
|
"tests/test_budget.py": "ef6277464b66a56a7a87c6caa672d6bd854683a163dd9262073daea517dddcf7",
|
||||||
|
"tests/test_costs.py": "860d3246e3e8667f7aebc5b9035da9513752db49b5c5c7c789b61c90cb0d2963",
|
||||||
|
"tests/test_exceptions.py": "3063039d2370fc4fcbbfe324e1c8f7bada8ed2c92f90261e4f745a32816ebc0b",
|
||||||
|
"tests/test_factory.py": "cd5870f3133eb9ca8bf06bbc27d030088dc1ac9ba65a388c65331491d8d228cf",
|
||||||
|
"tests/test_models.py": "987dd2f8cc2af8f855c92ad10cdec585ad76249fc225ac35c06b3429f4113f2f",
|
||||||
|
"tests/test_payload.py": "520fb17bfa1955241f832a189657311a4c7406fc4f2fda3772b6b49361726928",
|
||||||
|
"tests/test_problem_classes.py": "0764bf6798c983c62e23872da188d795e31025acb749082cdaccea6b2e02284f",
|
||||||
|
"tests/test_replay.py": "dd33a2c7ca0c825cc25f2da51d333d9b34c2cd76fca9e3fe8a572ff10a927e84",
|
||||||
|
"tests/test_routing.py": "6c29238900c04935c95715aba5dc011876e669ed763fe3e2a4053af1d7b8271b",
|
||||||
|
"tests/test_server.py": "05fb1a5f739d6f9ad5a07ad2c5b755fae0e50c2a70ea3e7da00f0831868708d7",
|
||||||
|
"tests/test_structured_output_smoke.py": "c23648eaa0df948a5e2b8272a9883fd3047593b1ea9bec4129706efcd56bdd72",
|
||||||
|
"workplans/LLM-WP-0009-owner-metered-messages-transport.md": "773129d4ff62c375aa5d8065aee212cd0a18812a37444b66e280fede0bb9cf1e"
|
||||||
|
},
|
||||||
|
"limitations": [
|
||||||
|
"Protected Railiance runtime b6e4e8a4 remains unchanged; these source quality changes are not deployed in it",
|
||||||
|
"Windows lock branch not executed by Linux tests",
|
||||||
|
"T03 native delivery, service admission, accepted live policy and live proof remain open"
|
||||||
|
],
|
||||||
|
"workplan_review": {
|
||||||
|
"source_workplans": 10,
|
||||||
|
"finished": 9,
|
||||||
|
"blocked": [
|
||||||
|
"LLM-WP-0009"
|
||||||
|
],
|
||||||
|
"remaining_tasks": [
|
||||||
|
"LLM-WP-0009-T03"
|
||||||
|
],
|
||||||
|
"new_tasks": 0,
|
||||||
|
"new_workplans": 0,
|
||||||
|
"normalized_to_finished": [
|
||||||
|
"LLM-WP-0001",
|
||||||
|
"LLM-WP-0002",
|
||||||
|
"LLM-WP-0003",
|
||||||
|
"LLM-WP-0004"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
@ -4,6 +4,9 @@
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
|
|
||||||
|
# Imports follow the source-checkout path bootstrap below.
|
||||||
|
# ruff: noqa: E402
|
||||||
import sys
|
import sys
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
|
||||||
|
|
@ -13,14 +13,6 @@ Quick start::
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from llm_connect.adapter import ErrorLLMAdapter, LLMAdapter, MockLLMAdapter
|
from llm_connect.adapter import ErrorLLMAdapter, LLMAdapter, MockLLMAdapter
|
||||||
from llm_connect.claude_code import ClaudeCodeAdapter
|
|
||||||
from llm_connect.config import LLMConfig, load_config
|
|
||||||
from llm_connect.costs import CostEstimate, CostModel, estimate_cost
|
|
||||||
from llm_connect.fx import FxRate, resolve_fx_rate
|
|
||||||
from llm_connect.embedding_adapter import EmbeddingAdapter
|
|
||||||
from llm_connect.embedding_cache import EmbeddingCache
|
|
||||||
from llm_connect.embedding_factory import create_embedding_adapter
|
|
||||||
from llm_connect.embedding_openai import OpenAICompatibleEmbeddingAdapter
|
|
||||||
from llm_connect.balance import (
|
from llm_connect.balance import (
|
||||||
AccountBalance,
|
AccountBalance,
|
||||||
BalanceClientRegistry,
|
BalanceClientRegistry,
|
||||||
|
|
@ -30,6 +22,13 @@ from llm_connect.balance import (
|
||||||
get_account_balance,
|
get_account_balance,
|
||||||
resolve_balance_provider,
|
resolve_balance_provider,
|
||||||
)
|
)
|
||||||
|
from llm_connect.claude_code import ClaudeCodeAdapter
|
||||||
|
from llm_connect.config import LLMConfig, load_config
|
||||||
|
from llm_connect.costs import CostEstimate, CostModel, estimate_cost
|
||||||
|
from llm_connect.embedding_adapter import EmbeddingAdapter
|
||||||
|
from llm_connect.embedding_cache import EmbeddingCache
|
||||||
|
from llm_connect.embedding_factory import create_embedding_adapter
|
||||||
|
from llm_connect.embedding_openai import OpenAICompatibleEmbeddingAdapter
|
||||||
from llm_connect.exceptions import (
|
from llm_connect.exceptions import (
|
||||||
LLMAPIError,
|
LLMAPIError,
|
||||||
LLMBalanceUnsupportedError,
|
LLMBalanceUnsupportedError,
|
||||||
|
|
@ -41,6 +40,7 @@ from llm_connect.exceptions import (
|
||||||
LLMTimeoutError,
|
LLMTimeoutError,
|
||||||
)
|
)
|
||||||
from llm_connect.factory import create_adapter
|
from llm_connect.factory import create_adapter
|
||||||
|
from llm_connect.fx import FxRate, resolve_fx_rate
|
||||||
from llm_connect.gemini import GeminiAdapter
|
from llm_connect.gemini import GeminiAdapter
|
||||||
from llm_connect.grading import (
|
from llm_connect.grading import (
|
||||||
BaselineGrader,
|
BaselineGrader,
|
||||||
|
|
|
||||||
|
|
@ -4,13 +4,13 @@ from __future__ import annotations
|
||||||
|
|
||||||
import copy
|
import copy
|
||||||
import json
|
import json
|
||||||
|
from collections.abc import Iterator, Mapping
|
||||||
from contextlib import contextmanager
|
from contextlib import contextmanager
|
||||||
from contextvars import ContextVar
|
from contextvars import ContextVar
|
||||||
from dataclasses import dataclass, field
|
from dataclasses import dataclass, field
|
||||||
from typing import Any, Iterator, Mapping
|
from typing import Any
|
||||||
from urllib.parse import parse_qsl, urlencode, urlsplit, urlunsplit
|
from urllib.parse import parse_qsl, urlencode, urlsplit, urlunsplit
|
||||||
|
|
||||||
|
|
||||||
_SECRET_QUERY_KEYS = {"key", "api_key", "apikey", "access_token", "token"}
|
_SECRET_QUERY_KEYS = {"key", "api_key", "apikey", "access_token", "token"}
|
||||||
_SECRET_HEADER_TOKENS = ("authorization", "api-key", "apikey", "token", "secret", "key")
|
_SECRET_HEADER_TOKENS = ("authorization", "api-key", "apikey", "token", "secret", "key")
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -7,7 +7,7 @@ Translates HTTP errors into typed :mod:`markitect.llm.exceptions`.
|
||||||
import json
|
import json
|
||||||
import urllib.error
|
import urllib.error
|
||||||
import urllib.request
|
import urllib.request
|
||||||
from typing import Any, Dict, Optional
|
from typing import Any, cast
|
||||||
|
|
||||||
from llm_connect._diagnostics import record_provider_request, record_provider_response
|
from llm_connect._diagnostics import record_provider_request, record_provider_response
|
||||||
from llm_connect.exceptions import (
|
from llm_connect.exceptions import (
|
||||||
|
|
@ -19,10 +19,10 @@ from llm_connect.exceptions import (
|
||||||
|
|
||||||
def post_json(
|
def post_json(
|
||||||
url: str,
|
url: str,
|
||||||
payload: Dict[str, Any],
|
payload: dict[str, Any],
|
||||||
headers: Optional[Dict[str, str]] = None,
|
headers: dict[str, str] | None = None,
|
||||||
timeout: int = 300,
|
timeout: int = 300,
|
||||||
) -> Dict[str, Any]:
|
) -> dict[str, Any]:
|
||||||
"""POST *payload* as JSON and return the parsed response body.
|
"""POST *payload* as JSON and return the parsed response body.
|
||||||
|
|
||||||
Raises:
|
Raises:
|
||||||
|
|
@ -43,9 +43,9 @@ def post_json(
|
||||||
|
|
||||||
def get_json(
|
def get_json(
|
||||||
url: str,
|
url: str,
|
||||||
headers: Optional[Dict[str, str]] = None,
|
headers: dict[str, str] | None = None,
|
||||||
timeout: int = 60,
|
timeout: int = 60,
|
||||||
) -> Dict[str, Any]:
|
) -> dict[str, Any]:
|
||||||
"""GET *url* and return the parsed JSON response body.
|
"""GET *url* and return the parsed JSON response body.
|
||||||
|
|
||||||
Raises:
|
Raises:
|
||||||
|
|
@ -67,14 +67,14 @@ def _read_json_response(
|
||||||
req: urllib.request.Request,
|
req: urllib.request.Request,
|
||||||
*,
|
*,
|
||||||
timeout: int,
|
timeout: int,
|
||||||
) -> Dict[str, Any]:
|
) -> dict[str, Any]:
|
||||||
try:
|
try:
|
||||||
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
||||||
body = resp.read().decode()
|
body = resp.read().decode()
|
||||||
try:
|
try:
|
||||||
parsed = json.loads(body)
|
parsed = json.loads(body)
|
||||||
record_provider_response(status=resp.status, body=parsed)
|
record_provider_response(status=resp.status, body=parsed)
|
||||||
return parsed
|
return cast(dict[str, Any], parsed)
|
||||||
except json.JSONDecodeError as exc:
|
except json.JSONDecodeError as exc:
|
||||||
record_provider_response(status=resp.status, body=body)
|
record_provider_response(status=resp.status, body=body)
|
||||||
preview = body[:300].replace("\n", "\\n")
|
preview = body[:300].replace("\n", "\\n")
|
||||||
|
|
|
||||||
|
|
@ -11,7 +11,6 @@ from llm_connect._diagnostics import (
|
||||||
record_adapter_transformation,
|
record_adapter_transformation,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
# OpenAI Chat Completions fields that map straight through from model_params.
|
# OpenAI Chat Completions fields that map straight through from model_params.
|
||||||
# Anything not in this set is provider-specific and must be either translated
|
# Anything not in this set is provider-specific and must be either translated
|
||||||
# or dropped. Blind merges are deliberately avoided because OpenAI-compatible
|
# or dropped. Blind merges are deliberately avoided because OpenAI-compatible
|
||||||
|
|
|
||||||
|
|
@ -7,10 +7,9 @@ multiple providers (OpenAI, Anthropic, local models, etc.).
|
||||||
|
|
||||||
import asyncio
|
import asyncio
|
||||||
from abc import ABC, abstractmethod
|
from abc import ABC, abstractmethod
|
||||||
from typing import Dict, Any
|
|
||||||
|
|
||||||
from llm_connect.models import RunConfig, LLMResponse, BudgetTracker
|
|
||||||
from llm_connect.exceptions import LLMBudgetExceededError
|
from llm_connect.exceptions import LLMBudgetExceededError
|
||||||
|
from llm_connect.models import LLMResponse, RunConfig
|
||||||
|
|
||||||
|
|
||||||
class LLMAdapter(ABC):
|
class LLMAdapter(ABC):
|
||||||
|
|
@ -131,8 +130,8 @@ class MockLLMAdapter(LLMAdapter):
|
||||||
"""
|
"""
|
||||||
self.mock_response = mock_response
|
self.mock_response = mock_response
|
||||||
self.call_count = 0
|
self.call_count = 0
|
||||||
self.last_prompt = None
|
self.last_prompt: str | None = None
|
||||||
self.last_config = None
|
self.last_config: RunConfig | None = None
|
||||||
|
|
||||||
def execute_prompt(
|
def execute_prompt(
|
||||||
self,
|
self,
|
||||||
|
|
|
||||||
|
|
@ -140,7 +140,7 @@ class BalanceClientRegistry:
|
||||||
return factory()
|
return factory()
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def default(cls) -> "BalanceClientRegistry":
|
def default(cls) -> BalanceClientRegistry:
|
||||||
"""Built-in registry (OpenRouter first; more backends later)."""
|
"""Built-in registry (OpenRouter first; more backends later)."""
|
||||||
return cls(
|
return cls(
|
||||||
{
|
{
|
||||||
|
|
|
||||||
|
|
@ -7,7 +7,6 @@ import json
|
||||||
import os
|
import os
|
||||||
import subprocess
|
import subprocess
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Optional
|
|
||||||
|
|
||||||
from llm_connect._diagnostics import (
|
from llm_connect._diagnostics import (
|
||||||
record_adapter_transformation,
|
record_adapter_transformation,
|
||||||
|
|
@ -30,9 +29,9 @@ class ClaudeCodeAdapter(LLMAdapter):
|
||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
cli_path: Optional[str] = None,
|
cli_path: str | None = None,
|
||||||
model: Optional[str] = None,
|
model: str | None = None,
|
||||||
config: Optional[LLMConfig] = None,
|
config: LLMConfig | None = None,
|
||||||
):
|
):
|
||||||
self._config = config or LLMConfig(provider="claude-code")
|
self._config = config or LLMConfig(provider="claude-code")
|
||||||
self._cli_path = cli_path or self._resolve_cli_path()
|
self._cli_path = cli_path or self._resolve_cli_path()
|
||||||
|
|
@ -124,6 +123,7 @@ class ClaudeCodeAdapter(LLMAdapter):
|
||||||
status=proc.returncode,
|
status=proc.returncode,
|
||||||
body={"stdout": stdout, "stderr": stderr},
|
body={"stdout": stdout, "stderr": stderr},
|
||||||
)
|
)
|
||||||
|
assert proc.returncode is not None # communicate() has reaped the process.
|
||||||
if proc.returncode != 0:
|
if proc.returncode != 0:
|
||||||
raise LLMSubprocessError(
|
raise LLMSubprocessError(
|
||||||
f"claude CLI exited with code {proc.returncode}",
|
f"claude CLI exited with code {proc.returncode}",
|
||||||
|
|
|
||||||
|
|
@ -2,10 +2,10 @@
|
||||||
LLM configuration and API key resolution.
|
LLM configuration and API key resolution.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
from dataclasses import dataclass, field
|
from dataclasses import dataclass, field
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Optional, Dict, Any
|
from typing import Any
|
||||||
import os
|
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
|
|
@ -25,19 +25,19 @@ class LLMConfig:
|
||||||
|
|
||||||
provider: str = "openrouter"
|
provider: str = "openrouter"
|
||||||
model: str = "moonshotai/kimi-k3"
|
model: str = "moonshotai/kimi-k3"
|
||||||
api_key: Optional[str] = None
|
api_key: str | None = None
|
||||||
api_base: str = "https://openrouter.ai/api/v1"
|
api_base: str = "https://openrouter.ai/api/v1"
|
||||||
claude_cli_path: str = "claude"
|
claude_cli_path: str = "claude"
|
||||||
timeout_seconds: int = 300
|
timeout_seconds: int = 300
|
||||||
max_retries: int = 3
|
max_retries: int = 3
|
||||||
extra: Dict[str, Any] = field(default_factory=dict)
|
extra: dict[str, Any] = field(default_factory=dict)
|
||||||
|
|
||||||
|
|
||||||
def resolve_api_key(
|
def resolve_api_key(
|
||||||
explicit: Optional[str] = None,
|
explicit: str | None = None,
|
||||||
env_var: str = "OPENROUTER_API_KEY",
|
env_var: str = "OPENROUTER_API_KEY",
|
||||||
key_file_paths: Optional[list[Path]] = None,
|
key_file_paths: list[Path] | None = None,
|
||||||
) -> Optional[str]:
|
) -> str | None:
|
||||||
"""Return an API key from the first available source.
|
"""Return an API key from the first available source.
|
||||||
|
|
||||||
Resolution order:
|
Resolution order:
|
||||||
|
|
@ -65,7 +65,7 @@ def resolve_api_key(
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
def find_project_root(start: Optional[Path] = None) -> Optional[Path]:
|
def find_project_root(start: Path | None = None) -> Path | None:
|
||||||
"""Walk up from *start* (default CWD) looking for ``pyproject.toml``.
|
"""Walk up from *start* (default CWD) looking for ``pyproject.toml``.
|
||||||
|
|
||||||
Returns the directory containing the marker file, or ``None``.
|
Returns the directory containing the marker file, or ``None``.
|
||||||
|
|
@ -79,8 +79,8 @@ def find_project_root(start: Optional[Path] = None) -> Optional[Path]:
|
||||||
|
|
||||||
def load_config(
|
def load_config(
|
||||||
provider: str = "openrouter",
|
provider: str = "openrouter",
|
||||||
model: Optional[str] = None,
|
model: str | None = None,
|
||||||
api_key: Optional[str] = None,
|
api_key: str | None = None,
|
||||||
**overrides: Any,
|
**overrides: Any,
|
||||||
) -> LLMConfig:
|
) -> LLMConfig:
|
||||||
"""Build an :class:`LLMConfig` with sensible defaults.
|
"""Build an :class:`LLMConfig` with sensible defaults.
|
||||||
|
|
@ -99,7 +99,7 @@ def load_config(
|
||||||
key_file_paths=key_file_paths,
|
key_file_paths=key_file_paths,
|
||||||
)
|
)
|
||||||
|
|
||||||
defaults: Dict[str, Any] = {
|
defaults: dict[str, Any] = {
|
||||||
"provider": provider,
|
"provider": provider,
|
||||||
"model": model or "moonshotai/kimi-k3",
|
"model": model or "moonshotai/kimi-k3",
|
||||||
"api_key": resolved_key,
|
"api_key": resolved_key,
|
||||||
|
|
|
||||||
|
|
@ -8,7 +8,12 @@ automatically invalidated when entity content changes.
|
||||||
|
|
||||||
import json
|
import json
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Optional
|
from typing import TypedDict
|
||||||
|
|
||||||
|
|
||||||
|
class _CacheEntry(TypedDict):
|
||||||
|
digest: str
|
||||||
|
vector: list[float]
|
||||||
|
|
||||||
|
|
||||||
class EmbeddingCache:
|
class EmbeddingCache:
|
||||||
|
|
@ -24,12 +29,12 @@ class EmbeddingCache:
|
||||||
|
|
||||||
def __init__(self, cache_dir: Path):
|
def __init__(self, cache_dir: Path):
|
||||||
self._path = cache_dir / "embeddings.json"
|
self._path = cache_dir / "embeddings.json"
|
||||||
self._data: dict[str, dict] = {}
|
self._data: dict[str, _CacheEntry] = {}
|
||||||
self._hits = 0
|
self._hits = 0
|
||||||
self._misses = 0
|
self._misses = 0
|
||||||
self._load()
|
self._load()
|
||||||
|
|
||||||
def get(self, slug: str, content_digest: str) -> Optional[list[float]]:
|
def get(self, slug: str, content_digest: str) -> list[float] | None:
|
||||||
"""Return the cached vector if *content_digest* matches, else ``None``."""
|
"""Return the cached vector if *content_digest* matches, else ``None``."""
|
||||||
entry = self._data.get(slug)
|
entry = self._data.get(slug)
|
||||||
if entry is not None and entry.get("digest") == content_digest:
|
if entry is not None and entry.get("digest") == content_digest:
|
||||||
|
|
|
||||||
|
|
@ -2,7 +2,8 @@
|
||||||
Factory for creating embedding adapters by provider name.
|
Factory for creating embedding adapters by provider name.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from typing import Optional, Any
|
from collections.abc import Callable
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
from llm_connect.embedding_adapter import EmbeddingAdapter
|
from llm_connect.embedding_adapter import EmbeddingAdapter
|
||||||
from llm_connect.exceptions import LLMConfigurationError
|
from llm_connect.exceptions import LLMConfigurationError
|
||||||
|
|
@ -15,8 +16,8 @@ _EMBEDDING_PROVIDERS = {
|
||||||
|
|
||||||
def create_embedding_adapter(
|
def create_embedding_adapter(
|
||||||
provider: str = "openai",
|
provider: str = "openai",
|
||||||
model: Optional[str] = None,
|
model: str | None = None,
|
||||||
api_key: Optional[str] = None,
|
api_key: str | None = None,
|
||||||
**kwargs: Any,
|
**kwargs: Any,
|
||||||
) -> EmbeddingAdapter:
|
) -> EmbeddingAdapter:
|
||||||
"""Instantiate an :class:`EmbeddingAdapter` for the given *provider*.
|
"""Instantiate an :class:`EmbeddingAdapter` for the given *provider*.
|
||||||
|
|
@ -45,6 +46,6 @@ def create_embedding_adapter(
|
||||||
module_path, class_name = fqn.rsplit(".", 1)
|
module_path, class_name = fqn.rsplit(".", 1)
|
||||||
import importlib
|
import importlib
|
||||||
mod = importlib.import_module(module_path)
|
mod = importlib.import_module(module_path)
|
||||||
cls = getattr(mod, class_name)
|
cls: Callable[..., EmbeddingAdapter] = getattr(mod, class_name)
|
||||||
|
|
||||||
return cls(model=model, api_key=api_key, provider=provider, **kwargs)
|
return cls(model=model, api_key=api_key, provider=provider, **kwargs)
|
||||||
|
|
|
||||||
|
|
@ -8,20 +8,20 @@ API key environment variable.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import time
|
import time
|
||||||
from typing import Optional, Dict, Any
|
from typing import Any
|
||||||
|
|
||||||
from llm_connect.embedding_adapter import EmbeddingAdapter
|
|
||||||
from llm_connect.config import resolve_api_key, find_project_root
|
|
||||||
from llm_connect._http import post_json
|
from llm_connect._http import post_json
|
||||||
|
from llm_connect.config import find_project_root, resolve_api_key
|
||||||
|
from llm_connect.embedding_adapter import EmbeddingAdapter
|
||||||
from llm_connect.exceptions import (
|
from llm_connect.exceptions import (
|
||||||
LLMConfigurationError,
|
|
||||||
LLMAPIError,
|
LLMAPIError,
|
||||||
|
LLMConfigurationError,
|
||||||
LLMRateLimitError,
|
LLMRateLimitError,
|
||||||
)
|
)
|
||||||
|
|
||||||
_DEFAULT_MODEL = "text-embedding-3-small"
|
_DEFAULT_MODEL = "text-embedding-3-small"
|
||||||
|
|
||||||
_PROVIDER_DEFAULTS: Dict[str, Dict[str, str]] = {
|
_PROVIDER_DEFAULTS: dict[str, dict[str, str]] = {
|
||||||
"openai": {
|
"openai": {
|
||||||
"api_base": "https://api.openai.com/v1",
|
"api_base": "https://api.openai.com/v1",
|
||||||
"env_var": "OPENAI_API_KEY",
|
"env_var": "OPENAI_API_KEY",
|
||||||
|
|
@ -42,9 +42,9 @@ class OpenAICompatibleEmbeddingAdapter(EmbeddingAdapter):
|
||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
model: Optional[str] = None,
|
model: str | None = None,
|
||||||
api_key: Optional[str] = None,
|
api_key: str | None = None,
|
||||||
api_base: Optional[str] = None,
|
api_base: str | None = None,
|
||||||
provider: str = "openai",
|
provider: str = "openai",
|
||||||
max_retries: int = 3,
|
max_retries: int = 3,
|
||||||
):
|
):
|
||||||
|
|
@ -85,7 +85,7 @@ class OpenAICompatibleEmbeddingAdapter(EmbeddingAdapter):
|
||||||
)
|
)
|
||||||
|
|
||||||
url = f"{self._api_base}/embeddings"
|
url = f"{self._api_base}/embeddings"
|
||||||
payload: Dict[str, Any] = {
|
payload: dict[str, Any] = {
|
||||||
"model": self._model,
|
"model": self._model,
|
||||||
"input": texts,
|
"input": texts,
|
||||||
}
|
}
|
||||||
|
|
@ -105,10 +105,10 @@ class OpenAICompatibleEmbeddingAdapter(EmbeddingAdapter):
|
||||||
def _post_with_retries(
|
def _post_with_retries(
|
||||||
self,
|
self,
|
||||||
url: str,
|
url: str,
|
||||||
payload: Dict[str, Any],
|
payload: dict[str, Any],
|
||||||
headers: Dict[str, str],
|
headers: dict[str, str],
|
||||||
) -> Dict[str, Any]:
|
) -> dict[str, Any]:
|
||||||
last_exc: Optional[Exception] = None
|
last_exc: Exception | None = None
|
||||||
for attempt in range(self._max_retries + 1):
|
for attempt in range(self._max_retries + 1):
|
||||||
try:
|
try:
|
||||||
return post_json(url, payload, headers)
|
return post_json(url, payload, headers)
|
||||||
|
|
|
||||||
|
|
@ -2,7 +2,7 @@
|
||||||
LLM-specific exceptions.
|
LLM-specific exceptions.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from typing import Optional, Dict, Any
|
from typing import Any
|
||||||
|
|
||||||
|
|
||||||
class LLMError(Exception):
|
class LLMError(Exception):
|
||||||
|
|
@ -11,8 +11,8 @@ class LLMError(Exception):
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
message: str,
|
message: str,
|
||||||
cause: Optional[Exception] = None,
|
cause: Exception | None = None,
|
||||||
context: Optional[Dict[str, Any]] = None,
|
context: dict[str, Any] | None = None,
|
||||||
):
|
):
|
||||||
super().__init__(message)
|
super().__init__(message)
|
||||||
self.cause = cause
|
self.cause = cause
|
||||||
|
|
@ -46,8 +46,8 @@ class LLMAPIError(LLMError):
|
||||||
message: str,
|
message: str,
|
||||||
status_code: int = 0,
|
status_code: int = 0,
|
||||||
response_body: str = "",
|
response_body: str = "",
|
||||||
cause: Optional[Exception] = None,
|
cause: Exception | None = None,
|
||||||
context: Optional[Dict[str, Any]] = None,
|
context: dict[str, Any] | None = None,
|
||||||
):
|
):
|
||||||
super().__init__(message, cause=cause, context=context)
|
super().__init__(message, cause=cause, context=context)
|
||||||
self.status_code = status_code
|
self.status_code = status_code
|
||||||
|
|
@ -79,8 +79,8 @@ class LLMBudgetExceededError(LLMError):
|
||||||
total: int = 0,
|
total: int = 0,
|
||||||
spent: int = 0,
|
spent: int = 0,
|
||||||
requested: int = 0,
|
requested: int = 0,
|
||||||
cause: Optional[Exception] = None,
|
cause: Exception | None = None,
|
||||||
context: Optional[Dict[str, Any]] = None,
|
context: dict[str, Any] | None = None,
|
||||||
):
|
):
|
||||||
if context is None:
|
if context is None:
|
||||||
context = {"total": total, "spent": spent, "requested": requested}
|
context = {"total": total, "spent": spent, "requested": requested}
|
||||||
|
|
@ -102,9 +102,9 @@ class LLMBalanceUnsupportedError(LLMConfigurationError):
|
||||||
self,
|
self,
|
||||||
message: str,
|
message: str,
|
||||||
provider: str = "",
|
provider: str = "",
|
||||||
supported: Optional[list[str]] = None,
|
supported: list[str] | None = None,
|
||||||
cause: Optional[Exception] = None,
|
cause: Exception | None = None,
|
||||||
context: Optional[Dict[str, Any]] = None,
|
context: dict[str, Any] | None = None,
|
||||||
):
|
):
|
||||||
supported_list = list(supported or [])
|
supported_list = list(supported or [])
|
||||||
if context is None:
|
if context is None:
|
||||||
|
|
@ -127,8 +127,8 @@ class LLMSubprocessError(LLMError):
|
||||||
message: str,
|
message: str,
|
||||||
return_code: int = 1,
|
return_code: int = 1,
|
||||||
stderr: str = "",
|
stderr: str = "",
|
||||||
cause: Optional[Exception] = None,
|
cause: Exception | None = None,
|
||||||
context: Optional[Dict[str, Any]] = None,
|
context: dict[str, Any] | None = None,
|
||||||
):
|
):
|
||||||
super().__init__(message, cause=cause, context=context)
|
super().__init__(message, cause=cause, context=context)
|
||||||
self.return_code = return_code
|
self.return_code = return_code
|
||||||
|
|
|
||||||
|
|
@ -3,13 +3,14 @@ Factory for creating LLM adapters by provider name.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import os
|
import os
|
||||||
from typing import Optional, Dict, Any
|
from collections.abc import Callable
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
from llm_connect.adapter import LLMAdapter
|
from llm_connect.adapter import LLMAdapter
|
||||||
from llm_connect.exceptions import LLMConfigurationError
|
from llm_connect.exceptions import LLMConfigurationError
|
||||||
|
|
||||||
# Lazy imports to avoid pulling in every adapter at module load time.
|
# Lazy imports to avoid pulling in every adapter at module load time.
|
||||||
_PROVIDERS: Dict[str, str] = {
|
_PROVIDERS: dict[str, str] = {
|
||||||
"openrouter": "llm_connect.openrouter.OpenRouterAdapter",
|
"openrouter": "llm_connect.openrouter.OpenRouterAdapter",
|
||||||
"claude-code": "llm_connect.claude_code.ClaudeCodeAdapter",
|
"claude-code": "llm_connect.claude_code.ClaudeCodeAdapter",
|
||||||
"gemini": "llm_connect.gemini.GeminiAdapter",
|
"gemini": "llm_connect.gemini.GeminiAdapter",
|
||||||
|
|
@ -20,9 +21,9 @@ _PROVIDERS: Dict[str, str] = {
|
||||||
|
|
||||||
def create_adapter(
|
def create_adapter(
|
||||||
provider: str = "openrouter",
|
provider: str = "openrouter",
|
||||||
model: Optional[str] = None,
|
model: str | None = None,
|
||||||
api_key: Optional[str] = None,
|
api_key: str | None = None,
|
||||||
system_prompt: Optional[str] = None,
|
system_prompt: str | None = None,
|
||||||
**kwargs: Any,
|
**kwargs: Any,
|
||||||
) -> LLMAdapter:
|
) -> LLMAdapter:
|
||||||
"""Instantiate an :class:`LLMAdapter` for the given *provider*.
|
"""Instantiate an :class:`LLMAdapter` for the given *provider*.
|
||||||
|
|
@ -52,7 +53,7 @@ def create_adapter(
|
||||||
module_path, class_name = fqn.rsplit(".", 1)
|
module_path, class_name = fqn.rsplit(".", 1)
|
||||||
import importlib
|
import importlib
|
||||||
mod = importlib.import_module(module_path)
|
mod = importlib.import_module(module_path)
|
||||||
cls = getattr(mod, class_name)
|
cls: Callable[..., LLMAdapter] = getattr(mod, class_name)
|
||||||
|
|
||||||
if provider in ("openrouter", "gemini", "openai"):
|
if provider in ("openrouter", "gemini", "openai"):
|
||||||
return cls(model=model, api_key=api_key, system_prompt=system_prompt, **kwargs)
|
return cls(model=model, api_key=api_key, system_prompt=system_prompt, **kwargs)
|
||||||
|
|
|
||||||
|
|
@ -10,7 +10,6 @@ import os
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from typing import Any
|
from typing import Any
|
||||||
|
|
||||||
|
|
||||||
# Snapshot: euros per one US dollar. Operator can override via env.
|
# Snapshot: euros per one US dollar. Operator can override via env.
|
||||||
DEFAULT_EUR_PER_USD = 0.92
|
DEFAULT_EUR_PER_USD = 0.92
|
||||||
DEFAULT_FX_CAPTURED_AT = "2026-08-03"
|
DEFAULT_FX_CAPTURED_AT = "2026-08-03"
|
||||||
|
|
|
||||||
|
|
@ -3,14 +3,14 @@ Google Gemini adapter — calls the Generative Language REST API directly.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import time
|
import time
|
||||||
from typing import Optional, Dict, Any
|
from typing import Any
|
||||||
|
|
||||||
from llm_connect.adapter import LLMAdapter
|
|
||||||
from llm_connect.models import RunConfig, LLMResponse
|
|
||||||
from llm_connect.config import resolve_api_key, find_project_root
|
|
||||||
from llm_connect._http import post_json
|
from llm_connect._http import post_json
|
||||||
from llm_connect._payload import merge_gemini_model_params
|
from llm_connect._payload import merge_gemini_model_params
|
||||||
|
from llm_connect.adapter import LLMAdapter
|
||||||
|
from llm_connect.config import find_project_root, resolve_api_key
|
||||||
from llm_connect.exceptions import LLMConfigurationError
|
from llm_connect.exceptions import LLMConfigurationError
|
||||||
|
from llm_connect.models import LLMResponse, RunConfig
|
||||||
|
|
||||||
_DEFAULT_MODEL = "gemini-2.5-flash"
|
_DEFAULT_MODEL = "gemini-2.5-flash"
|
||||||
_API_BASE = "https://generativelanguage.googleapis.com/v1beta"
|
_API_BASE = "https://generativelanguage.googleapis.com/v1beta"
|
||||||
|
|
@ -24,9 +24,9 @@ class GeminiAdapter(LLMAdapter):
|
||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
model: Optional[str] = None,
|
model: str | None = None,
|
||||||
api_key: Optional[str] = None,
|
api_key: str | None = None,
|
||||||
system_prompt: Optional[str] = None,
|
system_prompt: str | None = None,
|
||||||
**_kwargs: Any,
|
**_kwargs: Any,
|
||||||
):
|
):
|
||||||
self._model = model or _DEFAULT_MODEL
|
self._model = model or _DEFAULT_MODEL
|
||||||
|
|
@ -53,7 +53,7 @@ class GeminiAdapter(LLMAdapter):
|
||||||
model = self._model
|
model = self._model
|
||||||
|
|
||||||
# Build Gemini request
|
# Build Gemini request
|
||||||
contents: list[Dict[str, Any]] = []
|
contents: list[dict[str, Any]] = []
|
||||||
if self._system_prompt:
|
if self._system_prompt:
|
||||||
contents.append({
|
contents.append({
|
||||||
"role": "user",
|
"role": "user",
|
||||||
|
|
@ -68,7 +68,7 @@ class GeminiAdapter(LLMAdapter):
|
||||||
"parts": [{"text": prompt}],
|
"parts": [{"text": prompt}],
|
||||||
})
|
})
|
||||||
|
|
||||||
payload: Dict[str, Any] = {
|
payload: dict[str, Any] = {
|
||||||
"contents": contents,
|
"contents": contents,
|
||||||
"generationConfig": {
|
"generationConfig": {
|
||||||
"temperature": config.temperature,
|
"temperature": config.temperature,
|
||||||
|
|
|
||||||
|
|
@ -18,7 +18,7 @@ from llm_connect.models import LLMResponse, RunConfig
|
||||||
from llm_connect.similarity import cosine_similarity
|
from llm_connect.similarity import cosine_similarity
|
||||||
|
|
||||||
|
|
||||||
def _validate_score(value: float) -> float:
|
def _validate_score(value: object) -> float:
|
||||||
if not isinstance(value, (int, float)):
|
if not isinstance(value, (int, float)):
|
||||||
raise ValueError("quality_score must be a number between 0 and 1")
|
raise ValueError("quality_score must be a number between 0 and 1")
|
||||||
score = float(value)
|
score = float(value)
|
||||||
|
|
|
||||||
|
|
@ -7,7 +7,7 @@ markitect.prompts.execution.models for backward compatibility.
|
||||||
|
|
||||||
import threading
|
import threading
|
||||||
from dataclasses import dataclass, field
|
from dataclasses import dataclass, field
|
||||||
from typing import Dict, Any, Optional
|
from typing import Any, Optional
|
||||||
|
|
||||||
from llm_connect.exceptions import LLMBudgetExceededError
|
from llm_connect.exceptions import LLMBudgetExceededError
|
||||||
|
|
||||||
|
|
@ -70,13 +70,13 @@ class RunConfig:
|
||||||
model_name: str = "gpt-4"
|
model_name: str = "gpt-4"
|
||||||
temperature: float = 0.7
|
temperature: float = 0.7
|
||||||
max_tokens: int = 2000
|
max_tokens: int = 2000
|
||||||
model_params: Dict[str, Any] = field(default_factory=dict)
|
model_params: dict[str, Any] = field(default_factory=dict)
|
||||||
max_depth: int = 3
|
max_depth: int = 3
|
||||||
skip_if_exists: bool = True
|
skip_if_exists: bool = True
|
||||||
timeout_seconds: int = 300
|
timeout_seconds: int = 300
|
||||||
budget_tracker: Optional["BudgetTracker"] = field(default=None, repr=False)
|
budget_tracker: Optional["BudgetTracker"] = field(default=None, repr=False)
|
||||||
|
|
||||||
def to_dict(self) -> Dict[str, Any]:
|
def to_dict(self) -> dict[str, Any]:
|
||||||
"""Convert to dictionary. ``budget_tracker`` is excluded (runtime object)."""
|
"""Convert to dictionary. ``budget_tracker`` is excluded (runtime object)."""
|
||||||
return {
|
return {
|
||||||
"model_name": self.model_name,
|
"model_name": self.model_name,
|
||||||
|
|
@ -89,7 +89,7 @@ class RunConfig:
|
||||||
}
|
}
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def from_dict(cls, data: Dict[str, Any]) -> "RunConfig":
|
def from_dict(cls, data: dict[str, Any]) -> "RunConfig":
|
||||||
"""Create from dictionary."""
|
"""Create from dictionary."""
|
||||||
return cls(
|
return cls(
|
||||||
model_name=data.get("model_name", "gpt-4"),
|
model_name=data.get("model_name", "gpt-4"),
|
||||||
|
|
@ -116,11 +116,11 @@ class LLMResponse:
|
||||||
"""
|
"""
|
||||||
content: str
|
content: str
|
||||||
model: str
|
model: str
|
||||||
usage: Dict[str, int] = field(default_factory=dict)
|
usage: dict[str, int] = field(default_factory=dict)
|
||||||
finish_reason: str = "stop"
|
finish_reason: str = "stop"
|
||||||
metadata: Dict[str, Any] = field(default_factory=dict)
|
metadata: dict[str, Any] = field(default_factory=dict)
|
||||||
|
|
||||||
def to_dict(self) -> Dict[str, Any]:
|
def to_dict(self) -> dict[str, Any]:
|
||||||
"""Convert to dictionary."""
|
"""Convert to dictionary."""
|
||||||
return {
|
return {
|
||||||
"content": self.content,
|
"content": self.content,
|
||||||
|
|
|
||||||
|
|
@ -3,18 +3,18 @@ OpenAI (ChatGPT) adapter — calls the OpenAI chat completions API.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import time
|
import time
|
||||||
from typing import Optional, Dict, Any
|
from typing import Any
|
||||||
|
|
||||||
from llm_connect.adapter import LLMAdapter
|
|
||||||
from llm_connect.models import RunConfig, LLMResponse
|
|
||||||
from llm_connect.config import resolve_api_key, find_project_root
|
|
||||||
from llm_connect._http import post_json
|
from llm_connect._http import post_json
|
||||||
from llm_connect._payload import merge_openai_chat_model_params
|
from llm_connect._payload import merge_openai_chat_model_params
|
||||||
|
from llm_connect.adapter import LLMAdapter
|
||||||
|
from llm_connect.config import find_project_root, resolve_api_key
|
||||||
from llm_connect.exceptions import (
|
from llm_connect.exceptions import (
|
||||||
LLMConfigurationError,
|
|
||||||
LLMAPIError,
|
LLMAPIError,
|
||||||
|
LLMConfigurationError,
|
||||||
LLMRateLimitError,
|
LLMRateLimitError,
|
||||||
)
|
)
|
||||||
|
from llm_connect.models import LLMResponse, RunConfig
|
||||||
|
|
||||||
_DEFAULT_MODEL = "gpt-4.1-mini"
|
_DEFAULT_MODEL = "gpt-4.1-mini"
|
||||||
_API_BASE = "https://api.openai.com/v1"
|
_API_BASE = "https://api.openai.com/v1"
|
||||||
|
|
@ -25,9 +25,9 @@ class OpenAIAdapter(LLMAdapter):
|
||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
model: Optional[str] = None,
|
model: str | None = None,
|
||||||
api_key: Optional[str] = None,
|
api_key: str | None = None,
|
||||||
system_prompt: Optional[str] = None,
|
system_prompt: str | None = None,
|
||||||
max_retries: int = 3,
|
max_retries: int = 3,
|
||||||
**_kwargs: Any,
|
**_kwargs: Any,
|
||||||
):
|
):
|
||||||
|
|
@ -55,12 +55,12 @@ class OpenAIAdapter(LLMAdapter):
|
||||||
self._preflight_budget(config)
|
self._preflight_budget(config)
|
||||||
model = self._model
|
model = self._model
|
||||||
|
|
||||||
messages: list[Dict[str, str]] = []
|
messages: list[dict[str, str]] = []
|
||||||
if self._system_prompt:
|
if self._system_prompt:
|
||||||
messages.append({"role": "system", "content": self._system_prompt})
|
messages.append({"role": "system", "content": self._system_prompt})
|
||||||
messages.append({"role": "user", "content": prompt})
|
messages.append({"role": "user", "content": prompt})
|
||||||
|
|
||||||
payload: Dict[str, Any] = {
|
payload: dict[str, Any] = {
|
||||||
"model": model,
|
"model": model,
|
||||||
"messages": messages,
|
"messages": messages,
|
||||||
"temperature": config.temperature,
|
"temperature": config.temperature,
|
||||||
|
|
@ -114,11 +114,11 @@ class OpenAIAdapter(LLMAdapter):
|
||||||
def _post_with_retries(
|
def _post_with_retries(
|
||||||
self,
|
self,
|
||||||
url: str,
|
url: str,
|
||||||
payload: Dict[str, Any],
|
payload: dict[str, Any],
|
||||||
headers: Dict[str, str],
|
headers: dict[str, str],
|
||||||
timeout: int,
|
timeout: int,
|
||||||
) -> Dict[str, Any]:
|
) -> dict[str, Any]:
|
||||||
last_exc: Optional[Exception] = None
|
last_exc: Exception | None = None
|
||||||
for attempt in range(self._max_retries + 1):
|
for attempt in range(self._max_retries + 1):
|
||||||
try:
|
try:
|
||||||
return post_json(url, payload, headers, timeout=timeout)
|
return post_json(url, payload, headers, timeout=timeout)
|
||||||
|
|
|
||||||
|
|
@ -3,7 +3,7 @@ OpenRouter adapter - calls the OpenAI-compatible chat completions API.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import time
|
import time
|
||||||
from typing import Any, Dict, Optional
|
from typing import Any
|
||||||
|
|
||||||
from llm_connect._http import post_json
|
from llm_connect._http import post_json
|
||||||
from llm_connect._payload import merge_openai_chat_model_params
|
from llm_connect._payload import merge_openai_chat_model_params
|
||||||
|
|
@ -25,13 +25,13 @@ class OpenRouterAdapter(LLMAdapter):
|
||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
model: Optional[str] = None,
|
model: str | None = None,
|
||||||
api_key: Optional[str] = None,
|
api_key: str | None = None,
|
||||||
api_base: Optional[str] = None,
|
api_base: str | None = None,
|
||||||
config: Optional[LLMConfig] = None,
|
config: LLMConfig | None = None,
|
||||||
system_prompt: Optional[str] = None,
|
system_prompt: str | None = None,
|
||||||
extra_headers: Optional[Dict[str, str]] = None,
|
extra_headers: dict[str, str] | None = None,
|
||||||
max_retries: Optional[int] = None,
|
max_retries: int | None = None,
|
||||||
):
|
):
|
||||||
self._config = config or LLMConfig()
|
self._config = config or LLMConfig()
|
||||||
# Track whether the model was explicitly supplied (constructor or
|
# Track whether the model was explicitly supplied (constructor or
|
||||||
|
|
@ -69,12 +69,12 @@ class OpenRouterAdapter(LLMAdapter):
|
||||||
else:
|
else:
|
||||||
model = config.model_name or self._model
|
model = config.model_name or self._model
|
||||||
|
|
||||||
messages: list[Dict[str, str]] = []
|
messages: list[dict[str, str]] = []
|
||||||
if self._system_prompt:
|
if self._system_prompt:
|
||||||
messages.append({"role": "system", "content": self._system_prompt})
|
messages.append({"role": "system", "content": self._system_prompt})
|
||||||
messages.append({"role": "user", "content": prompt})
|
messages.append({"role": "user", "content": prompt})
|
||||||
|
|
||||||
payload: Dict[str, Any] = {
|
payload: dict[str, Any] = {
|
||||||
"model": model,
|
"model": model,
|
||||||
"messages": messages,
|
"messages": messages,
|
||||||
"temperature": config.temperature,
|
"temperature": config.temperature,
|
||||||
|
|
@ -137,11 +137,11 @@ class OpenRouterAdapter(LLMAdapter):
|
||||||
def _post_with_retries(
|
def _post_with_retries(
|
||||||
self,
|
self,
|
||||||
url: str,
|
url: str,
|
||||||
payload: Dict[str, Any],
|
payload: dict[str, Any],
|
||||||
headers: Dict[str, str],
|
headers: dict[str, str],
|
||||||
timeout: int,
|
timeout: int,
|
||||||
) -> Dict[str, Any]:
|
) -> dict[str, Any]:
|
||||||
last_exc: Optional[Exception] = None
|
last_exc: Exception | None = None
|
||||||
for attempt in range(self._max_retries + 1):
|
for attempt in range(self._max_retries + 1):
|
||||||
try:
|
try:
|
||||||
return post_json(url, payload, headers, timeout=timeout)
|
return post_json(url, payload, headers, timeout=timeout)
|
||||||
|
|
@ -158,6 +158,6 @@ class OpenRouterAdapter(LLMAdapter):
|
||||||
raise last_exc # type: ignore[misc]
|
raise last_exc # type: ignore[misc]
|
||||||
|
|
||||||
|
|
||||||
def _uses_json_schema_response_format(payload: Dict[str, Any]) -> bool:
|
def _uses_json_schema_response_format(payload: dict[str, Any]) -> bool:
|
||||||
response_format = payload.get("response_format")
|
response_format = payload.get("response_format")
|
||||||
return isinstance(response_format, dict) and response_format.get("type") == "json_schema"
|
return isinstance(response_format, dict) and response_format.get("type") == "json_schema"
|
||||||
|
|
|
||||||
|
|
@ -6,7 +6,6 @@ from collections.abc import Mapping, Sequence
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from typing import Any, Protocol
|
from typing import Any, Protocol
|
||||||
|
|
||||||
|
|
||||||
DEFAULT_WORDS_PER_TOKEN = 0.75
|
DEFAULT_WORDS_PER_TOKEN = 0.75
|
||||||
|
|
||||||
|
|
||||||
|
|
@ -66,7 +65,7 @@ class ProblemClass(Protocol):
|
||||||
observations: Sequence[Any],
|
observations: Sequence[Any],
|
||||||
*,
|
*,
|
||||||
min_observations: int = 3,
|
min_observations: int = 3,
|
||||||
) -> "ProblemClass":
|
) -> ProblemClass:
|
||||||
"""Return an estimator with params adapted from observed token use."""
|
"""Return an estimator with params adapted from observed token use."""
|
||||||
...
|
...
|
||||||
|
|
||||||
|
|
@ -99,7 +98,7 @@ class ProblemClassRegistry:
|
||||||
self._classes[name] = problem_class
|
self._classes[name] = problem_class
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def default(cls) -> "ProblemClassRegistry":
|
def default(cls) -> ProblemClassRegistry:
|
||||||
"""Return the built-in problem-class registry."""
|
"""Return the built-in problem-class registry."""
|
||||||
return cls(
|
return cls(
|
||||||
[
|
[
|
||||||
|
|
|
||||||
|
|
@ -5,9 +5,10 @@ from __future__ import annotations
|
||||||
import json
|
import json
|
||||||
import os
|
import os
|
||||||
import threading
|
import threading
|
||||||
|
from collections.abc import Callable, Mapping
|
||||||
from dataclasses import dataclass, field, replace
|
from dataclasses import dataclass, field, replace
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Callable, Mapping
|
from typing import Any
|
||||||
|
|
||||||
from llm_connect.adapter import LLMAdapter
|
from llm_connect.adapter import LLMAdapter
|
||||||
from llm_connect.exceptions import LLMConfigurationError
|
from llm_connect.exceptions import LLMConfigurationError
|
||||||
|
|
|
||||||
|
|
@ -8,13 +8,14 @@ from __future__ import annotations
|
||||||
|
|
||||||
import json
|
import json
|
||||||
import os
|
import os
|
||||||
|
import sys
|
||||||
import threading
|
import threading
|
||||||
|
from collections.abc import Iterator
|
||||||
from contextlib import contextmanager
|
from contextlib import contextmanager
|
||||||
from dataclasses import dataclass, field
|
from dataclasses import dataclass, field
|
||||||
from datetime import datetime, timedelta, timezone
|
from datetime import datetime, timedelta, timezone
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Iterator, TextIO
|
from typing import Any, Literal, TextIO
|
||||||
|
|
||||||
|
|
||||||
_PATH_LOCKS: dict[Path, threading.Lock] = {}
|
_PATH_LOCKS: dict[Path, threading.Lock] = {}
|
||||||
_PATH_LOCKS_GUARD = threading.Lock()
|
_PATH_LOCKS_GUARD = threading.Lock()
|
||||||
|
|
@ -62,7 +63,7 @@ def _path_lock(path: Path) -> threading.Lock:
|
||||||
|
|
||||||
|
|
||||||
def _lock_file(handle: TextIO) -> None:
|
def _lock_file(handle: TextIO) -> None:
|
||||||
if os.name == "nt":
|
if sys.platform == "win32":
|
||||||
import msvcrt
|
import msvcrt
|
||||||
|
|
||||||
msvcrt.locking(handle.fileno(), msvcrt.LK_LOCK, 1)
|
msvcrt.locking(handle.fileno(), msvcrt.LK_LOCK, 1)
|
||||||
|
|
@ -73,7 +74,7 @@ def _lock_file(handle: TextIO) -> None:
|
||||||
|
|
||||||
|
|
||||||
def _unlock_file(handle: TextIO) -> None:
|
def _unlock_file(handle: TextIO) -> None:
|
||||||
if os.name == "nt":
|
if sys.platform == "win32":
|
||||||
import msvcrt
|
import msvcrt
|
||||||
|
|
||||||
msvcrt.locking(handle.fileno(), msvcrt.LK_UNLCK, 1)
|
msvcrt.locking(handle.fileno(), msvcrt.LK_UNLCK, 1)
|
||||||
|
|
@ -84,7 +85,7 @@ def _unlock_file(handle: TextIO) -> None:
|
||||||
|
|
||||||
|
|
||||||
@contextmanager
|
@contextmanager
|
||||||
def _locked_file(path: Path, mode: str) -> Iterator[TextIO]:
|
def _locked_file(path: Path, mode: Literal["a", "a+", "r"]) -> Iterator[TextIO]:
|
||||||
path.parent.mkdir(parents=True, exist_ok=True)
|
path.parent.mkdir(parents=True, exist_ok=True)
|
||||||
local_lock = _path_lock(path)
|
local_lock = _path_lock(path)
|
||||||
with local_lock:
|
with local_lock:
|
||||||
|
|
@ -157,7 +158,7 @@ class QualityObservation:
|
||||||
}
|
}
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def from_dict(cls, data: dict[str, Any]) -> "QualityObservation":
|
def from_dict(cls, data: dict[str, Any]) -> QualityObservation:
|
||||||
"""Create an observation from a JSON-decoded dictionary."""
|
"""Create an observation from a JSON-decoded dictionary."""
|
||||||
return cls(
|
return cls(
|
||||||
task_type=data["task_type"],
|
task_type=data["task_type"],
|
||||||
|
|
|
||||||
|
|
@ -7,7 +7,6 @@ from dataclasses import dataclass
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any
|
from typing import Any
|
||||||
|
|
||||||
|
|
||||||
DEFAULT_RATE_SOURCE_URL = "https://openrouter.ai/models"
|
DEFAULT_RATE_SOURCE_URL = "https://openrouter.ai/models"
|
||||||
DEFAULT_RATE_CAPTURED_AT = "2026-05-17"
|
DEFAULT_RATE_CAPTURED_AT = "2026-05-17"
|
||||||
DEFAULT_RATE_CURRENCY = "USD"
|
DEFAULT_RATE_CURRENCY = "USD"
|
||||||
|
|
@ -60,12 +59,12 @@ class ModelRateRegistry:
|
||||||
return dict(self._rates)
|
return dict(self._rates)
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def default(cls) -> "ModelRateRegistry":
|
def default(cls) -> ModelRateRegistry:
|
||||||
"""Return the bundled OpenRouter list-price snapshot."""
|
"""Return the bundled OpenRouter list-price snapshot."""
|
||||||
return cls(_default_rate_payload())
|
return cls(_default_rate_payload())
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def from_yaml(cls, path: Path | str) -> "ModelRateRegistry":
|
def from_yaml(cls, path: Path | str) -> ModelRateRegistry:
|
||||||
"""Load rates from a YAML file.
|
"""Load rates from a YAML file.
|
||||||
|
|
||||||
The expected shape matches the historic infospace-bench table::
|
The expected shape matches the historic infospace-bench table::
|
||||||
|
|
@ -84,7 +83,7 @@ class ModelRateRegistry:
|
||||||
payload = _load_yaml_mapping(Path(path))
|
payload = _load_yaml_mapping(Path(path))
|
||||||
return cls(_rates_from_payload(payload))
|
return cls(_rates_from_payload(payload))
|
||||||
|
|
||||||
def merged_with(self, override: "ModelRateRegistry") -> "ModelRateRegistry":
|
def merged_with(self, override: ModelRateRegistry) -> ModelRateRegistry:
|
||||||
"""Return a new registry where *override* entries win by model id."""
|
"""Return a new registry where *override* entries win by model id."""
|
||||||
merged = self.all()
|
merged = self.all()
|
||||||
merged.update(override.all())
|
merged.update(override.all())
|
||||||
|
|
@ -111,9 +110,9 @@ def _default_rate_payload() -> dict[str, ModelRate]:
|
||||||
rates: dict[str, ModelRate] = {}
|
rates: dict[str, ModelRate] = {}
|
||||||
for model_id, values in _DEFAULT_RATES.items():
|
for model_id, values in _DEFAULT_RATES.items():
|
||||||
if len(values) == 3:
|
if len(values) == 3:
|
||||||
prompt_rate, completion_rate, captured_at = values # type: ignore[misc]
|
prompt_rate, completion_rate, captured_at = values
|
||||||
else:
|
else:
|
||||||
prompt_rate, completion_rate = values # type: ignore[misc]
|
prompt_rate, completion_rate = values
|
||||||
captured_at = DEFAULT_RATE_CAPTURED_AT
|
captured_at = DEFAULT_RATE_CAPTURED_AT
|
||||||
rates[model_id] = ModelRate(
|
rates[model_id] = ModelRate(
|
||||||
model_id=model_id,
|
model_id=model_id,
|
||||||
|
|
|
||||||
|
|
@ -5,7 +5,7 @@ from __future__ import annotations
|
||||||
import argparse
|
import argparse
|
||||||
import json
|
import json
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any
|
from typing import Any, cast
|
||||||
|
|
||||||
from llm_connect.claude_code import _unwrap_cli_json_envelope
|
from llm_connect.claude_code import _unwrap_cli_json_envelope
|
||||||
from llm_connect.models import RunConfig
|
from llm_connect.models import RunConfig
|
||||||
|
|
@ -51,7 +51,7 @@ def _parse_provider_response(provider: str | None, body: Any, config: RunConfig)
|
||||||
if provider in {"openai", "openrouter"}:
|
if provider in {"openai", "openrouter"}:
|
||||||
if isinstance(body, dict):
|
if isinstance(body, dict):
|
||||||
choice = (body.get("choices") or [{}])[0]
|
choice = (body.get("choices") or [{}])[0]
|
||||||
return choice.get("message", {}).get("content", "")
|
return cast(str, choice.get("message", {}).get("content", ""))
|
||||||
return ""
|
return ""
|
||||||
|
|
||||||
if provider == "gemini":
|
if provider == "gemini":
|
||||||
|
|
|
||||||
|
|
@ -4,9 +4,9 @@ RoutingPolicy — task-type-aware adapter selection (FR-2).
|
||||||
Maps task types to preferred adapters with optional cost-cap fallback.
|
Maps task types to preferred adapters with optional cost-cap fallback.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
from collections.abc import Mapping
|
||||||
from dataclasses import dataclass, field
|
from dataclasses import dataclass, field
|
||||||
from datetime import datetime, timedelta, timezone
|
from datetime import datetime, timedelta, timezone
|
||||||
from typing import List, Mapping, Optional
|
|
||||||
|
|
||||||
from llm_connect.adapter import LLMAdapter
|
from llm_connect.adapter import LLMAdapter
|
||||||
from llm_connect.quality import QualityLedger, QualityObservation
|
from llm_connect.quality import QualityLedger, QualityObservation
|
||||||
|
|
@ -27,8 +27,8 @@ class RoutingRule:
|
||||||
|
|
||||||
task_type: str
|
task_type: str
|
||||||
prefer: LLMAdapter
|
prefer: LLMAdapter
|
||||||
max_cost_per_1k: Optional[float] = None
|
max_cost_per_1k: float | None = None
|
||||||
fallback: Optional[LLMAdapter] = None
|
fallback: LLMAdapter | None = None
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
|
|
@ -50,13 +50,13 @@ class RoutingPolicy:
|
||||||
adapter = policy.resolve("triage")
|
adapter = policy.resolve("triage")
|
||||||
"""
|
"""
|
||||||
|
|
||||||
rules: List[RoutingRule] = field(default_factory=list)
|
rules: list[RoutingRule] = field(default_factory=list)
|
||||||
default: Optional[LLMAdapter] = None
|
default: LLMAdapter | None = None
|
||||||
|
|
||||||
def resolve(
|
def resolve(
|
||||||
self,
|
self,
|
||||||
task_type: str,
|
task_type: str,
|
||||||
estimated_cost_per_1k: Optional[float] = None,
|
estimated_cost_per_1k: float | None = None,
|
||||||
) -> LLMAdapter:
|
) -> LLMAdapter:
|
||||||
"""Return the adapter for *task_type*.
|
"""Return the adapter for *task_type*.
|
||||||
|
|
||||||
|
|
@ -111,11 +111,11 @@ class AdaptiveRoutingPolicy(RoutingPolicy):
|
||||||
caller can use the same policy on day zero and after observations accrue.
|
caller can use the same policy on day zero and after observations accrue.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
ledger: Optional[QualityLedger] = None
|
ledger: QualityLedger | None = None
|
||||||
adapters_by_id: Mapping[str, LLMAdapter] = field(default_factory=dict)
|
adapters_by_id: Mapping[str, LLMAdapter] = field(default_factory=dict)
|
||||||
window_size: int = 20
|
window_size: int = 20
|
||||||
min_observations: int = 1
|
min_observations: int = 1
|
||||||
max_age: Optional[timedelta] = None
|
max_age: timedelta | None = None
|
||||||
|
|
||||||
def __post_init__(self) -> None:
|
def __post_init__(self) -> None:
|
||||||
if self.window_size <= 0:
|
if self.window_size <= 0:
|
||||||
|
|
@ -128,9 +128,9 @@ class AdaptiveRoutingPolicy(RoutingPolicy):
|
||||||
def resolve(
|
def resolve(
|
||||||
self,
|
self,
|
||||||
task_type: str,
|
task_type: str,
|
||||||
estimated_cost_per_1k: Optional[float] = None,
|
estimated_cost_per_1k: float | None = None,
|
||||||
*,
|
*,
|
||||||
quality_floor: Optional[float] = None,
|
quality_floor: float | None = None,
|
||||||
) -> LLMAdapter:
|
) -> LLMAdapter:
|
||||||
"""Return the adaptive adapter for *task_type*.
|
"""Return the adaptive adapter for *task_type*.
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -30,14 +30,14 @@ import time
|
||||||
import uuid
|
import uuid
|
||||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Optional
|
from typing import Any
|
||||||
from urllib.parse import parse_qs, urlsplit
|
from urllib.parse import parse_qs, urlsplit
|
||||||
|
|
||||||
from llm_connect._diagnostics import capture_diagnostics
|
from llm_connect._diagnostics import capture_diagnostics
|
||||||
from llm_connect.adapter import LLMAdapter
|
from llm_connect.adapter import LLMAdapter
|
||||||
from llm_connect.exceptions import (
|
from llm_connect.exceptions import (
|
||||||
LLMBudgetExceededError,
|
|
||||||
LLMAPIError,
|
LLMAPIError,
|
||||||
|
LLMBudgetExceededError,
|
||||||
LLMConfigurationError,
|
LLMConfigurationError,
|
||||||
LLMError,
|
LLMError,
|
||||||
LLMRateLimitError,
|
LLMRateLimitError,
|
||||||
|
|
@ -48,15 +48,21 @@ from llm_connect.profiles import ProfiledLLMAdapter, default_runtime_profiles
|
||||||
from llm_connect.usage import maybe_record_usage, suppress_auto_usage_record
|
from llm_connect.usage import maybe_record_usage, suppress_auto_usage_record
|
||||||
|
|
||||||
|
|
||||||
|
class _AdapterHTTPServer(ThreadingHTTPServer):
|
||||||
|
adapter: LLMAdapter
|
||||||
|
|
||||||
|
|
||||||
class _Handler(BaseHTTPRequestHandler):
|
class _Handler(BaseHTTPRequestHandler):
|
||||||
"""Request handler — adapter injected via server.adapter."""
|
"""Request handler — adapter injected via server.adapter."""
|
||||||
|
|
||||||
def log_message(self, format, *args): # suppress default access log
|
server: _AdapterHTTPServer
|
||||||
|
|
||||||
|
def log_message(self, format: str, *args: Any) -> None: # suppress default access log
|
||||||
pass
|
pass
|
||||||
|
|
||||||
# ── GET ────────────────────────────────────────────────────────
|
# ── GET ────────────────────────────────────────────────────────
|
||||||
|
|
||||||
def do_GET(self):
|
def do_GET(self) -> None:
|
||||||
parsed = urlsplit(self.path)
|
parsed = urlsplit(self.path)
|
||||||
if parsed.path == "/health":
|
if parsed.path == "/health":
|
||||||
self._respond(200, {"status": "ok"})
|
self._respond(200, {"status": "ok"})
|
||||||
|
|
@ -65,7 +71,7 @@ class _Handler(BaseHTTPRequestHandler):
|
||||||
|
|
||||||
# ── POST ───────────────────────────────────────────────────────
|
# ── POST ───────────────────────────────────────────────────────
|
||||||
|
|
||||||
def do_POST(self):
|
def do_POST(self) -> None:
|
||||||
parsed = urlsplit(self.path)
|
parsed = urlsplit(self.path)
|
||||||
if parsed.path != "/execute":
|
if parsed.path != "/execute":
|
||||||
self._respond(404, {"error": "not found"})
|
self._respond(404, {"error": "not found"})
|
||||||
|
|
@ -96,7 +102,7 @@ class _Handler(BaseHTTPRequestHandler):
|
||||||
diagnostics_enabled = debug_enabled or bool(audit_dir)
|
diagnostics_enabled = debug_enabled or bool(audit_dir)
|
||||||
try:
|
try:
|
||||||
with capture_diagnostics(diagnostics_enabled) as diagnostics:
|
with capture_diagnostics(diagnostics_enabled) as diagnostics:
|
||||||
adapter = self.server.adapter # type: ignore[attr-defined]
|
adapter = self.server.adapter
|
||||||
if not adapter.validate_config(config):
|
if not adapter.validate_config(config):
|
||||||
raise LLMConfigurationError(
|
raise LLMConfigurationError(
|
||||||
"Adapter rejected RunConfig",
|
"Adapter rejected RunConfig",
|
||||||
|
|
@ -152,9 +158,9 @@ class LLMServer:
|
||||||
host: str = "127.0.0.1",
|
host: str = "127.0.0.1",
|
||||||
port: int = 8080,
|
port: int = 8080,
|
||||||
) -> None:
|
) -> None:
|
||||||
self._httpd = ThreadingHTTPServer((host, port), _Handler)
|
self._httpd = _AdapterHTTPServer((host, port), _Handler)
|
||||||
self._httpd.adapter = adapter # type: ignore[attr-defined]
|
self._httpd.adapter = adapter
|
||||||
self._thread: Optional[threading.Thread] = None
|
self._thread: threading.Thread | None = None
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def port(self) -> int:
|
def port(self) -> int:
|
||||||
|
|
@ -163,7 +169,7 @@ class LLMServer:
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def host(self) -> str:
|
def host(self) -> str:
|
||||||
return self._httpd.server_address[0]
|
return str(self._httpd.server_address[0])
|
||||||
|
|
||||||
def start(self) -> None:
|
def start(self) -> None:
|
||||||
"""Start serving in a daemon background thread."""
|
"""Start serving in a daemon background thread."""
|
||||||
|
|
@ -185,7 +191,7 @@ class LLMServer:
|
||||||
|
|
||||||
def _build_adapter(
|
def _build_adapter(
|
||||||
provider: str,
|
provider: str,
|
||||||
model: Optional[str],
|
model: str | None,
|
||||||
*,
|
*,
|
||||||
enable_profiles: bool = True,
|
enable_profiles: bool = True,
|
||||||
strict_profiles: bool = False,
|
strict_profiles: bool = False,
|
||||||
|
|
@ -240,7 +246,7 @@ def _error_response(exc: Exception) -> tuple[int, dict]:
|
||||||
|
|
||||||
|
|
||||||
def _error_body(code: str, exc: Exception) -> dict:
|
def _error_body(code: str, exc: Exception) -> dict:
|
||||||
body = {
|
body: dict[str, Any] = {
|
||||||
"error": code,
|
"error": code,
|
||||||
"message": _sanitize_text(_message(exc)),
|
"message": _sanitize_text(_message(exc)),
|
||||||
"type": exc.__class__.__name__,
|
"type": exc.__class__.__name__,
|
||||||
|
|
@ -260,7 +266,7 @@ def _message(exc: Exception) -> str:
|
||||||
|
|
||||||
|
|
||||||
def _safe_context(context: dict) -> dict:
|
def _safe_context(context: dict) -> dict:
|
||||||
safe = {}
|
safe: dict[str, Any] = {}
|
||||||
for key, value in context.items():
|
for key, value in context.items():
|
||||||
lowered = str(key).lower()
|
lowered = str(key).lower()
|
||||||
if any(secret_word in lowered for secret_word in ("key", "secret", "token", "password")):
|
if any(secret_word in lowered for secret_word in ("key", "secret", "token", "password")):
|
||||||
|
|
@ -321,7 +327,7 @@ def _safe_filename(value: str) -> str:
|
||||||
return re.sub(r"[^A-Za-z0-9_.-]+", "-", value).strip("-") or "response"
|
return re.sub(r"[^A-Za-z0-9_.-]+", "-", value).strip("-") or "response"
|
||||||
|
|
||||||
|
|
||||||
def main(argv=None) -> None:
|
def main(argv: list[str] | None = None) -> None:
|
||||||
parser = argparse.ArgumentParser(
|
parser = argparse.ArgumentParser(
|
||||||
prog="python -m llm_connect.server",
|
prog="python -m llm_connect.server",
|
||||||
description="Start llm_connect HTTP serve mode.",
|
description="Start llm_connect HTTP serve mode.",
|
||||||
|
|
|
||||||
|
|
@ -5,9 +5,10 @@ from __future__ import annotations
|
||||||
import asyncio
|
import asyncio
|
||||||
import random
|
import random
|
||||||
import threading
|
import threading
|
||||||
|
from collections.abc import Callable, Mapping
|
||||||
from concurrent.futures import Future, ThreadPoolExecutor
|
from concurrent.futures import Future, ThreadPoolExecutor
|
||||||
from dataclasses import dataclass, field, replace
|
from dataclasses import dataclass, field, replace
|
||||||
from typing import Any, Callable, Mapping
|
from typing import Any
|
||||||
|
|
||||||
from llm_connect.adapter import LLMAdapter
|
from llm_connect.adapter import LLMAdapter
|
||||||
from llm_connect.grading import BaselineGrader
|
from llm_connect.grading import BaselineGrader
|
||||||
|
|
|
||||||
|
|
@ -18,7 +18,6 @@ Resolution order (highest → lowest):
|
||||||
import os
|
import os
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Optional
|
|
||||||
|
|
||||||
import toml
|
import toml
|
||||||
|
|
||||||
|
|
@ -55,8 +54,8 @@ def _dir_config_name(app_name: str) -> str:
|
||||||
@dataclass
|
@dataclass
|
||||||
class LLMLayer:
|
class LLMLayer:
|
||||||
"""One layer of provider/model configuration (may be partial)."""
|
"""One layer of provider/model configuration (may be partial)."""
|
||||||
provider: Optional[str] = None
|
provider: str | None = None
|
||||||
model: Optional[str] = None
|
model: str | None = None
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
|
|
@ -129,7 +128,7 @@ def _clear_llm_section(path: Path, section: str) -> bool:
|
||||||
|
|
||||||
# ── Directory config path helper ─────────────────────────────────────────
|
# ── Directory config path helper ─────────────────────────────────────────
|
||||||
|
|
||||||
def _dir_config_path(app_name: str = "markitect") -> Optional[Path]:
|
def _dir_config_path(app_name: str = "markitect") -> Path | None:
|
||||||
root = find_project_root()
|
root = find_project_root()
|
||||||
if root is None:
|
if root is None:
|
||||||
return None
|
return None
|
||||||
|
|
@ -139,8 +138,8 @@ def _dir_config_path(app_name: str = "markitect") -> Optional[Path]:
|
||||||
# ── Resolution ───────────────────────────────────────────────────────────
|
# ── Resolution ───────────────────────────────────────────────────────────
|
||||||
|
|
||||||
def resolve_llm(
|
def resolve_llm(
|
||||||
cli_provider: Optional[str] = None,
|
cli_provider: str | None = None,
|
||||||
cli_model: Optional[str] = None,
|
cli_model: str | None = None,
|
||||||
app_name: str = "markitect",
|
app_name: str = "markitect",
|
||||||
) -> ResolvedLLM:
|
) -> ResolvedLLM:
|
||||||
"""Walk the 7-level priority chain and return a fully resolved config.
|
"""Walk the 7-level priority chain and return a fully resolved config.
|
||||||
|
|
|
||||||
|
|
@ -9,12 +9,14 @@ from __future__ import annotations
|
||||||
import contextvars
|
import contextvars
|
||||||
import json
|
import json
|
||||||
import os
|
import os
|
||||||
|
import sys
|
||||||
import threading
|
import threading
|
||||||
|
from collections.abc import Iterator
|
||||||
from contextlib import contextmanager
|
from contextlib import contextmanager
|
||||||
from dataclasses import dataclass, field
|
from dataclasses import dataclass, field
|
||||||
from datetime import datetime, timedelta, timezone
|
from datetime import datetime, timedelta, timezone
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Iterator, TextIO
|
from typing import Any, Literal, TextIO
|
||||||
from zoneinfo import ZoneInfo
|
from zoneinfo import ZoneInfo
|
||||||
|
|
||||||
from llm_connect.costs import CostEstimate, estimate_cost
|
from llm_connect.costs import CostEstimate, estimate_cost
|
||||||
|
|
@ -22,7 +24,6 @@ from llm_connect.fx import FxRate
|
||||||
from llm_connect.models import LLMResponse
|
from llm_connect.models import LLMResponse
|
||||||
from llm_connect.rates import ModelRateRegistry
|
from llm_connect.rates import ModelRateRegistry
|
||||||
|
|
||||||
|
|
||||||
ENV_USAGE_LEDGER = "LLM_CONNECT_USAGE_LEDGER"
|
ENV_USAGE_LEDGER = "LLM_CONNECT_USAGE_LEDGER"
|
||||||
ENV_TZ = "LLM_CONNECT_TZ"
|
ENV_TZ = "LLM_CONNECT_TZ"
|
||||||
DEFAULT_TZ = "Europe/Berlin"
|
DEFAULT_TZ = "Europe/Berlin"
|
||||||
|
|
@ -82,7 +83,7 @@ def _path_lock(path: Path) -> threading.Lock:
|
||||||
|
|
||||||
|
|
||||||
def _lock_file(handle: TextIO) -> None:
|
def _lock_file(handle: TextIO) -> None:
|
||||||
if os.name == "nt":
|
if sys.platform == "win32":
|
||||||
import msvcrt
|
import msvcrt
|
||||||
|
|
||||||
msvcrt.locking(handle.fileno(), msvcrt.LK_LOCK, 1)
|
msvcrt.locking(handle.fileno(), msvcrt.LK_LOCK, 1)
|
||||||
|
|
@ -93,7 +94,7 @@ def _lock_file(handle: TextIO) -> None:
|
||||||
|
|
||||||
|
|
||||||
def _unlock_file(handle: TextIO) -> None:
|
def _unlock_file(handle: TextIO) -> None:
|
||||||
if os.name == "nt":
|
if sys.platform == "win32":
|
||||||
import msvcrt
|
import msvcrt
|
||||||
|
|
||||||
msvcrt.locking(handle.fileno(), msvcrt.LK_UNLCK, 1)
|
msvcrt.locking(handle.fileno(), msvcrt.LK_UNLCK, 1)
|
||||||
|
|
@ -104,7 +105,7 @@ def _unlock_file(handle: TextIO) -> None:
|
||||||
|
|
||||||
|
|
||||||
@contextmanager
|
@contextmanager
|
||||||
def _locked_file(path: Path, mode: str) -> Iterator[TextIO]:
|
def _locked_file(path: Path, mode: Literal["a", "a+", "r"]) -> Iterator[TextIO]:
|
||||||
path.parent.mkdir(parents=True, exist_ok=True)
|
path.parent.mkdir(parents=True, exist_ok=True)
|
||||||
local_lock = _path_lock(path)
|
local_lock = _path_lock(path)
|
||||||
with local_lock:
|
with local_lock:
|
||||||
|
|
@ -201,7 +202,7 @@ class UsageEvent:
|
||||||
}
|
}
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def from_dict(cls, data: dict[str, Any]) -> "UsageEvent":
|
def from_dict(cls, data: dict[str, Any]) -> UsageEvent:
|
||||||
"""Create an event from a JSON-decoded dictionary."""
|
"""Create an event from a JSON-decoded dictionary."""
|
||||||
return cls(
|
return cls(
|
||||||
provider=data["provider"],
|
provider=data["provider"],
|
||||||
|
|
|
||||||
|
|
@ -36,6 +36,7 @@ dev = [
|
||||||
|
|
||||||
[tool.pytest.ini_options]
|
[tool.pytest.ini_options]
|
||||||
testpaths = ["tests"]
|
testpaths = ["tests"]
|
||||||
|
pythonpath = ["."]
|
||||||
addopts = "-v"
|
addopts = "-v"
|
||||||
|
|
||||||
[tool.ruff]
|
[tool.ruff]
|
||||||
|
|
|
||||||
|
|
@ -4,8 +4,8 @@ Shared pytest fixtures for llm-connect tests.
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
from llm_connect.models import RunConfig, LLMResponse
|
|
||||||
from llm_connect.adapter import MockLLMAdapter
|
from llm_connect.adapter import MockLLMAdapter
|
||||||
|
from llm_connect.models import LLMResponse, RunConfig
|
||||||
|
|
||||||
|
|
||||||
@pytest.fixture
|
@pytest.fixture
|
||||||
|
|
|
||||||
|
|
@ -7,7 +7,6 @@ from llm_connect.models import RunConfig
|
||||||
from llm_connect.profiles import CUSTODIAN_TRIAGE_BALANCED, ProfiledLLMAdapter, RuntimeProfile
|
from llm_connect.profiles import CUSTODIAN_TRIAGE_BALANCED, ProfiledLLMAdapter, RuntimeProfile
|
||||||
from llm_connect.server import LLMServer
|
from llm_connect.server import LLMServer
|
||||||
|
|
||||||
|
|
||||||
ROOT = Path(__file__).resolve().parents[1]
|
ROOT = Path(__file__).resolve().parents[1]
|
||||||
SCRIPT = ROOT / "scripts" / "smoke_activity_core_endpoint.py"
|
SCRIPT = ROOT / "scripts" / "smoke_activity_core_endpoint.py"
|
||||||
FIXTURE_DIR = ROOT / "fixtures" / "activity_core"
|
FIXTURE_DIR = ROOT / "fixtures" / "activity_core"
|
||||||
|
|
|
||||||
|
|
@ -3,8 +3,9 @@ Tests for MockLLMAdapter and ErrorLLMAdapter (Core adapter utilities).
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
from llm_connect.adapter import MockLLMAdapter, ErrorLLMAdapter
|
|
||||||
from llm_connect.models import RunConfig, LLMResponse
|
from llm_connect.adapter import ErrorLLMAdapter, MockLLMAdapter
|
||||||
|
from llm_connect.models import LLMResponse
|
||||||
|
|
||||||
|
|
||||||
class TestMockLLMAdapter:
|
class TestMockLLMAdapter:
|
||||||
|
|
|
||||||
|
|
@ -3,11 +3,12 @@ Tests for async_execute_prompt (FR-3).
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import asyncio
|
import asyncio
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
from llm_connect.models import RunConfig, BudgetTracker
|
|
||||||
from llm_connect.adapter import MockLLMAdapter
|
from llm_connect.adapter import MockLLMAdapter
|
||||||
from llm_connect.exceptions import LLMBudgetExceededError
|
from llm_connect.exceptions import LLMBudgetExceededError
|
||||||
|
from llm_connect.models import BudgetTracker, RunConfig
|
||||||
|
|
||||||
|
|
||||||
class TestAsyncExecutePrompt:
|
class TestAsyncExecutePrompt:
|
||||||
|
|
@ -46,7 +47,6 @@ class TestAsyncExecutePrompt:
|
||||||
|
|
||||||
def test_concurrent_faster_than_sequential(self):
|
def test_concurrent_faster_than_sequential(self):
|
||||||
"""Gathering N async calls should not be N× slower than one call."""
|
"""Gathering N async calls should not be N× slower than one call."""
|
||||||
import time
|
|
||||||
|
|
||||||
adapter = MockLLMAdapter()
|
adapter = MockLLMAdapter()
|
||||||
config = RunConfig()
|
config = RunConfig()
|
||||||
|
|
|
||||||
|
|
@ -3,11 +3,12 @@ Tests for BudgetTracker (FR-4) and LLMBudgetExceededError.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import threading
|
import threading
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
from llm_connect.models import BudgetTracker, RunConfig
|
|
||||||
from llm_connect.adapter import MockLLMAdapter
|
from llm_connect.adapter import MockLLMAdapter
|
||||||
from llm_connect.exceptions import LLMBudgetExceededError, LLMError
|
from llm_connect.exceptions import LLMBudgetExceededError, LLMError
|
||||||
|
from llm_connect.models import BudgetTracker, RunConfig
|
||||||
|
|
||||||
|
|
||||||
class TestBudgetTracker:
|
class TestBudgetTracker:
|
||||||
|
|
|
||||||
|
|
@ -1,6 +1,6 @@
|
||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
from llm_connect.costs import CostEstimate, CostModel, estimate_cost
|
from llm_connect.costs import CostModel, estimate_cost
|
||||||
from llm_connect.rates import ModelRate, ModelRateRegistry
|
from llm_connect.rates import ModelRate, ModelRateRegistry
|
||||||
|
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -3,13 +3,14 @@ Tests for the LLMError exception hierarchy (Core).
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
from llm_connect.exceptions import (
|
from llm_connect.exceptions import (
|
||||||
LLMError,
|
|
||||||
LLMConfigurationError,
|
|
||||||
LLMAPIError,
|
LLMAPIError,
|
||||||
|
LLMConfigurationError,
|
||||||
|
LLMError,
|
||||||
LLMRateLimitError,
|
LLMRateLimitError,
|
||||||
LLMTimeoutError,
|
|
||||||
LLMSubprocessError,
|
LLMSubprocessError,
|
||||||
|
LLMTimeoutError,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -3,16 +3,17 @@ Tests for create_adapter() and create_embedding_adapter() factories.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
from llm_connect.factory import create_adapter
|
|
||||||
from llm_connect.embedding_factory import create_embedding_adapter
|
|
||||||
from llm_connect.exceptions import LLMConfigurationError
|
|
||||||
from llm_connect.adapter import LLMAdapter
|
from llm_connect.adapter import LLMAdapter
|
||||||
from llm_connect.embedding_adapter import EmbeddingAdapter
|
|
||||||
from llm_connect.openrouter import OpenRouterAdapter
|
|
||||||
from llm_connect.claude_code import ClaudeCodeAdapter
|
from llm_connect.claude_code import ClaudeCodeAdapter
|
||||||
from llm_connect.openai import OpenAIAdapter
|
from llm_connect.embedding_adapter import EmbeddingAdapter
|
||||||
from llm_connect.gemini import GeminiAdapter
|
from llm_connect.embedding_factory import create_embedding_adapter
|
||||||
from llm_connect.embedding_openai import OpenAICompatibleEmbeddingAdapter
|
from llm_connect.embedding_openai import OpenAICompatibleEmbeddingAdapter
|
||||||
|
from llm_connect.exceptions import LLMConfigurationError
|
||||||
|
from llm_connect.factory import create_adapter
|
||||||
|
from llm_connect.gemini import GeminiAdapter
|
||||||
|
from llm_connect.openai import OpenAIAdapter
|
||||||
|
from llm_connect.openrouter import OpenRouterAdapter
|
||||||
|
|
||||||
|
|
||||||
class TestCreateAdapter:
|
class TestCreateAdapter:
|
||||||
|
|
|
||||||
|
|
@ -2,8 +2,7 @@
|
||||||
Tests for RunConfig and LLMResponse (Core models).
|
Tests for RunConfig and LLMResponse (Core models).
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import pytest
|
from llm_connect.models import LLMResponse, RunConfig
|
||||||
from llm_connect.models import RunConfig, LLMResponse
|
|
||||||
|
|
||||||
|
|
||||||
class TestRunConfig:
|
class TestRunConfig:
|
||||||
|
|
|
||||||
|
|
@ -1,6 +1,5 @@
|
||||||
from llm_connect._payload import merge_gemini_model_params, merge_openai_chat_model_params
|
from llm_connect._payload import merge_gemini_model_params, merge_openai_chat_model_params
|
||||||
|
|
||||||
|
|
||||||
STRUCTURED_SCHEMA = {
|
STRUCTURED_SCHEMA = {
|
||||||
"type": "object",
|
"type": "object",
|
||||||
"properties": {
|
"properties": {
|
||||||
|
|
|
||||||
|
|
@ -10,7 +10,6 @@ from llm_connect.problem_classes import (
|
||||||
)
|
)
|
||||||
from llm_connect.quality import QualityObservation
|
from llm_connect.quality import QualityObservation
|
||||||
|
|
||||||
|
|
||||||
DIMENSIONS_BY_CLASS = {
|
DIMENSIONS_BY_CLASS = {
|
||||||
"chunk-summarization": [
|
"chunk-summarization": [
|
||||||
{"chunk_words": 900, "template_words": 150},
|
{"chunk_words": 900, "template_words": 150},
|
||||||
|
|
|
||||||
|
|
@ -1,6 +1,5 @@
|
||||||
from llm_connect.replay import parse_audit_record
|
from llm_connect.replay import parse_audit_record
|
||||||
|
|
||||||
|
|
||||||
STRUCTURED_SCHEMA = {
|
STRUCTURED_SCHEMA = {
|
||||||
"type": "object",
|
"type": "object",
|
||||||
"properties": {
|
"properties": {
|
||||||
|
|
|
||||||
|
|
@ -4,8 +4,8 @@ Tests for RoutingPolicy (FR-2).
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
from llm_connect.routing import RoutingPolicy, RoutingRule
|
|
||||||
from llm_connect.adapter import MockLLMAdapter
|
from llm_connect.adapter import MockLLMAdapter
|
||||||
|
from llm_connect.routing import RoutingPolicy, RoutingRule
|
||||||
|
|
||||||
|
|
||||||
class TestRoutingPolicy:
|
class TestRoutingPolicy:
|
||||||
|
|
|
||||||
|
|
@ -2,12 +2,12 @@
|
||||||
Tests for LLMServer HTTP serve mode (FR-1).
|
Tests for LLMServer HTTP serve mode (FR-1).
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
import json
|
||||||
import threading
|
import threading
|
||||||
import time
|
import time
|
||||||
from concurrent.futures import ThreadPoolExecutor
|
|
||||||
import json
|
|
||||||
import urllib.error
|
import urllib.error
|
||||||
import urllib.request
|
import urllib.request
|
||||||
|
from concurrent.futures import ThreadPoolExecutor
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
|
|
@ -16,7 +16,7 @@ from llm_connect._diagnostics import (
|
||||||
record_provider_request,
|
record_provider_request,
|
||||||
record_provider_response,
|
record_provider_response,
|
||||||
)
|
)
|
||||||
from llm_connect.adapter import MockLLMAdapter, ErrorLLMAdapter
|
from llm_connect.adapter import ErrorLLMAdapter, MockLLMAdapter
|
||||||
from llm_connect.exceptions import LLMAPIError, LLMConfigurationError, LLMTimeoutError
|
from llm_connect.exceptions import LLMAPIError, LLMConfigurationError, LLMTimeoutError
|
||||||
from llm_connect.models import LLMResponse, RunConfig
|
from llm_connect.models import LLMResponse, RunConfig
|
||||||
from llm_connect.profiles import CUSTODIAN_TRIAGE_BALANCED, ProfiledLLMAdapter, RuntimeProfile
|
from llm_connect.profiles import CUSTODIAN_TRIAGE_BALANCED, ProfiledLLMAdapter, RuntimeProfile
|
||||||
|
|
@ -136,6 +136,7 @@ class TestExecute:
|
||||||
except urllib.error.HTTPError as exc:
|
except urllib.error.HTTPError as exc:
|
||||||
status, body = exc.code, json.loads(exc.read())
|
status, body = exc.code, json.loads(exc.read())
|
||||||
assert status == 400
|
assert status == 400
|
||||||
|
assert body == {"error": "invalid JSON body"}
|
||||||
|
|
||||||
def test_unknown_post_path_returns_404(self, server):
|
def test_unknown_post_path_returns_404(self, server):
|
||||||
status, body = _post(
|
status, body = _post(
|
||||||
|
|
|
||||||
|
|
@ -5,7 +5,6 @@ from llm_connect.models import RunConfig
|
||||||
from llm_connect.openai import OpenAIAdapter
|
from llm_connect.openai import OpenAIAdapter
|
||||||
from llm_connect.openrouter import OpenRouterAdapter
|
from llm_connect.openrouter import OpenRouterAdapter
|
||||||
|
|
||||||
|
|
||||||
STRUCTURED_SCHEMA = {
|
STRUCTURED_SCHEMA = {
|
||||||
"type": "object",
|
"type": "object",
|
||||||
"properties": {
|
"properties": {
|
||||||
|
|
|
||||||
|
|
@ -4,12 +4,12 @@ type: workplan
|
||||||
title: "Owner-metered Messages transport for bounded factory execution"
|
title: "Owner-metered Messages transport for bounded factory execution"
|
||||||
domain: agents
|
domain: agents
|
||||||
repo: llm-connect
|
repo: llm-connect
|
||||||
status: active
|
status: blocked
|
||||||
flavor: implementation
|
flavor: implementation
|
||||||
owner: codex
|
owner: codex
|
||||||
topic_slug: llm-connect
|
topic_slug: llm-connect
|
||||||
created: "2026-09-09"
|
created: "2026-09-09"
|
||||||
updated: "2026-09-09"
|
updated: "2026-09-27"
|
||||||
related:
|
related:
|
||||||
- HFACT-WP-0001
|
- HFACT-WP-0001
|
||||||
- REINAH-WP-0003
|
- REINAH-WP-0003
|
||||||
|
|
@ -58,7 +58,7 @@ No actual inference, provider credential or live price/FX policy is involved.
|
||||||
id: LLM-WP-0009-T03
|
id: LLM-WP-0009-T03
|
||||||
status: wait
|
status: wait
|
||||||
priority: high
|
priority: high
|
||||||
blocking_reason: "Local Unix hosting, worker lease/token lifecycle and bwrap confinement proved; requires admitted credential-to-owner bootstrap, matched protected artifact and Railiance custody/placement under HFACT T03/T04; live tariff/FX and G0 remain HFACT T01."
|
blocking_reason: "Corrected Railiance owner/code installation and synthetic two-request tool proof are complete. Remaining: accepted native spend grant/window with FX/tariffs and accounting continuity, regenerated recipient pins and attended per-lane delivery approvals, then real provider and natural queue/tool/commit/recovery evidence under SECRETS-WP-0009-T03, HFACT-WP-0001-T01/T03/T04/T05 and REINAH-WP-0003-T05/T06."
|
||||||
state_hub_task_id: "98a38d75-73ab-5f37-b710-5df9d9681e49"
|
state_hub_task_id: "98a38d75-73ab-5f37-b710-5df9d9681e49"
|
||||||
```
|
```
|
||||||
|
|
||||||
|
|
@ -111,6 +111,113 @@ admission, live provider compatibility and accepted bounds/tariffs/FX, then G0 a
|
||||||
natural model/queue evidence. No protected runtime was installed or promoted,
|
natural model/queue evidence. No protected runtime was installed or promoted,
|
||||||
no existing CCR changed, no secret read or paid execution took place.
|
no existing CCR changed, no secret read or paid execution took place.
|
||||||
|
|
||||||
|
### Release quality and dependency reconciliation — 2026-09-27
|
||||||
|
|
||||||
|
Resolved the reproduced 177 Ruff findings and 36 mypy errors without disabling
|
||||||
|
repository-wide rules. Modernized annotations/imports, typed lazy adapter
|
||||||
|
constructors and cache entries, narrowed text-file modes and platform branches,
|
||||||
|
and declared the HTTP server's injected adapter. JSON boundary casts retain the
|
||||||
|
existing response contracts. The example's source-path bootstrap has an explicit
|
||||||
|
E402 exception; invalid-JSON server coverage now checks the returned error body.
|
||||||
|
Pytest explicitly includes the repository root so the example integration test
|
||||||
|
collects under both the pytest executable and module invocation. `make check`
|
||||||
|
now runs lint, typecheck and tests together: lint clean, all 35 source files type
|
||||||
|
check, **264 tests passed**. Evidence:
|
||||||
|
`docs/evidence/2026-09-27-request-admission-quality.json`.
|
||||||
|
|
||||||
|
Consumed newer owner evidence from
|
||||||
|
`../prj-helixforge-factory/operations/owner-bootstrap-admission.md` and
|
||||||
|
`../prj-helixforge-factory/evidence/2026-09-10-runtime-placement.json`.
|
||||||
|
The one-cycle bootstrap and protected artifact
|
||||||
|
`5371156d2027dde6e8f140cc0a1833c4e90b8c75b3bec862e05f06a957f5fd34`
|
||||||
|
were already installed and synthetically proved on workstation and Railiance.
|
||||||
|
These receipts supersede the earlier statement that no protected artifact had
|
||||||
|
been installed. They do not establish native service or credential admission.
|
||||||
|
Today's source quality changes are not installed in that immutable artifact;
|
||||||
|
any replacement needs a new source pin/build and the same artifact checks.
|
||||||
|
|
||||||
|
T03 remains `wait` and the workplan is `blocked` on these existing owner tasks:
|
||||||
|
|
||||||
|
| Remaining acceptance | Existing owner record |
|
||||||
|
| --- | --- |
|
||||||
|
| Exact recipient configuration and native provider-key delivery | SECRETS-WP-0009-T03 / HFACT-WP-0001-T03 |
|
||||||
|
| Service/profile/consumer admission, isolated queue and private-state recovery | HFACT-WP-0001-T04 / REINAH-WP-0003-T05 |
|
||||||
|
| Accepted live provider compatibility, bounds, maximum tariffs, validity, FX and G0 | HFACT-WP-0001-T01 |
|
||||||
|
| Admitted real model and natural queue proof | REINAH-WP-0003-T06 / HFACT-WP-0001-T05 |
|
||||||
|
|
||||||
|
No accepted live policy or native delivery receipt was found in the owning
|
||||||
|
records. Fixture inputs cannot substitute for them. Closing T03 would contradict
|
||||||
|
its explicit acceptance condition. No credential retrieval, paid request,
|
||||||
|
production mutation or profile activation was performed in this session.
|
||||||
|
|
||||||
|
### Direct cross-repository follow-up — 2026-09-27
|
||||||
|
|
||||||
|
Followed the user's instruction into secrets-engine, rein-aharness and the
|
||||||
|
factory project. The prior summary was stale: the dedicated metered worker
|
||||||
|
identity, private owner state and newer b6e4e8a4 runtime already exist.
|
||||||
|
The installed policy nevertheless reserved 200k tokens against Sonnet 5's 1M
|
||||||
|
context, refused the CLI's actual 64k output request and omitted its additional
|
||||||
|
primary-request beta. Its prior runtime proof always selected profile 1.0.0.
|
||||||
|
|
||||||
|
Corrected rein's proof to require the exact profile/model and record request
|
||||||
|
shape metadata. The real Railiance artifact now passes a synthetic Sonnet 5 /
|
||||||
|
profile 1.1.1 first response, zero-forward exhausted-capacity refusal, private
|
||||||
|
state exclusion, read-only unchanged artifact and cleanup. Prepared the corrected
|
||||||
|
owner config and source catalog pin, plus six exact unapproved action requests
|
||||||
|
in `../secrets-engine/docs/proposals/glas-metered-20260927/README.md`.
|
||||||
|
Secrets Engine's full suite passes 498 tests. The installed old policy is
|
||||||
|
explicitly refused by the new offline review; the candidate passes.
|
||||||
|
|
||||||
|
Maximum request hold is USD 4.64. Only one fits the existing USD 5.74 allowance;
|
||||||
|
a successful tool loop must not be promised under the current EUR 5 cap. The
|
||||||
|
candidate preserves all spend limits and the ledger. Source changes and owner
|
||||||
|
records are updated in their actual repositories. Corrected production config
|
||||||
|
installation and native action approvals remain pending; no secret read, paid
|
||||||
|
request or production owner mutation occurred. The broader useful-change G0 is
|
||||||
|
separate from this disposable Glas proof. Factory receipt:
|
||||||
|
`../prj-helixforge-factory/evidence/2026-09-27-metered-owner-followup.json`.
|
||||||
|
|
||||||
|
### Approved installation and tool-session proposal — 2026-09-27
|
||||||
|
|
||||||
|
The user approved the configuration/code update and separately requested preparation
|
||||||
|
of the €10 tool-session proposal. Installed Secrets Engine `11cc0d5` and corrected
|
||||||
|
owner `e0d3fb84` on Railiance; exact path/hash checks, substituted-command refusal,
|
||||||
|
standalone companion refusal and backend-free owner check pass. Standing worker,
|
||||||
|
spend limits and existing ledger are unchanged; no credentials read or paid calls.
|
||||||
|
Deployment receipt: `../secrets-engine/docs/evidence/2026-09-27-metered-owner-deployment.json`.
|
||||||
|
|
||||||
|
Actual pinned CLI/profile 1.1.1/runtime b6e4e8a4 passes a synthetic two-request Bash
|
||||||
|
tool/result exchange and zero-forward underfunded refusal, with teardown and
|
||||||
|
unchanged artifact. Receipt: `../rein-aharness/docs/evidence/2026-09-27-sonnet5-tool-session-proof.json`.
|
||||||
|
The inactive proposal is `../prj-helixforge-factory/operations/metered-tool-session-proposal.md`:
|
||||||
|
EUR 10 run cap, USD 10 liability, EUR/USD 1.00 treatment, existing native USD 5
|
||||||
|
threshold/daily EUR 20/total EUR 500 retained. Two USD 4.64 holds fit. Installed
|
||||||
|
0.87 FX is below the latest observed 0.876962 reference; fresh validity/FX acceptance
|
||||||
|
is required. No new grant or paid execution is authorized by proposal preparation.
|
||||||
|
|
||||||
|
Remaining owner tasks retain their waiting status: fresh spend grant/window,
|
||||||
|
accounting continuity and replacement recipient pins; attended native per-lane
|
||||||
|
approval/delivery/revocation; then natural queue/model/tool/commit/recovery proof.
|
||||||
|
This return supersedes earlier installation-pending statements, not those gates.
|
||||||
|
|
||||||
|
|
||||||
|
### Repository loose-end review — 2026-09-27
|
||||||
|
|
||||||
|
Reviewed all ten source workplans and their task blocks, including the four
|
||||||
|
legacy `completed` plans. Only T03 remains unfinished; there are no locally
|
||||||
|
ready, active or proposed tasks to implement. LLM-WP-0001 through LLM-WP-0008
|
||||||
|
and the historical ad-hoc plan have all tasks done. Normalized LLM-WP-0001–0004
|
||||||
|
to the canonical `finished` state without changing identities or task history.
|
||||||
|
|
||||||
|
Completed the local release-quality repairs described above and included them
|
||||||
|
in the repository commit: lint/type fixes, typed server/factory/cache boundaries,
|
||||||
|
pytest example imports, and the combined `make check` target. Removed the stale
|
||||||
|
installation-pending blocker from T03's structured record. Current synthetic
|
||||||
|
and deployment receipts establish the local completion; they do not satisfy
|
||||||
|
T03's explicit live acceptance. Retain `status: blocked` / task `wait` for the
|
||||||
|
existing native-spend, delivery and natural-run owner dependencies. No new task
|
||||||
|
or workplan was opened, and no paid or production action is part of this review.
|
||||||
|
|
||||||
## Repair historical source identities blocking primary synchronization
|
## Repair historical source identities blocking primary synchronization
|
||||||
|
|
||||||
```task
|
```task
|
||||||
|
|
|
||||||
|
|
@ -3,13 +3,14 @@ id: LLM-WP-0001
|
||||||
type: workplan
|
type: workplan
|
||||||
title: llm-connect — Foundation & GAAF Baseline
|
title: llm-connect — Foundation & GAAF Baseline
|
||||||
domain: agents
|
domain: agents
|
||||||
status: completed
|
status: finished
|
||||||
owner: llm-connect
|
owner: llm-connect
|
||||||
created: 2026-04-01
|
created: 2026-04-01
|
||||||
repo: llm-connect
|
repo: llm-connect
|
||||||
planning_priority: high
|
planning_priority: high
|
||||||
planning_order: 1
|
planning_order: 1
|
||||||
state_hub_workstream_id: f7f08327-753f-4175-8591-ffa1c3188ebc
|
state_hub_workstream_id: f7f08327-753f-4175-8591-ffa1c3188ebc
|
||||||
|
updated: "2026-09-27"
|
||||||
---
|
---
|
||||||
|
|
||||||
# LLM-WP-0001 — Foundation & GAAF Baseline
|
# LLM-WP-0001 — Foundation & GAAF Baseline
|
||||||
|
|
|
||||||
|
|
@ -3,13 +3,14 @@ id: LLM-WP-0002
|
||||||
type: workplan
|
type: workplan
|
||||||
title: llm-connect — Core Extensions (FR-4 BudgetTracker + FR-3 async)
|
title: llm-connect — Core Extensions (FR-4 BudgetTracker + FR-3 async)
|
||||||
domain: agents
|
domain: agents
|
||||||
status: completed
|
status: finished
|
||||||
owner: llm-connect
|
owner: llm-connect
|
||||||
created: 2026-04-01
|
created: 2026-04-01
|
||||||
repo: llm-connect
|
repo: llm-connect
|
||||||
planning_priority: high
|
planning_priority: high
|
||||||
planning_order: 2
|
planning_order: 2
|
||||||
state_hub_workstream_id: 448fa379-eb9e-4808-b3fa-0078f1e4eaba
|
state_hub_workstream_id: 448fa379-eb9e-4808-b3fa-0078f1e4eaba
|
||||||
|
updated: "2026-09-27"
|
||||||
---
|
---
|
||||||
|
|
||||||
# LLM-WP-0002 — Core Extensions (FR-4 + FR-3)
|
# LLM-WP-0002 — Core Extensions (FR-4 + FR-3)
|
||||||
|
|
|
||||||
|
|
@ -3,13 +3,14 @@ id: LLM-WP-0003
|
||||||
type: workplan
|
type: workplan
|
||||||
title: llm-connect — Functional Extensions (FR-2 RoutingPolicy + FR-1 HTTP server)
|
title: llm-connect — Functional Extensions (FR-2 RoutingPolicy + FR-1 HTTP server)
|
||||||
domain: agents
|
domain: agents
|
||||||
status: completed
|
status: finished
|
||||||
owner: llm-connect
|
owner: llm-connect
|
||||||
created: 2026-04-01
|
created: 2026-04-01
|
||||||
repo: llm-connect
|
repo: llm-connect
|
||||||
planning_priority: high
|
planning_priority: high
|
||||||
planning_order: 3
|
planning_order: 3
|
||||||
state_hub_workstream_id: 7b463cdc-40a2-4cc5-8b55-b59cc5ae3443
|
state_hub_workstream_id: 7b463cdc-40a2-4cc5-8b55-b59cc5ae3443
|
||||||
|
updated: "2026-09-27"
|
||||||
---
|
---
|
||||||
|
|
||||||
# LLM-WP-0003 — Functional Extensions (FR-2 + FR-1)
|
# LLM-WP-0003 — Functional Extensions (FR-2 + FR-1)
|
||||||
|
|
|
||||||
|
|
@ -3,13 +3,14 @@ id: LLM-WP-0004
|
||||||
type: workplan
|
type: workplan
|
||||||
title: Adaptive Cost-Quality Routing
|
title: Adaptive Cost-Quality Routing
|
||||||
domain: agents
|
domain: agents
|
||||||
status: completed
|
status: finished
|
||||||
owner: llm-connect
|
owner: llm-connect
|
||||||
created: 2026-05-17
|
created: 2026-05-17
|
||||||
repo: llm-connect
|
repo: llm-connect
|
||||||
planning_priority: high
|
planning_priority: high
|
||||||
planning_order: 4
|
planning_order: 4
|
||||||
state_hub_workstream_id: e1807fab-e29e-4517-b362-95737a96582d
|
state_hub_workstream_id: e1807fab-e29e-4517-b362-95737a96582d
|
||||||
|
updated: "2026-09-27"
|
||||||
---
|
---
|
||||||
|
|
||||||
# LLM-WP-0004 — Adaptive Cost-Quality Routing
|
# LLM-WP-0004 — Adaptive Cost-Quality Routing
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue