mirror of
https://github.com/Routstr/routstr-core.git
synced 2026-10-05 20:28:23 +00:00
refactor: simplify certification comments
This commit is contained in:
+19
-26
@@ -1267,12 +1267,10 @@ def _served_model_for_provider(model_id: str, provider_pk: int) -> Model | None:
|
|||||||
class _ModelEvaluation:
|
class _ModelEvaluation:
|
||||||
"""One enabled model row's facts, built once and shared by every row.
|
"""One enabled model row's facts, built once and shared by every row.
|
||||||
|
|
||||||
``configured`` is the fee-applied USD view built fresh from the row, or
|
``configured`` is the fee-applied USD view of the row, or ``None`` when the
|
||||||
``None`` when the stored row could not be parsed (``build_error`` then
|
row could not be parsed (``build_error`` carries the exception). ``served``
|
||||||
carries the exception). ``served`` is this provider's live candidate for
|
is this provider's live candidate, or ``None`` when the model is withheld
|
||||||
the model, or ``None`` when it is not being served at all — e.g. an
|
from the served map despite the row being enabled.
|
||||||
unusable stored price holds it back from the served map even though the
|
|
||||||
row itself is enabled.
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
model_id: str
|
model_id: str
|
||||||
@@ -1422,10 +1420,9 @@ def _report_row_cache_rate(
|
|||||||
checked = 0
|
checked = 0
|
||||||
unknown: list[dict[str, object]] = []
|
unknown: list[dict[str, object]] = []
|
||||||
for ev in evaluations:
|
for ev in evaluations:
|
||||||
# Scoped to served models only, like every sibling row: a model the
|
# Served models only, like every sibling row: an unserved model has no
|
||||||
# routing algorithm withholds from the served map (e.g. an unusable
|
# cache-billing behaviour to certify, and
|
||||||
# stored price) has no cache-billing behaviour to certify here —
|
# ``pricing.enabled_models_served`` already flags it.
|
||||||
# ``pricing.enabled_models_served`` already flags it as unserved.
|
|
||||||
if ev.served is None or ev.configured is None:
|
if ev.served is None or ev.configured is None:
|
||||||
continue
|
continue
|
||||||
checked += 1
|
checked += 1
|
||||||
@@ -1514,18 +1511,15 @@ async def certify_upstream_provider(
|
|||||||
) -> dict[str, object]:
|
) -> dict[str, object]:
|
||||||
"""Live certification checks for a configured upstream provider.
|
"""Live certification checks for a configured upstream provider.
|
||||||
|
|
||||||
Unlike the read-only ``GET …/report``, this endpoint probes the
|
Unlike the read-only ``GET …/report``, this probes the upstream over the
|
||||||
upstream over the network: it calls ``/models`` and sends a one-token
|
network and runs the node's cost engine on the real response. It never
|
||||||
completion, then runs the node's own cost engine on the real response.
|
enters the billing path, so it costs at most one completion's worth of
|
||||||
It never enters the billing path — no reservation, no Cashu, no wallet
|
upstream credit and nothing from the node's wallet.
|
||||||
— so it cannot spend the node's wallet. It costs at most one
|
|
||||||
completion's worth of upstream credit.
|
|
||||||
|
|
||||||
The response carries the four ``pricing.*`` rows from the read-only
|
Returns the read-only report's four ``pricing.*`` rows (re-derived here so
|
||||||
report (re-derived here so the certification is self-contained) plus
|
the certification is self-contained), the live rows from
|
||||||
the five live/derived rows from
|
:mod:`routstr.upstream.certification`, and a ``checklist`` of the
|
||||||
:mod:`routstr.upstream.certification`, and a ``checklist`` summarising
|
operator-facing goals.
|
||||||
the four operator-facing goals with ``ok``/``warn``/``fail`` ticks.
|
|
||||||
"""
|
"""
|
||||||
from ..payment.price import sats_usd_price
|
from ..payment.price import sats_usd_price
|
||||||
from ..upstream.certification import (
|
from ..upstream.certification import (
|
||||||
@@ -1558,9 +1552,8 @@ async def certify_upstream_provider(
|
|||||||
|
|
||||||
model_id = payload.model_id
|
model_id = payload.model_id
|
||||||
if not model_id and enabled_rows:
|
if not model_id and enabled_rows:
|
||||||
# Pick the first enabled row that is actually being served — a
|
# Prefer a served model: one withheld from the served map would fail
|
||||||
# model withheld from the served map would fail the chat probe for
|
# the chat probe for a reason unrelated to the endpoint's health.
|
||||||
# a reason unrelated to the endpoint's health.
|
|
||||||
for ev in evaluations:
|
for ev in evaluations:
|
||||||
if ev.served is not None:
|
if ev.served is not None:
|
||||||
model_id = ev.served.id
|
model_id = ev.served.id
|
||||||
@@ -1634,8 +1627,8 @@ async def certify_upstream_provider(
|
|||||||
]
|
]
|
||||||
else:
|
else:
|
||||||
sats_to_usd = sats_usd_price()
|
sats_to_usd = sats_usd_price()
|
||||||
# Clamp the admin-supplied timeout: the probe must never be able to
|
# Clamp the admin-supplied timeout so a probe cannot hold the request
|
||||||
# hold the request open indefinitely.
|
# open indefinitely.
|
||||||
requested = (
|
requested = (
|
||||||
payload.timeout_seconds
|
payload.timeout_seconds
|
||||||
if payload.timeout_seconds is not None
|
if payload.timeout_seconds is not None
|
||||||
|
|||||||
@@ -55,12 +55,10 @@ class NormalizedUsage(BaseModel):
|
|||||||
def parse_token_count(value: object) -> int:
|
def parse_token_count(value: object) -> int:
|
||||||
"""Parse a token count from various formats (int, float, str, bool).
|
"""Parse a token count from various formats (int, float, str, bool).
|
||||||
|
|
||||||
A non-finite count is not a count. ``json.loads`` accepts the bare
|
``json.loads`` accepts bare ``Infinity``/``NaN`` and overflows ``1e999`` to
|
||||||
``Infinity``/``NaN`` literals and overflows ``1e999`` to ``inf``, so an
|
``inf``, so an upstream can put them on the wire. ``int()`` raises on both,
|
||||||
upstream — or an attacker who controls one — can put them on the wire.
|
which would turn a billing path into a 500; reject them like
|
||||||
``int(inf)`` raises ``OverflowError`` and ``int(nan)`` raises
|
``is_usable_rate`` does instead.
|
||||||
``ValueError``; either would turn a billing path into a 500. Same rule as
|
|
||||||
``is_usable_rate``: reject the value, do not crash on it.
|
|
||||||
"""
|
"""
|
||||||
if isinstance(value, bool):
|
if isinstance(value, bool):
|
||||||
return 0
|
return 0
|
||||||
|
|||||||
@@ -1,33 +1,12 @@
|
|||||||
"""Certification checks for an upstream provider endpoint.
|
"""Live certification checks for an upstream provider endpoint.
|
||||||
|
|
||||||
PR #717 established the row contract — ``{id, status, title, detail,
|
Extends the read-only pricing rows, which never touch the network, with the
|
||||||
evidence}`` with ``status`` in ``{ok, warn, fail}`` — and the four pricing
|
ones that must: a ``/models`` heartbeat and a one-token completion.
|
||||||
rows derived from the database row plus the in-process served map. Those
|
|
||||||
rows deliberately never touch the network. This module adds the checks that
|
|
||||||
*must* touch the network, and the checklist view that maps the
|
|
||||||
operator-facing goals onto rows:
|
|
||||||
|
|
||||||
========================= =========================================
|
Probes call the upstream directly with ``httpx``, never through the node's
|
||||||
Goal Row(s)
|
billing path — no reservation, no Cashu, at most one token of upstream spend.
|
||||||
========================= =========================================
|
They sit behind ``POST …/certify`` rather than the read-only ``GET …/report``
|
||||||
Heartbeat ``endpoint.reachable``
|
because they can block for the length of the timeout.
|
||||||
Usage data ``usage.capture``
|
|
||||||
Cost data ``cost.prompt_completion``
|
|
||||||
Pricing in ``/v1/models`` ``pricing.served_matches_configured``,
|
|
||||||
``pricing.enabled_models_served``
|
|
||||||
========================= =========================================
|
|
||||||
|
|
||||||
**Money safety.** Every live check calls the upstream directly with
|
|
||||||
``httpx`` — exactly like the existing ``POST /api/models/test`` probe — and
|
|
||||||
never enters the node's billing path. No reservation is taken, no Cashu
|
|
||||||
token is minted or spent, and the probe asks for a single token
|
|
||||||
(``max_tokens=1``). A probe therefore costs the operator at most one
|
|
||||||
completion's worth of upstream spend and nothing from the node's wallet.
|
|
||||||
|
|
||||||
**Why a separate endpoint.** ``GET …/report`` promises the operator a
|
|
||||||
cheap, non-blocking read. A live probe can hang for the length of its
|
|
||||||
timeout and spends upstream credit, so it lives behind
|
|
||||||
``POST …/certify`` instead of being folded into the read.
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
@@ -61,31 +40,22 @@ STATUS_FAIL = "fail"
|
|||||||
|
|
||||||
TICKS = {STATUS_OK: "☑️", STATUS_WARN: "⚠️", STATUS_FAIL: "❌"}
|
TICKS = {STATUS_OK: "☑️", STATUS_WARN: "⚠️", STATUS_FAIL: "❌"}
|
||||||
|
|
||||||
# A probe must never be able to wedge an admin request. Fifteen seconds is
|
# Bounded so a dead upstream fails the row rather than wedging the request.
|
||||||
# generous for a `/models` listing or a one-token completion on a healthy
|
|
||||||
# upstream, and bounded enough that a dead host fails the row rather than
|
|
||||||
# the request.
|
|
||||||
PROBE_TIMEOUT_SECONDS = 15.0
|
PROBE_TIMEOUT_SECONDS = 15.0
|
||||||
|
|
||||||
# An upper bound for a caller-supplied timeout. The admin endpoint accepts a
|
# Ceiling for the caller-supplied timeout override.
|
||||||
# timeout override, and without a ceiling that override could hold the
|
|
||||||
# request open for as long as the caller likes.
|
|
||||||
MAX_PROBE_TIMEOUT_SECONDS = 60.0
|
MAX_PROBE_TIMEOUT_SECONDS = 60.0
|
||||||
|
|
||||||
# The cheapest request that still exercises the usage/cost path: one token
|
# The cheapest request that still exercises the usage/cost path.
|
||||||
# out. Anything larger only spends more upstream credit for no extra
|
|
||||||
# signal.
|
|
||||||
PROBE_MAX_TOKENS = 1
|
PROBE_MAX_TOKENS = 1
|
||||||
PROBE_PROMPT = "ping"
|
PROBE_PROMPT = "ping"
|
||||||
|
|
||||||
# The reservation ceiling is irrelevant to the token-priced path — it is
|
# ``calculate_cost`` demands a reservation ceiling; any value at or above the
|
||||||
# only the amount held before settlement — but ``calculate_cost`` requires
|
# real charge behaves identically.
|
||||||
# one. Any value at or above the real charge behaves identically.
|
|
||||||
_PROBE_MAX_COST_MSATS = 1_000_000_000
|
_PROBE_MAX_COST_MSATS = 1_000_000_000
|
||||||
|
|
||||||
# Rounding in ``_calculate_from_tokens`` truncates the output component and
|
# ``_calculate_from_tokens`` truncates the output component and folds the
|
||||||
# folds the remainder into the input component, so a one-millisatoshi
|
# remainder into the input one, so a one-msat difference is arithmetic.
|
||||||
# difference is arithmetic, not drift.
|
|
||||||
COST_TOLERANCE_MSATS = 1
|
COST_TOLERANCE_MSATS = 1
|
||||||
|
|
||||||
|
|
||||||
@@ -96,12 +66,8 @@ def certification_row(
|
|||||||
detail: str,
|
detail: str,
|
||||||
evidence: dict[str, Any] | None = None,
|
evidence: dict[str, Any] | None = None,
|
||||||
) -> dict[str, Any]:
|
) -> dict[str, Any]:
|
||||||
"""Build one row of the certification report.
|
"""Build one row, coercing ``evidence`` to a dict so the row contract
|
||||||
|
holds by construction rather than by caller discipline."""
|
||||||
``evidence`` is coerced to a dict so the row contract holds by
|
|
||||||
construction rather than by caller discipline — a caller that passes a
|
|
||||||
list or a string still produces a row a client can read.
|
|
||||||
"""
|
|
||||||
return {
|
return {
|
||||||
"id": row_id,
|
"id": row_id,
|
||||||
"status": status,
|
"status": status,
|
||||||
@@ -116,13 +82,8 @@ def safe_row(
|
|||||||
title: str,
|
title: str,
|
||||||
builder: Callable[[], dict[str, Any]],
|
builder: Callable[[], dict[str, Any]],
|
||||||
) -> dict[str, Any]:
|
) -> dict[str, Any]:
|
||||||
"""Run a row builder, turning any raise into a ``fail`` row.
|
"""Run a row builder, turning any raise into a ``fail`` row: the report is
|
||||||
|
the diagnostic, so it must never be the thing that 500s."""
|
||||||
The report is the diagnostic; it must never be the thing that fails. A
|
|
||||||
builder tripping over a hostile payload — a non-finite count, a body of
|
|
||||||
the wrong shape — becomes a ``fail`` row carrying the exception instead
|
|
||||||
of escaping the endpoint as a 500.
|
|
||||||
"""
|
|
||||||
try:
|
try:
|
||||||
return builder()
|
return builder()
|
||||||
except Exception as exc: # noqa: BLE001 - a raising check is a row status
|
except Exception as exc: # noqa: BLE001 - a raising check is a row status
|
||||||
@@ -140,10 +101,8 @@ def safe_row(
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
# The operator-facing goals, each mapped onto the rows that decide it. A
|
# Operator-facing goals mapped onto the rows that decide them: ``ok`` only when
|
||||||
# goal is ``ok`` only when every row it names is ``ok``; any ``fail`` makes
|
# every named row is ``ok``, ``fail`` if any fails, ``warn`` otherwise.
|
||||||
# it ``fail``; anything else (a ``warn``, or a row that did not run) makes
|
|
||||||
# it ``warn``. Kept as data so the checklist and the row set cannot drift.
|
|
||||||
CHECKLIST_GOALS: tuple[tuple[str, str, tuple[str, ...]], ...] = (
|
CHECKLIST_GOALS: tuple[tuple[str, str, tuple[str, ...]], ...] = (
|
||||||
(
|
(
|
||||||
"heartbeat",
|
"heartbeat",
|
||||||
@@ -169,7 +128,6 @@ CHECKLIST_GOALS: tuple[tuple[str, str, tuple[str, ...]], ...] = (
|
|||||||
|
|
||||||
|
|
||||||
def build_checklist(rows: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
def build_checklist(rows: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
||||||
"""Summarise the rows as the four operator-facing goals with ticks."""
|
|
||||||
by_id = {row["id"]: row for row in rows}
|
by_id = {row["id"]: row for row in rows}
|
||||||
checklist: list[dict[str, Any]] = []
|
checklist: list[dict[str, Any]] = []
|
||||||
for goal, label, row_ids in CHECKLIST_GOALS:
|
for goal, label, row_ids in CHECKLIST_GOALS:
|
||||||
@@ -295,25 +253,17 @@ async def probe_upstream(
|
|||||||
return result
|
return result
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# Row builders are pure: the network lives only in ``probe_upstream`` and
|
||||||
# Row builders
|
# ``run_live_checks``, so every verdict is testable without a socket.
|
||||||
#
|
|
||||||
# Every builder below is pure: it turns an already-fetched fact (a probe
|
|
||||||
# result, a model, a computed cost) into a row. The network lives only in
|
|
||||||
# ``probe_upstream`` and ``run_live_checks``, so a test can exercise each
|
|
||||||
# verdict — including the failure ones — without a socket.
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
def endpoint_validity_row(base_url: str) -> dict[str, Any]:
|
def endpoint_validity_row(base_url: str) -> dict[str, Any]:
|
||||||
"""Check the configured base URL is a well-formed http(s) endpoint."""
|
|
||||||
parsed = urlparse(base_url or "")
|
parsed = urlparse(base_url or "")
|
||||||
problems: list[str] = []
|
problems: list[str] = []
|
||||||
if parsed.scheme not in ("http", "https"):
|
if parsed.scheme not in ("http", "https"):
|
||||||
problems.append(f"scheme {parsed.scheme!r} is not http or https")
|
problems.append(f"scheme {parsed.scheme!r} is not http or https")
|
||||||
# ``netloc`` is truthy for a hostless authority like ``http://:8080``
|
# ``netloc`` is truthy for a hostless authority like ``http://:8080``;
|
||||||
# (``.netloc == ':8080'``) even though there is no host to connect to —
|
# only ``.hostname`` answers whether there is a host to connect to.
|
||||||
# only ``.hostname`` answers "is there a host here".
|
|
||||||
if not parsed.hostname:
|
if not parsed.hostname:
|
||||||
problems.append("no host component")
|
problems.append("no host component")
|
||||||
evidence: dict[str, Any] = {
|
evidence: dict[str, Any] = {
|
||||||
@@ -343,7 +293,6 @@ def endpoint_validity_row(base_url: str) -> dict[str, Any]:
|
|||||||
|
|
||||||
|
|
||||||
def heartbeat_row(probe: ProbeResult) -> dict[str, Any]:
|
def heartbeat_row(probe: ProbeResult) -> dict[str, Any]:
|
||||||
"""Check the upstream's ``/models`` responds — the heartbeat."""
|
|
||||||
evidence: dict[str, Any] = {
|
evidence: dict[str, Any] = {
|
||||||
"url": probe.models_url,
|
"url": probe.models_url,
|
||||||
"status_code": probe.models_status,
|
"status_code": probe.models_status,
|
||||||
@@ -377,7 +326,6 @@ def heartbeat_row(probe: ProbeResult) -> dict[str, Any]:
|
|||||||
|
|
||||||
|
|
||||||
def models_payload_row(probe: ProbeResult) -> dict[str, Any]:
|
def models_payload_row(probe: ProbeResult) -> dict[str, Any]:
|
||||||
"""Check the ``/models`` payload matches the OpenAI list shape."""
|
|
||||||
payload = probe.models_payload
|
payload = probe.models_payload
|
||||||
if not isinstance(payload, dict):
|
if not isinstance(payload, dict):
|
||||||
return certification_row(
|
return certification_row(
|
||||||
@@ -402,8 +350,6 @@ def models_payload_row(probe: ProbeResult) -> dict[str, Any]:
|
|||||||
},
|
},
|
||||||
)
|
)
|
||||||
|
|
||||||
# An empty id is not an id — the CLI discovery path refuses it, so the
|
|
||||||
# row must not certify it either.
|
|
||||||
ids = [
|
ids = [
|
||||||
item["id"]
|
item["id"]
|
||||||
for item in data
|
for item in data
|
||||||
@@ -436,10 +382,8 @@ def models_payload_row(probe: ProbeResult) -> dict[str, Any]:
|
|||||||
def usage_capture_row(probe: ProbeResult) -> dict[str, Any]:
|
def usage_capture_row(probe: ProbeResult) -> dict[str, Any]:
|
||||||
"""Check a completion comes back with token usage the node can bill on.
|
"""Check a completion comes back with token usage the node can bill on.
|
||||||
|
|
||||||
A missing ``usage`` object is the root of the ``(0+0)`` billing bug —
|
A missing ``usage`` object means the node has nothing to price and the
|
||||||
the node has nothing to price, so the request settles for free. That is
|
request settles for free. Broken, but still usable, so ``warn``.
|
||||||
a real defect in the upstream's OpenAI compatibility, but it does not
|
|
||||||
make the endpoint unusable, so it is a ``warn`` rather than a ``fail``.
|
|
||||||
"""
|
"""
|
||||||
evidence: dict[str, Any] = {
|
evidence: dict[str, Any] = {
|
||||||
"url": probe.chat_url,
|
"url": probe.chat_url,
|
||||||
@@ -532,14 +476,9 @@ def _truncate(value: Any, limit: int = 400) -> Any:
|
|||||||
def _reported_usd_cost(payload: dict[str, Any]) -> float:
|
def _reported_usd_cost(payload: dict[str, Any]) -> float:
|
||||||
"""The upstream-reported USD cost, or 0.0 when it reported none.
|
"""The upstream-reported USD cost, or 0.0 when it reported none.
|
||||||
|
|
||||||
Mirrors ``_resolve_usd_cost``'s priority (``cost_details.total_cost``
|
Mirrors ``_resolve_usd_cost``'s priority and shares ``coerce_rate``, so
|
||||||
then ``total_cost`` then ``cost``) so this check knows which branch of
|
this helper and the engine agree on *whether* a cost was reported; only
|
||||||
the engine it is verifying. Coercion goes through the shared
|
the arithmetic below is re-derived independently.
|
||||||
``coerce_rate`` — the one definition of what an upstream-supplied
|
|
||||||
number is — so this helper and the engine agree on *whether* a cost was
|
|
||||||
reported; only the arithmetic below is re-derived independently. Using
|
|
||||||
a private coercion here would disagree with the engine on numeric
|
|
||||||
strings and booleans and manufacture false failures.
|
|
||||||
"""
|
"""
|
||||||
usage = payload.get("usage")
|
usage = payload.get("usage")
|
||||||
if not isinstance(usage, dict):
|
if not isinstance(usage, dict):
|
||||||
@@ -560,19 +499,12 @@ def _reported_usd_cost(payload: dict[str, Any]) -> float:
|
|||||||
def _expected_token_msats(sats_pricing: Any, usage: Any) -> tuple[int, int, int]:
|
def _expected_token_msats(sats_pricing: Any, usage: Any) -> tuple[int, int, int]:
|
||||||
"""Re-derive the token-priced charge independently of the engine.
|
"""Re-derive the token-priced charge independently of the engine.
|
||||||
|
|
||||||
``_calculate_from_tokens`` prices at *msats per 1000 tokens*, rounds
|
Reproduces ``_calculate_from_tokens``'s arithmetic rather than calling the
|
||||||
each component to three decimals, ceilings the sum, then folds the
|
engine and comparing it to itself, so a swapped rate, a dropped cache term
|
||||||
cache cost into the input component by truncating the output one. The
|
or a changed rounding rule shows up as a mismatch.
|
||||||
arithmetic is reproduced here — rather than calling the engine and
|
|
||||||
comparing it to itself — so a swapped input/output rate, a dropped
|
|
||||||
cache term or a changed rounding rule shows up as a mismatch.
|
|
||||||
|
|
||||||
Returns ``(total_msats, input_msats, output_msats)``.
|
Returns ``(total_msats, input_msats, output_msats)``. Raises ``ValueError``
|
||||||
|
on a non-finite rate, which would otherwise crash ``math.ceil`` downstream.
|
||||||
Raises ``ValueError`` when a rate is not finite: ``math.ceil`` on an
|
|
||||||
infinite sum raises ``ValueError`` and on ``NaN`` produces an
|
|
||||||
unrepresentable result, so a non-finite rate is rejected explicitly
|
|
||||||
here rather than surfacing as an opaque crash.
|
|
||||||
"""
|
"""
|
||||||
input_rate = float(sats_pricing.prompt) * 1_000_000.0
|
input_rate = float(sats_pricing.prompt) * 1_000_000.0
|
||||||
output_rate = float(sats_pricing.completion) * 1_000_000.0
|
output_rate = float(sats_pricing.completion) * 1_000_000.0
|
||||||
@@ -619,10 +551,8 @@ def cost_prompt_completion_row(
|
|||||||
) -> dict[str, Any]:
|
) -> dict[str, Any]:
|
||||||
"""Check the node's cost engine prices a real completion correctly.
|
"""Check the node's cost engine prices a real completion correctly.
|
||||||
|
|
||||||
Both the prompt and the completion component are checked: the engine
|
Both components are checked, since the engine folds the truncated output
|
||||||
truncates the output component and folds the remainder into the input
|
remainder into the input one to keep ``input + output == total``.
|
||||||
component so that ``input + output == total`` exactly, which means a
|
|
||||||
wrong rate on *either* side shows up as a mismatch here.
|
|
||||||
"""
|
"""
|
||||||
from ..payment.cost_calculation import CostDataError
|
from ..payment.cost_calculation import CostDataError
|
||||||
|
|
||||||
@@ -842,17 +772,9 @@ async def run_live_checks(
|
|||||||
return rows
|
return rows
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# The standalone runner certifies a URL before it is configured, so it reads
|
||||||
# Standalone runner
|
# nothing from the node's database: the pricing rows do not apply, and the cost
|
||||||
#
|
# row falls back to litellm's cost map or explicit prices.
|
||||||
# ``certify_upstream_url`` deliberately reads nothing from the node's
|
|
||||||
# database: the point of the CLI is to certify a URL *before* it is
|
|
||||||
# configured, or one the operator does not want to write into the node at
|
|
||||||
# all. The four pricing rows therefore do not apply here — they compare a
|
|
||||||
# stored row against the served map, neither of which exists for a bare
|
|
||||||
# URL — and the cost row falls back to litellm's cost map (or explicit
|
|
||||||
# prices) instead of a configured row.
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
def _first_model_id(probe: ProbeResult) -> str | None:
|
def _first_model_id(probe: ProbeResult) -> str | None:
|
||||||
@@ -870,9 +792,8 @@ def _first_model_id(probe: ProbeResult) -> str | None:
|
|||||||
def _as_price(value: Any) -> float | None:
|
def _as_price(value: Any) -> float | None:
|
||||||
"""A USD-per-token price from outside the node, or ``None``.
|
"""A USD-per-token price from outside the node, or ``None``.
|
||||||
|
|
||||||
Shares ``coerce_rate`` — the one definition of a usable rate — so an
|
Shares ``coerce_rate`` so an explicit ``--prompt-price`` is validated
|
||||||
explicit ``--prompt-price`` is validated exactly like a litellm-derived
|
exactly like a litellm-derived one.
|
||||||
one: a boolean, a negative or a non-finite value is not a price.
|
|
||||||
"""
|
"""
|
||||||
return coerce_rate(value)
|
return coerce_rate(value)
|
||||||
|
|
||||||
@@ -1036,7 +957,6 @@ async def certify_upstream_url(
|
|||||||
|
|
||||||
|
|
||||||
def render_checklist(result: dict[str, Any]) -> str:
|
def render_checklist(result: dict[str, Any]) -> str:
|
||||||
"""Render one certification result as the operator-facing checklist."""
|
|
||||||
target = result.get("target", {})
|
target = result.get("target", {})
|
||||||
lines = [f"Upstream certification — {target.get('base_url')}"]
|
lines = [f"Upstream certification — {target.get('base_url')}"]
|
||||||
if target.get("model_id"):
|
if target.get("model_id"):
|
||||||
@@ -1054,12 +974,8 @@ def render_checklist(result: dict[str, Any]) -> str:
|
|||||||
|
|
||||||
|
|
||||||
def _route_logs_to_stderr() -> None:
|
def _route_logs_to_stderr() -> None:
|
||||||
"""Move the app's stdout log handlers to stderr.
|
"""Move the app's stdout log handlers to stderr, so log records cannot
|
||||||
|
interleave with the report."""
|
||||||
``routstr.core.logging`` configures its handlers onto ``sys.stdout``, so
|
|
||||||
a machine-readable run would otherwise interleave log records with the
|
|
||||||
document. Stdout is the report's channel; logs belong on stderr.
|
|
||||||
"""
|
|
||||||
import logging
|
import logging
|
||||||
|
|
||||||
loggers = [logging.getLogger()]
|
loggers = [logging.getLogger()]
|
||||||
|
|||||||
@@ -224,7 +224,6 @@ async def test_certify_all_ok(
|
|||||||
assert "rows" in body
|
assert "rows" in body
|
||||||
assert "checklist" in body
|
assert "checklist" in body
|
||||||
|
|
||||||
# All live rows should be ok
|
|
||||||
live_row_ids = [
|
live_row_ids = [
|
||||||
"endpoint.validity",
|
"endpoint.validity",
|
||||||
"endpoint.reachable",
|
"endpoint.reachable",
|
||||||
@@ -236,7 +235,6 @@ async def test_certify_all_ok(
|
|||||||
row = _find_row(body["rows"], row_id)
|
row = _find_row(body["rows"], row_id)
|
||||||
assert row["status"] == "ok", f"{row_id}: {row}"
|
assert row["status"] == "ok", f"{row_id}: {row}"
|
||||||
|
|
||||||
# All checklist goals should be ok
|
|
||||||
for item in body["checklist"]:
|
for item in body["checklist"]:
|
||||||
assert item["status"] == "ok", f"{item['goal']}: {item}"
|
assert item["status"] == "ok", f"{item['goal']}: {item}"
|
||||||
|
|
||||||
@@ -267,7 +265,6 @@ async def test_certify_heartbeat_fail_on_500(
|
|||||||
assert row["status"] == "fail"
|
assert row["status"] == "fail"
|
||||||
assert row["evidence"]["status_code"] == 500
|
assert row["evidence"]["status_code"] == 500
|
||||||
|
|
||||||
# heartbeat goal should be fail
|
|
||||||
heartbeat_goal = next(
|
heartbeat_goal = next(
|
||||||
item for item in body["checklist"] if item["goal"] == "heartbeat"
|
item for item in body["checklist"] if item["goal"] == "heartbeat"
|
||||||
)
|
)
|
||||||
@@ -338,7 +335,6 @@ async def test_certify_usage_warn_when_no_usage(
|
|||||||
respx.get("https://certify-upstream.example/v1/models").mock(
|
respx.get("https://certify-upstream.example/v1/models").mock(
|
||||||
return_value=Response(200, json=_mock_models_response())
|
return_value=Response(200, json=_mock_models_response())
|
||||||
)
|
)
|
||||||
# No "usage" key in the chat response
|
|
||||||
respx.post("https://certify-upstream.example/v1/chat/completions").mock(
|
respx.post("https://certify-upstream.example/v1/chat/completions").mock(
|
||||||
return_value=Response(
|
return_value=Response(
|
||||||
200,
|
200,
|
||||||
@@ -494,7 +490,6 @@ async def test_certify_with_no_served_model(
|
|||||||
)
|
)
|
||||||
assert resp.status_code == 200, resp.text
|
assert resp.status_code == 200, resp.text
|
||||||
body = resp.json()
|
body = resp.json()
|
||||||
# Live rows should be warn (skipped)
|
|
||||||
for row_id in ["endpoint.reachable", "usage.capture", "cost.prompt_completion"]:
|
for row_id in ["endpoint.reachable", "usage.capture", "cost.prompt_completion"]:
|
||||||
row = _find_row(body["rows"], row_id)
|
row = _find_row(body["rows"], row_id)
|
||||||
assert row["status"] == "warn", f"{row_id}: {row}"
|
assert row["status"] == "warn", f"{row_id}: {row}"
|
||||||
|
|||||||
@@ -50,13 +50,10 @@ def _probe(**kwargs: Any) -> ProbeResult:
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# Regression: a non-finite token count crashed the billing path.
|
||||||
# Defect: a non-finite token count crashed the billing path.
|
|
||||||
#
|
|
||||||
# ``json.loads`` accepts the bare ``Infinity``/``NaN`` literals, so an
|
# ``json.loads`` accepts the bare ``Infinity``/``NaN`` literals, so an
|
||||||
# upstream can put them on the wire; ``int(inf)`` raised OverflowError and
|
# upstream can put them on the wire; ``int(inf)`` raised OverflowError and
|
||||||
# ``int(nan)`` raised ValueError inside ``parse_token_count``.
|
# ``int(nan)`` raised ValueError inside ``parse_token_count``.
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
class TestNonFiniteTokenCounts:
|
class TestNonFiniteTokenCounts:
|
||||||
@@ -111,10 +108,8 @@ class TestNonFiniteTokenCounts:
|
|||||||
assert row["status"] == STATUS_WARN
|
assert row["status"] == STATUS_WARN
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# Regression: ``certification_row`` stored non-dict evidence verbatim, so the
|
||||||
# Defect: ``certification_row`` stored non-dict evidence verbatim, so the
|
|
||||||
# row contract ("evidence is always a dict") held only by caller discipline.
|
# row contract ("evidence is always a dict") held only by caller discipline.
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
class TestEvidenceContract:
|
class TestEvidenceContract:
|
||||||
@@ -128,10 +123,8 @@ class TestEvidenceContract:
|
|||||||
assert row["evidence"] == {"a": 1}
|
assert row["evidence"] == {"a": 1}
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# Regression: ``http://:8080/v1`` was certified as a valid endpoint because
|
||||||
# Defect: ``http://:8080/v1`` was certified as a valid endpoint because
|
|
||||||
# ``netloc`` is truthy for a hostless authority.
|
# ``netloc`` is truthy for a hostless authority.
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
class TestEndpointValidity:
|
class TestEndpointValidity:
|
||||||
@@ -157,11 +150,9 @@ class TestEndpointValidity:
|
|||||||
assert row["status"] == STATUS_OK, url
|
assert row["status"] == STATUS_OK, url
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# Regression: the payload builders called ``.get()`` on whatever they were
|
||||||
# Defect: the payload builders called ``.get()`` on whatever they were
|
|
||||||
# given, so a wrong-typed body raised AttributeError instead of producing a
|
# given, so a wrong-typed body raised AttributeError instead of producing a
|
||||||
# verdict.
|
# verdict.
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
class TestPayloadTypeGuards:
|
class TestPayloadTypeGuards:
|
||||||
@@ -184,10 +175,8 @@ class TestPayloadTypeGuards:
|
|||||||
assert row["evidence"]["usable_ids"] == 0
|
assert row["evidence"]["usable_ids"] == 0
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# Regression: an empty-string id was counted as "usable" by the payload row but
|
||||||
# Defect: an empty-string id was counted as "usable" by the payload row but
|
|
||||||
# rejected by the CLI's discovery path — the two disagreed on one response.
|
# rejected by the CLI's discovery path — the two disagreed on one response.
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
class TestModelIdAgreement:
|
class TestModelIdAgreement:
|
||||||
@@ -204,10 +193,8 @@ class TestModelIdAgreement:
|
|||||||
assert row["evidence"]["usable_ids"] == 1
|
assert row["evidence"]["usable_ids"] == 1
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# Regression: the independent cost re-derivation disagreed with the engine on
|
||||||
# Defect: the independent cost re-derivation disagreed with the engine on
|
|
||||||
# coercion (numeric strings, booleans), manufacturing false failures.
|
# coercion (numeric strings, booleans), manufacturing false failures.
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
class TestReportedCostCoercionParity:
|
class TestReportedCostCoercionParity:
|
||||||
@@ -230,10 +217,8 @@ class TestReportedCostCoercionParity:
|
|||||||
assert _reported_usd_cost(payload) == pytest.approx(0.001)
|
assert _reported_usd_cost(payload) == pytest.approx(0.001)
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# Regression: ``_expected_token_msats`` ran ``math.ceil`` on a non-finite sum,
|
||||||
# Defect: ``_expected_token_msats`` ran ``math.ceil`` on a non-finite sum,
|
|
||||||
# raising an opaque error instead of a describable one.
|
# raising an opaque error instead of a describable one.
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
class TestNonFinitePricing:
|
class TestNonFinitePricing:
|
||||||
@@ -267,9 +252,7 @@ class TestNonFinitePricing:
|
|||||||
assert total == inp + outp
|
assert total == inp + outp
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# Regression: a row builder raising escaped as a 500 from the admin endpoint.
|
||||||
# Defect: a row builder raising escaped as a 500 from the admin endpoint.
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
class TestSafeRow:
|
class TestSafeRow:
|
||||||
@@ -289,10 +272,8 @@ class TestSafeRow:
|
|||||||
assert row["status"] == STATUS_OK
|
assert row["status"] == STATUS_OK
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# Regression: explicit ``--prompt-price`` bypassed validation, so a negative
|
||||||
# Defect: explicit ``--prompt-price`` bypassed validation, so a negative
|
|
||||||
# rate could be fed into the cost engine.
|
# rate could be fed into the cost engine.
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
class TestExplicitPriceValidation:
|
class TestExplicitPriceValidation:
|
||||||
@@ -314,12 +295,10 @@ class TestExplicitPriceValidation:
|
|||||||
assert _as_price("1e-7") == pytest.approx(1e-7)
|
assert _as_price("1e-7") == pytest.approx(1e-7)
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# Regression: the standalone CLI was dead on arrival — ``sats_usd_price()``
|
||||||
# Defect: the standalone CLI was dead on arrival — ``sats_usd_price()``
|
|
||||||
# raises in a fresh process because the module global is only populated by
|
# raises in a fresh process because the module global is only populated by
|
||||||
# the app's lifespan task. These run the CLI as a subprocess so the fresh
|
# the app's lifespan task. These run the CLI as a subprocess so the fresh
|
||||||
# process is the thing under test.
|
# process is the thing under test.
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
|
|
||||||
|
|
||||||
def _run_cli(*args: str, timeout: float = 90.0) -> subprocess.CompletedProcess[str]:
|
def _run_cli(*args: str, timeout: float = 90.0) -> subprocess.CompletedProcess[str]:
|
||||||
|
|||||||
Reference in New Issue
Block a user