fin-hub/tests/test_ingest.py
tegwick 09ef7fa967 feat: bound AI-plan work to resource-control and stop zeroing missing tokens
Record the FIN-WP-0007 split after checking resource-control: they keep
intelligence resource identity and provision-level class I metering;
session tokens stay with State Hub; work-effectiveness is a reporting
join. Missing token counts are now unknown, not zero.
2026-08-15 19:01:45 +02:00

94 lines
3.1 KiB
Python

from pathlib import Path
import pytest
from fin_hub.ingest.anthropic import parse_anthropic_billing_csv, parse_optional_token_count
from fin_hub.ingest.cloud import parse_cloud_cost_csv
from fin_hub.ingest.hosteurope import parse_hosteurope_csv
FIXTURES = Path(__file__).parent / "fixtures"
def test_parse_cloud_cost_csv():
rows = parse_cloud_cost_csv(FIXTURES / "cloud-costs.csv")
assert len(rows) == 3
assert rows[0].service == "state-hub"
assert rows[0].amount == 42.5
def test_parse_anthropic_billing_csv():
rows = parse_anthropic_billing_csv(FIXTURES / "anthropic-billing.csv")
assert len(rows) == 2
assert rows[0].provider == "anthropic"
assert rows[0].tokens_in == 12000
assert rows[0].tokens_out == 3000
def test_parse_optional_token_count_treats_empty_as_unknown():
assert parse_optional_token_count("") is None
assert parse_optional_token_count("0") == 0
assert parse_optional_token_count("12") == 12
with pytest.raises(ValueError, match="integer"):
parse_optional_token_count("n/a")
with pytest.raises(ValueError, match="negative"):
parse_optional_token_count("-1")
def test_parse_anthropic_cost_only_row_does_not_invent_zero_tokens(tmp_path: Path):
source = tmp_path / "anthropic-cost-only.csv"
source.write_text(
"model,cost,currency,usage_date\n"
"claude-opus-4-20250514,20.00,USD,2026-08-01\n",
encoding="utf-8",
)
row = parse_anthropic_billing_csv(source)[0]
assert row.cost == 20.0
assert row.tokens_in is None
assert row.tokens_out is None
def test_parse_anthropic_explicit_zero_tokens_remain_zero(tmp_path: Path):
source = tmp_path / "anthropic-zero-tokens.csv"
source.write_text(
"model,input_tokens,output_tokens,cost,currency,usage_date\n"
"claude-haiku-3-20240307,0,0,0.00,USD,2026-08-02\n",
encoding="utf-8",
)
row = parse_anthropic_billing_csv(source)[0]
assert row.tokens_in == 0
assert row.tokens_out == 0
assert row.cost == 0.0
def test_parse_hosteurope_csv():
rows = parse_hosteurope_csv(FIXTURES / "hosteurope.csv")
assert len(rows) == 2
assert rows[0].service_id == "dedicated-server-m"
assert rows[0].period_month == "2026-06"
def test_parse_hosteurope_client_attribution(tmp_path: Path):
source = tmp_path / "attributed.csv"
source.write_text(
"product,amount,currency,invoice_date,client_id,application_id,app_instance_id\n"
"Managed cluster,42.00,EUR,2026-07-01,acme,portal,prod-01\n",
encoding="utf-8",
)
row = parse_hosteurope_csv(source)[0]
assert row.client_id == "acme"
assert row.application_id == "portal"
assert row.app_instance_id == "prod-01"
assert row.cost_attribution_key == "client:acme|app:portal|instance:prod-01"
def test_parse_hosteurope_rejects_partial_attribution(tmp_path: Path):
source = tmp_path / "partial.csv"
source.write_text(
"product,amount,currency,invoice_date,client_id\n"
"Managed cluster,42.00,EUR,2026-07-01,acme\n",
encoding="utf-8",
)
with pytest.raises(ValueError, match="must be supplied together"):
parse_hosteurope_csv(source)