Files
loyaly-catalogue/backend/tests/test_auth_diagnostics.py
sriram c7e4d59188 Electronics Catalog: API, MCP server, frontend and deployment
Verified catalogue of mobiles and laptops sold in India, collected from
real retail listings (FastAPI backend, React frontend, Postgres/pgvector).

- REST API under /api/elec (read-only catalogue; admin endpoints need login)
- MCP server (FastMCP) at /mcp/ with list_categories, search_products,
  get_product and price_history tools
- Real ratings and reviews read from product pages and search results
- Production Dockerfile (requirements-api.txt, no PyTorch) and
  .env.production.example; remote database only via an explicit
  ELEC_ALLOW_REMOTE_DB host/name allowlist
- docs/API.md: endpoint and MCP reference with live examples

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
2026-10-01 12:17:42 +05:30

330 lines
14 KiB
Python

"""
The credential-diagnostics surface: hash fingerprints, config provenance, and
the `auth` block on /api/health.
These exist because of a real incident. Production rejected the correct admin
password while localhost accepted it, and every observable said the app was
healthy: /api/health was 200, CORS passed, the route table was current, and the
only log line was `Failed sign-in for 'admin'` - which is what a user with caps
lock on produces too. Nothing distinguished "wrong password" from "this image
was built from a different .env.production", so there was no way to tell which
of them it was without a shell on the box.
What is pinned here is therefore not a feature so much as the ability to answer
one question from outside a container: *is this deployment running the
credential I think it is?* The fingerprint is the answer, and these tests hold
it to the two properties that make it usable - it identifies a hash, and it
discloses nothing about the password behind it.
"""
from __future__ import annotations
import pytest
from app.infrastructure.security import (
api_key_fingerprint,
auth_config_summary,
describe_api_keys,
describe_password_hash,
hash_is_wellformed,
hash_password,
password_hash_fingerprint,
)
from app.infrastructure.settings import API_KEY_MIN_LENGTH, _parse_api_keys, config_source
from tests.conftest import TEST_ADMIN_PASSWORD
# ---------------------------------------------------------------------------
# Fingerprint
# ---------------------------------------------------------------------------
def test_fingerprint_is_stable_for_a_given_hash():
"""Comparing prod against local is the whole point, so the same input must
give the same answer on both machines and across runs."""
encoded = hash_password("whatever", iterations=1000)
assert password_hash_fingerprint(encoded) == password_hash_fingerprint(encoded)
def test_fingerprint_differs_when_the_hash_does():
"""Including for the same password: two deployments that hashed the same
password separately are NOT running the same credential, and a fingerprint
that hid that would defeat the comparison."""
a = hash_password("same-password", iterations=1000)
b = hash_password("same-password", iterations=1000)
assert a != b, "salts must differ"
assert password_hash_fingerprint(a) != password_hash_fingerprint(b)
@pytest.mark.parametrize("wrapper", ['"{}"', "'{}'", " {} ", "{}\r", "\n{}\n"])
def test_fingerprint_ignores_quotes_and_whitespace(wrapper):
"""A hash pasted into a platform's Environment tab arrives wrapped. It is
the same credential, so it must fingerprint the same - otherwise the
comparison reports a spurious mismatch in exactly the case it exists for."""
encoded = hash_password("p", iterations=1000)
assert password_hash_fingerprint(wrapper.format(encoded)) == password_hash_fingerprint(
encoded
)
def test_fingerprint_discloses_no_part_of_the_hash():
"""It is served unauthenticated, so it must be a digest OF the credential
and not a piece of it."""
encoded = hash_password("p", iterations=1000)
fp = password_hash_fingerprint(encoded)
assert len(fp) == 12
assert all(c in "0123456789abcdef" for c in fp)
assert fp not in encoded
# Nor any run of it long enough to be a foothold into salt or digest.
for start in range(len(fp) - 5):
assert fp[start : start + 6] not in encoded
def test_absent_hash_fingerprints_as_empty():
assert password_hash_fingerprint("") == ""
# ---------------------------------------------------------------------------
# describe_password_hash / hash_is_wellformed
# ---------------------------------------------------------------------------
@pytest.mark.parametrize(
"bad", ["", "not-a-hash", "pbkdf2_sha256$notanint$a$b", "a$b$c$d", "bcrypt$1$a$b"]
)
def test_a_malformed_hash_is_reported_invalid(bad):
"""Same inputs as test_malformed_hash_fails_closed, held against the shared
parser - the two must agree on what 'unusable' means, since one decides the
login and the other decides what the log calls it."""
assert hash_is_wellformed(bad) is False
assert describe_password_hash(bad)["valid"] is False
def test_a_real_hash_is_reported_valid_with_its_iteration_count():
described = describe_password_hash(hash_password("p", iterations=4321))
assert described["valid"] is True
assert described["iterations"] == 4321
assert described["algorithm"] == "pbkdf2_sha256"
def test_describe_never_returns_the_hash_itself():
encoded = hash_password("p", iterations=1000)
assert encoded not in str(describe_password_hash(encoded))
# ---------------------------------------------------------------------------
# Config provenance
# ---------------------------------------------------------------------------
def test_config_source_reports_process_env_for_harness_supplied_values():
"""conftest writes the AUTH_* values into os.environ before app.main is
imported - which is structurally the same thing a deployment platform's
Environment tab does. That this reads back as 'process-env' is the
executable proof that an override is detectable at all."""
assert config_source("AUTH_ADMIN_PASSWORD_HASH") == "process-env"
assert config_source("AUTH_ADMIN_USERNAME") == "process-env"
def test_config_source_reports_default_for_something_never_set():
assert config_source("AUTH_NOT_A_REAL_SETTING_XYZ") == "default"
# ---------------------------------------------------------------------------
# /api/health
# ---------------------------------------------------------------------------
def test_health_reports_the_effective_auth_configuration(client):
auth = client.get("/api/health").json()["auth"]
assert auth["enabled"] is True
assert auth["allow_any_login"] is False
assert auth["admin_username"] == "admin"
assert auth["password_hash_valid"] is True
assert auth["password_hash_iterations"] == 20_000 # conftest._hash
assert auth["password_hash_fingerprint"] == auth_config_summary()[
"password_hash_fingerprint"
]
assert auth["password_hash_source"] == "process-env"
def test_health_never_exposes_a_hash_or_a_password(client):
"""The leak canary on an unauthenticated endpoint. A configured digest
always contains '$' separators; a password would appear verbatim."""
body = client.get("/api/health").text
assert TEST_ADMIN_PASSWORD not in body
assert "pbkdf2_sha256$" not in body
assert "$" not in body
def test_health_stays_ok_shaped_when_auth_is_misconfigured(client, monkeypatch):
"""An unusable credential must NOT flip `status` to degraded: the container
healthcheck and the frontend's connectivity banner both read that field, so
doing so would turn a login problem into an outage and a misleading "database
unreachable" banner. The signal belongs in auth.password_hash_valid."""
from app.api.routers import health as health_router
monkeypatch.setattr(
health_router, "auth_config_summary", lambda: {**auth_config_summary(),
"password_hash_valid": False}
)
body = client.get("/api/health").json()
assert body["auth"]["password_hash_valid"] is False
assert body["status"] in {"ok", "degraded"} # decided by db/ollama only
# ---------------------------------------------------------------------------
# API keys
# ---------------------------------------------------------------------------
# Same incident, one layer out. A key added to .env.production and then merely
# restarted into a running container is absent from the process, because the
# Dockerfile copies that file in at BUILD time - and from outside, an undeployed
# key and a wrong key are both just a 401. These pin the ability to tell them
# apart without anyone sending the secret to find out.
_GOOD_SECRET = "cs3JwvApS5Je_Qfe1sNYq6YtUBDDeqp4OEgy2_41sQg"
def test_api_key_fingerprint_is_stable_and_hex():
fp = api_key_fingerprint("partner", _GOOD_SECRET)
assert fp == api_key_fingerprint("partner", _GOOD_SECRET)
assert len(fp) == 12
assert all(c in "0123456789abcdef" for c in fp)
def test_api_key_fingerprint_differs_when_the_secret_does():
assert api_key_fingerprint("partner", _GOOD_SECRET) != api_key_fingerprint(
"partner", _GOOD_SECRET[:-1] + "X"
)
def test_api_key_fingerprint_separates_consumers_sharing_a_secret():
"""The name is mixed in, so two consumers mistakenly issued the same secret
do not report the same fingerprint - which would hide the mistake behind the
very field meant to reveal it."""
assert api_key_fingerprint("console-a", _GOOD_SECRET) != api_key_fingerprint(
"console-b", _GOOD_SECRET
)
@pytest.mark.parametrize("wrapper", ["{}", "'{}'", '"{}"', " {} "])
def test_api_key_fingerprint_ignores_quotes_and_whitespace(wrapper):
"""A value pasted into a deployment platform's Environment tab arrives
wrapped often enough that settings strips it; the fingerprint must agree,
or comparing two ends reports a mismatch that is not real."""
assert api_key_fingerprint("partner", wrapper.format(_GOOD_SECRET)) == (
api_key_fingerprint("partner", _GOOD_SECRET)
)
def test_api_key_fingerprint_discloses_no_part_of_the_secret():
"""Served unauthenticated, so it must be a digest OF the key, not a piece."""
fp = api_key_fingerprint("partner", _GOOD_SECRET)
assert fp not in _GOOD_SECRET
for start in range(len(fp) - 5):
assert fp[start : start + 6] not in _GOOD_SECRET
def test_absent_secret_fingerprints_as_empty():
assert api_key_fingerprint("partner", "") == ""
def test_describe_api_keys_is_sorted_by_name(monkeypatch):
"""API_KEYS is keyed by secret, whose order says nothing. Sorting is what
lets two deployments' output be diffed line for line."""
from app.infrastructure import security
monkeypatch.setattr(
security, "API_KEYS",
{_GOOD_SECRET: ("zulu", "user"), _GOOD_SECRET[::-1]: ("alpha", "admin")},
)
assert [k["name"] for k in describe_api_keys()] == ["alpha", "zulu"]
assert [k["role"] for k in describe_api_keys()] == ["admin", "user"]
# ---------------------------------------------------------------------------
# API keys on /api/health
# ---------------------------------------------------------------------------
def test_health_reports_the_keys_this_deployment_actually_loaded(client):
"""Against the harness's own API_KEYS, not a monkeypatched one - this is the
end-to-end wiring from settings through the summary to the response body."""
from tests.conftest import TEST_API_KEY
auth = client.get("/api/health").json()["auth"]
assert auth["api_keys_count"] == 1
assert auth["api_keys"] == [{
"name": "test-machine",
"role": "user",
"fingerprint": api_key_fingerprint("test-machine", TEST_API_KEY),
}]
def test_health_reports_an_empty_list_when_no_keys_are_configured(client, monkeypatch):
"""The state production was in while the colleague's console got 401s: auth
enabled, admin login working, and not one machine consumer deployed."""
from app.infrastructure import security
monkeypatch.setattr(security, "API_KEYS", {})
auth = client.get("/api/health").json()["auth"]
assert auth["api_keys_count"] == 0
assert auth["api_keys"] == []
def test_health_names_configured_keys_and_fingerprints_them(client, monkeypatch):
from app.infrastructure import security
monkeypatch.setattr(security, "API_KEYS", {_GOOD_SECRET: ("colleague-console", "admin")})
auth = client.get("/api/health").json()["auth"]
assert auth["api_keys_count"] == 1
assert auth["api_keys"] == [{
"name": "colleague-console",
"role": "admin",
"fingerprint": api_key_fingerprint("colleague-console", _GOOD_SECRET),
}]
def test_health_never_exposes_an_api_key_secret(client, monkeypatch):
"""The leak canary, extended to machine credentials."""
from app.infrastructure import security
monkeypatch.setattr(security, "API_KEYS", {_GOOD_SECRET: ("colleague-console", "admin")})
body = client.get("/api/health").text
assert _GOOD_SECRET not in body
for start in range(0, len(_GOOD_SECRET) - 7):
assert _GOOD_SECRET[start : start + 8] not in body
def test_health_reports_where_the_keys_came_from(client):
"""Which of the two config sources won. Unlike the admin hash - where
"process-env" flags a stale Environment tab shadowing the image - API_KEYS is
deliberately supplied by that tab, so "process-env" is the expected value in
production and "env-file" would mean the tab entry has gone missing."""
auth = client.get("/api/health").json()["auth"]
assert auth["api_keys_source"] in {"process-env", "env-file", "default"}
# ---------------------------------------------------------------------------
# Parsing
# ---------------------------------------------------------------------------
def test_parse_api_keys_accepts_a_generated_secret():
parsed = _parse_api_keys(f"partner:admin:{_GOOD_SECRET}")
assert parsed == {_GOOD_SECRET: ("partner", "admin")}
def test_parse_api_keys_rejects_a_secret_too_short_to_fingerprint_safely():
"""A raw key carries no salt, so publishing its digest is only safe while the
key itself is unguessable offline. A hand-picked one must be refused at
startup rather than quietly fingerprinted onto a public endpoint."""
with pytest.raises(RuntimeError, match="at least"):
_parse_api_keys("partner:admin:changeme")
def test_parse_api_keys_length_limit_admits_the_documented_generator():
import secrets as _secrets
assert len(_secrets.token_urlsafe(32)) >= API_KEY_MIN_LENGTH