Production Login page passcode updates

This commit is contained in:
sriram
2026-08-24 22:55:06 +05:30
parent 2482fe43e3
commit 1b347f91db
11 changed files with 878 additions and 67 deletions

View File

@@ -34,6 +34,8 @@ from app.infrastructure.security import (
ROLE_PERMISSIONS,
Principal,
create_access_token,
hash_is_wellformed,
password_hash_fingerprint,
verify_password,
)
from app.infrastructure.settings import (
@@ -45,6 +47,7 @@ from app.infrastructure.settings import (
AUTH_MAX_LOGIN_ATTEMPTS,
AUTH_USER_PASSWORD_HASH,
AUTH_USER_USERNAME,
config_source,
)
logger = logging.getLogger(__name__)
@@ -216,12 +219,67 @@ def login(payload: LoginRequest, request: Request) -> LoginResponse:
# username and a bad password take the same time. Otherwise the response
# latency alone enumerates valid usernames.
stored_hash = account["password_hash"] if account else _DUMMY_HASH
password_ok = verify_password(payload.password, stored_hash)
# A hash that does not parse can never match, and verify_password bails
# out of one before doing any PBKDF2 work - measured here, 0.16ms against
# 439ms for a real digest. That inverts the very property _DUMMY_HASH
# exists to protect: an account whose configured hash is corrupt would
# answer ~2700x faster than every other username, announcing which
# account is broken to anyone with a stopwatch. So spend the same work
# regardless; the result is a rejection either way.
hash_usable = hash_is_wellformed(stored_hash)
password_ok = verify_password(
payload.password, stored_hash if hash_usable else _DUMMY_HASH
)
if account is None or not password_ok:
_record_failure(key)
logger.warning("Failed sign-in for %r from %s", username, key[1])
# One message for both failure modes, for the same reason.
# The reason goes to the LOG, never to the caller - the response
# below is byte-identical whichever of these it was, so nothing here
# can be used to enumerate usernames. It is computed after both the
# lookup and the PBKDF2 call above, so it adds no timing signal
# either. Without it, a deployment whose configured hash or admin
# username has drifted is indistinguishable from someone simply
# typing the wrong password, and this is exactly how a production
# sign-in outage stayed unexplained: the log said "Failed sign-in
# for 'admin'" and nothing more.
if account is None:
logger.warning(
"Failed sign-in for %r from %s: reason=unknown-username. "
"Configured accounts: %s (AUTH_ADMIN_USERNAME source=%s).",
username,
key[1],
", ".join(sorted(_accounts())),
config_source("AUTH_ADMIN_USERNAME"),
)
elif not hash_usable:
# ERROR, not WARNING: this is a broken deployment, not a bad
# guess. No password can ever match, so every sign-in to this
# account will 401 until the hash itself is replaced.
logger.error(
"Failed sign-in for %r from %s: reason=malformed-hash. The configured "
"password hash does not parse as pbkdf2_sha256$<iterations>$<b64 salt>$"
"<b64 digest> (fingerprint=%s, source=%s). Nobody can sign in to this "
"account until it is regenerated with scripts/make_auth_secrets.py.",
username,
key[1],
password_hash_fingerprint(stored_hash) or "(empty)",
config_source("AUTH_ADMIN_PASSWORD_HASH"),
)
else:
logger.warning(
"Failed sign-in for %r from %s: reason=bad-password. The account exists "
"and its hash parses (fingerprint=%s, source=%s); the password did not "
"match. If this IS the password you deployed, then the running config "
"carries a different hash than the file you are reading - compare that "
"fingerprint against: python scripts/make_auth_secrets.py "
"--fingerprint .env.production",
username,
key[1],
password_hash_fingerprint(stored_hash),
config_source("AUTH_ADMIN_PASSWORD_HASH"),
)
# One message for every failure mode, for the same reason.
raise HTTPException(
status_code=status.HTTP_401_UNAUTHORIZED,
detail="Invalid username or password.",

View File

@@ -5,7 +5,8 @@ import logging
import requests
from fastapi import APIRouter
from app.api.schemas import HealthOut
from app.api.schemas import AuthConfigOut, HealthOut
from app.infrastructure.security import auth_config_summary
from app.infrastructure.settings import OLLAMA_BASE_URL, OLLAMA_MODEL_NAME, EMBEDDINGS_MODEL
from app.services.vector_store import _connect # internal, but handy for a connectivity probe
@@ -35,7 +36,17 @@ def _check_ollama() -> bool:
@router.get("/health", response_model=HealthOut)
def health() -> HealthOut:
"""Liveness/readiness probe used by the React app to show a banner when
Postgres or Ollama aren't reachable, instead of failing silently."""
Postgres or Ollama aren't reachable, instead of failing silently.
The `auth` block serves the same purpose for credentials that `database`
does for Postgres: it makes a misconfiguration visible from outside the
container. See AuthConfigOut for why it is not behind a token - a
diagnostic for "nobody can sign in" cannot itself require signing in.
`status` deliberately does NOT go degraded on an auth problem: this
endpoint gates container routing in some deployments, and taking a
perfectly serving process out of rotation over a credential mismatch would
replace a login failure with an outage."""
db_ok = _check_database()
ollama_ok = _check_ollama()
return HealthOut(
@@ -44,4 +55,5 @@ def health() -> HealthOut:
ollama=ollama_ok,
ollama_model=OLLAMA_MODEL_NAME,
embeddings_model=EMBEDDINGS_MODEL,
auth=AuthConfigOut(**auth_config_summary()),
)