Production Login page passcode updates

This commit is contained in:
sriram
2026-08-24 22:55:06 +05:30
parent 2482fe43e3
commit 1b347f91db
11 changed files with 878 additions and 67 deletions

View File

@@ -29,15 +29,19 @@ import logging
import secrets
import time
from dataclasses import dataclass, field
from typing import Dict, List, Optional
from typing import Dict, List, Optional, Tuple
import jwt
from app.infrastructure.settings import (
API_KEYS,
AUTH_ADMIN_PASSWORD_HASH,
AUTH_ADMIN_USERNAME,
AUTH_ALLOW_ANY_LOGIN,
AUTH_ENABLED,
AUTH_SECRET_KEY,
AUTH_TOKEN_TTL_MINUTES,
config_source,
)
logger = logging.getLogger(__name__)
@@ -117,6 +121,110 @@ def hash_password(password: str, *, iterations: int = _PBKDF2_ITERATIONS) -> str
)
def _parse_encoded_hash(encoded: str) -> Optional[Tuple[bytes, bytes, int]]:
"""
Split an encoded digest into ``(salt, digest, iterations)``, or None if it
is not one.
One parser, three callers. `verify_password` needs the parts, while
`hash_is_wellformed` and `describe_password_hash` need only the verdict -
and a login failing because the *configured* hash is corrupt is a different
incident from a wrong password, so the two must agree on what "corrupt"
means. Two copies of this parse would eventually disagree.
Values arrive here straight from the environment, so a hash pasted into a
deployment platform's form field as "pbkdf2_sha256$..." is unwrapped rather
than rejected: the surrounding quotes are almost never intended as part of
the secret, and the failure they cause otherwise is a silent 401.
"""
if not encoded:
return None
encoded = encoded.strip().strip("'\"")
try:
prefix, raw_iterations, raw_salt, raw_digest = encoded.split("$")
if prefix != _PBKDF2_PREFIX:
return None
# validate=True so junk is rejected rather than silently discarded:
# b64decode's default drops non-alphabet characters, which would let a
# subtly corrupted hash decode to the wrong bytes and fail as a "wrong
# password" instead of as the configuration error it is.
# binascii.Error subclasses ValueError, so it is caught below.
salt = base64.b64decode(raw_salt, validate=True)
digest = base64.b64decode(raw_digest, validate=True)
iterations = int(raw_iterations)
except (ValueError, TypeError):
return None
# A structurally valid string that decodes to nothing is still unusable,
# and PBKDF2 rejects a non-positive iteration count by raising.
if not salt or not digest or iterations < 1:
return None
return salt, digest, iterations
def hash_is_wellformed(encoded: str) -> bool:
"""Whether a configured digest can be checked against at all.
Distinct from "does the password match": this asks whether the credential
*store* is usable, which is a deployment fault rather than a sign-in one.
"""
return _parse_encoded_hash(encoded) is not None
def password_hash_fingerprint(encoded: str) -> str:
"""
A short, non-reversible identifier for a configured digest.
Safe to log and to publish: it is a truncated SHA-256 of the *encoded
digest*, and that digest already embeds a 16-byte random salt, so this says
which credential is loaded without saying anything about the password
behind it. It exists so a running deployment can be compared against the
config it was supposed to have been built from - the failure this project
actually hit - without moving a secret in order to do the comparison.
"""
if not encoded:
return ""
return hashlib.sha256(encoded.strip().strip("'\"").encode("utf-8")).hexdigest()[:12]
def describe_password_hash(encoded: str) -> Dict[str, object]:
"""A loggable/publishable summary of a configured digest. Never its bytes."""
parsed = _parse_encoded_hash(encoded)
return {
"valid": parsed is not None,
"algorithm": _PBKDF2_PREFIX if parsed is not None else None,
"iterations": parsed[2] if parsed is not None else None,
"fingerprint": password_hash_fingerprint(encoded),
}
def auth_config_summary() -> Dict[str, object]:
"""
The effective authentication configuration, in a form safe to both log and
publish. Contains no password and no hash - only the fingerprint.
This is deliberately one function with two callers (the startup log in
app/main.py and GET /api/health), because its entire purpose is letting two
*deployments* be compared, and that only works if both report the same
fields computed the same way.
`*_source` is the field that earns this its keep. A value of "process-env"
means the container's own environment supplied it and the .env file baked
into the image was ignored - which is invisible from anywhere else, and is
precisely how a corrected credential can keep failing after a redeploy.
"""
described = describe_password_hash(AUTH_ADMIN_PASSWORD_HASH)
return {
"enabled": AUTH_ENABLED,
"allow_any_login": AUTH_ALLOW_ANY_LOGIN,
"admin_username": AUTH_ADMIN_USERNAME,
"password_hash_valid": bool(described["valid"]),
"password_hash_iterations": described["iterations"],
"password_hash_fingerprint": described["fingerprint"],
"admin_username_source": config_source("AUTH_ADMIN_USERNAME"),
"password_hash_source": config_source("AUTH_ADMIN_PASSWORD_HASH"),
}
def verify_password(password: str, encoded: str) -> bool:
"""
Check a password against an encoded digest.
@@ -127,21 +235,15 @@ def verify_password(password: str, encoded: str) -> bool:
"""
if not encoded:
return False
encoded = encoded.strip().strip("'\"")
try:
prefix, raw_iterations, raw_salt, raw_digest = encoded.split("$")
if prefix != _PBKDF2_PREFIX:
return False
expected = base64.b64decode(raw_salt), base64.b64decode(raw_digest)
salt, digest = expected
iterations = int(raw_iterations)
except (ValueError, TypeError):
parsed = _parse_encoded_hash(encoded)
if parsed is None:
logger.error(
"A configured password hash is malformed and cannot be used. Regenerate "
"it with: python scripts/make_auth_secrets.py"
)
return False
salt, digest, iterations = parsed
candidate = hashlib.pbkdf2_hmac("sha256", password.encode("utf-8"), salt, iterations)
return hmac.compare_digest(candidate, digest)

View File

@@ -27,6 +27,17 @@ from __future__ import annotations
import os
from pathlib import Path
# Snapshotted BEFORE load_dotenv, and that ordering is the entire point.
# load_dotenv() is called without override=True, so a variable already in the
# process environment silently beats the .env file and keeps beating it no
# matter how many times the file is corrected. That is not hypothetical here:
# the deployment platform injects its Environment tab into the container, so a
# stale value left in that tab overrides the credentials baked into the image
# (backend/Dockerfile copies .env.production to /app/.env) and the only symptom
# is a 401 that nothing explains. Comparing a name against this set answers
# "which of the two won?" - see config_source() below.
_PREEXISTING_ENV = frozenset(os.environ)
try:
from dotenv import load_dotenv
@@ -39,6 +50,45 @@ except ImportError:
pass
# Names whose raw value arrived wrapped in quotes or padded with whitespace.
# Recorded rather than merely fixed: stripping keeps the login working, but the
# only place the original shape is still visible is right here, before the value
# is normalised. A quoted hash is the signature of a value pasted into a web
# form, so surfacing it at startup is what stops the next person rediscovering
# it from a 401. See DB_PASSWORD in .env.production for the counter-case where
# the quotes ARE part of the secret - which is why this warns, and does not fail.
_ENV_NEEDED_CLEANUP = set()
def _clean(name: str, raw: str) -> str:
"""Strip surrounding quotes/whitespace off an env value, remembering if it mattered."""
cleaned = raw.strip().strip("'\"")
if cleaned != raw:
_ENV_NEEDED_CLEANUP.add(name)
return cleaned
def cleaned_env_names() -> list:
"""Which settings needed quote/whitespace stripping. Reported at startup."""
return sorted(_ENV_NEEDED_CLEANUP)
def config_source(name: str) -> str:
"""
Where a setting's value actually came from: the process environment, the
.env file, or this module's own default.
Reported at startup for the AUTH_* values (see app/main.py) so that an
override arriving from outside the image is visible in the logs instead of
being inferred from a failing login.
"""
if name in _PREEXISTING_ENV:
return "process-env"
if name in os.environ:
return "env-file"
return "default"
def _bool(name: str, default: str) -> bool:
return os.getenv(name, default).strip().lower() in {"1", "true", "yes"}
@@ -290,20 +340,29 @@ AUTH_TOKEN_TTL_MINUTES = int(os.getenv("AUTH_TOKEN_TTL_MINUTES", "720"))
# `make_auth_secrets.py` prints the lines ready to paste.
#
# `admin` is required whenever auth is on: without it nobody could sign in.
AUTH_ADMIN_USERNAME = os.getenv("AUTH_ADMIN_USERNAME", "admin").strip().strip("'\"")
AUTH_ADMIN_PASSWORD_HASH = (
_require("AUTH_ADMIN_PASSWORD_HASH", feature_flag="AUTH_ENABLED")
if AUTH_ENABLED
else os.getenv("AUTH_ADMIN_PASSWORD_HASH", "")
).strip().strip("'\"")
AUTH_ADMIN_USERNAME = _clean(
"AUTH_ADMIN_USERNAME", os.getenv("AUTH_ADMIN_USERNAME", "admin")
)
AUTH_ADMIN_PASSWORD_HASH = _clean(
"AUTH_ADMIN_PASSWORD_HASH",
(
_require("AUTH_ADMIN_PASSWORD_HASH", feature_flag="AUTH_ENABLED")
if AUTH_ENABLED
else os.getenv("AUTH_ADMIN_PASSWORD_HASH", "")
),
)
# The second `user` account is OPTIONAL, and left unset in this deployment.
# An empty hash is how the account is switched off: auth.py builds its account
# table from these values and omits any entry whose hash is blank, so there is
# nothing to sign in to. Setting the hash again re-enables it with no code
# change - which is exactly what the test suite does in tests/conftest.py.
AUTH_USER_USERNAME = os.getenv("AUTH_USER_USERNAME", "user").strip().strip("'\"")
AUTH_USER_PASSWORD_HASH = os.getenv("AUTH_USER_PASSWORD_HASH", "").strip().strip("'\"")
AUTH_USER_USERNAME = _clean(
"AUTH_USER_USERNAME", os.getenv("AUTH_USER_USERNAME", "user")
)
AUTH_USER_PASSWORD_HASH = _clean(
"AUTH_USER_PASSWORD_HASH", os.getenv("AUTH_USER_PASSWORD_HASH", "")
)
# Failed-login throttle, applied per username+client-IP. Prevents an exposed
# login endpoint from being a free password oracle.