Production Login page passcode updates
This commit is contained in:
@@ -29,15 +29,19 @@ import logging
|
||||
import secrets
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Dict, List, Optional
|
||||
from typing import Dict, List, Optional, Tuple
|
||||
|
||||
import jwt
|
||||
|
||||
from app.infrastructure.settings import (
|
||||
API_KEYS,
|
||||
AUTH_ADMIN_PASSWORD_HASH,
|
||||
AUTH_ADMIN_USERNAME,
|
||||
AUTH_ALLOW_ANY_LOGIN,
|
||||
AUTH_ENABLED,
|
||||
AUTH_SECRET_KEY,
|
||||
AUTH_TOKEN_TTL_MINUTES,
|
||||
config_source,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -117,6 +121,110 @@ def hash_password(password: str, *, iterations: int = _PBKDF2_ITERATIONS) -> str
|
||||
)
|
||||
|
||||
|
||||
def _parse_encoded_hash(encoded: str) -> Optional[Tuple[bytes, bytes, int]]:
|
||||
"""
|
||||
Split an encoded digest into ``(salt, digest, iterations)``, or None if it
|
||||
is not one.
|
||||
|
||||
One parser, three callers. `verify_password` needs the parts, while
|
||||
`hash_is_wellformed` and `describe_password_hash` need only the verdict -
|
||||
and a login failing because the *configured* hash is corrupt is a different
|
||||
incident from a wrong password, so the two must agree on what "corrupt"
|
||||
means. Two copies of this parse would eventually disagree.
|
||||
|
||||
Values arrive here straight from the environment, so a hash pasted into a
|
||||
deployment platform's form field as "pbkdf2_sha256$..." is unwrapped rather
|
||||
than rejected: the surrounding quotes are almost never intended as part of
|
||||
the secret, and the failure they cause otherwise is a silent 401.
|
||||
"""
|
||||
if not encoded:
|
||||
return None
|
||||
encoded = encoded.strip().strip("'\"")
|
||||
try:
|
||||
prefix, raw_iterations, raw_salt, raw_digest = encoded.split("$")
|
||||
if prefix != _PBKDF2_PREFIX:
|
||||
return None
|
||||
# validate=True so junk is rejected rather than silently discarded:
|
||||
# b64decode's default drops non-alphabet characters, which would let a
|
||||
# subtly corrupted hash decode to the wrong bytes and fail as a "wrong
|
||||
# password" instead of as the configuration error it is.
|
||||
# binascii.Error subclasses ValueError, so it is caught below.
|
||||
salt = base64.b64decode(raw_salt, validate=True)
|
||||
digest = base64.b64decode(raw_digest, validate=True)
|
||||
iterations = int(raw_iterations)
|
||||
except (ValueError, TypeError):
|
||||
return None
|
||||
# A structurally valid string that decodes to nothing is still unusable,
|
||||
# and PBKDF2 rejects a non-positive iteration count by raising.
|
||||
if not salt or not digest or iterations < 1:
|
||||
return None
|
||||
return salt, digest, iterations
|
||||
|
||||
|
||||
def hash_is_wellformed(encoded: str) -> bool:
|
||||
"""Whether a configured digest can be checked against at all.
|
||||
|
||||
Distinct from "does the password match": this asks whether the credential
|
||||
*store* is usable, which is a deployment fault rather than a sign-in one.
|
||||
"""
|
||||
return _parse_encoded_hash(encoded) is not None
|
||||
|
||||
|
||||
def password_hash_fingerprint(encoded: str) -> str:
|
||||
"""
|
||||
A short, non-reversible identifier for a configured digest.
|
||||
|
||||
Safe to log and to publish: it is a truncated SHA-256 of the *encoded
|
||||
digest*, and that digest already embeds a 16-byte random salt, so this says
|
||||
which credential is loaded without saying anything about the password
|
||||
behind it. It exists so a running deployment can be compared against the
|
||||
config it was supposed to have been built from - the failure this project
|
||||
actually hit - without moving a secret in order to do the comparison.
|
||||
"""
|
||||
if not encoded:
|
||||
return ""
|
||||
return hashlib.sha256(encoded.strip().strip("'\"").encode("utf-8")).hexdigest()[:12]
|
||||
|
||||
|
||||
def describe_password_hash(encoded: str) -> Dict[str, object]:
|
||||
"""A loggable/publishable summary of a configured digest. Never its bytes."""
|
||||
parsed = _parse_encoded_hash(encoded)
|
||||
return {
|
||||
"valid": parsed is not None,
|
||||
"algorithm": _PBKDF2_PREFIX if parsed is not None else None,
|
||||
"iterations": parsed[2] if parsed is not None else None,
|
||||
"fingerprint": password_hash_fingerprint(encoded),
|
||||
}
|
||||
|
||||
|
||||
def auth_config_summary() -> Dict[str, object]:
|
||||
"""
|
||||
The effective authentication configuration, in a form safe to both log and
|
||||
publish. Contains no password and no hash - only the fingerprint.
|
||||
|
||||
This is deliberately one function with two callers (the startup log in
|
||||
app/main.py and GET /api/health), because its entire purpose is letting two
|
||||
*deployments* be compared, and that only works if both report the same
|
||||
fields computed the same way.
|
||||
|
||||
`*_source` is the field that earns this its keep. A value of "process-env"
|
||||
means the container's own environment supplied it and the .env file baked
|
||||
into the image was ignored - which is invisible from anywhere else, and is
|
||||
precisely how a corrected credential can keep failing after a redeploy.
|
||||
"""
|
||||
described = describe_password_hash(AUTH_ADMIN_PASSWORD_HASH)
|
||||
return {
|
||||
"enabled": AUTH_ENABLED,
|
||||
"allow_any_login": AUTH_ALLOW_ANY_LOGIN,
|
||||
"admin_username": AUTH_ADMIN_USERNAME,
|
||||
"password_hash_valid": bool(described["valid"]),
|
||||
"password_hash_iterations": described["iterations"],
|
||||
"password_hash_fingerprint": described["fingerprint"],
|
||||
"admin_username_source": config_source("AUTH_ADMIN_USERNAME"),
|
||||
"password_hash_source": config_source("AUTH_ADMIN_PASSWORD_HASH"),
|
||||
}
|
||||
|
||||
|
||||
def verify_password(password: str, encoded: str) -> bool:
|
||||
"""
|
||||
Check a password against an encoded digest.
|
||||
@@ -127,21 +235,15 @@ def verify_password(password: str, encoded: str) -> bool:
|
||||
"""
|
||||
if not encoded:
|
||||
return False
|
||||
encoded = encoded.strip().strip("'\"")
|
||||
try:
|
||||
prefix, raw_iterations, raw_salt, raw_digest = encoded.split("$")
|
||||
if prefix != _PBKDF2_PREFIX:
|
||||
return False
|
||||
expected = base64.b64decode(raw_salt), base64.b64decode(raw_digest)
|
||||
salt, digest = expected
|
||||
iterations = int(raw_iterations)
|
||||
except (ValueError, TypeError):
|
||||
parsed = _parse_encoded_hash(encoded)
|
||||
if parsed is None:
|
||||
logger.error(
|
||||
"A configured password hash is malformed and cannot be used. Regenerate "
|
||||
"it with: python scripts/make_auth_secrets.py"
|
||||
)
|
||||
return False
|
||||
|
||||
salt, digest, iterations = parsed
|
||||
candidate = hashlib.pbkdf2_hmac("sha256", password.encode("utf-8"), salt, iterations)
|
||||
return hmac.compare_digest(candidate, digest)
|
||||
|
||||
|
||||
@@ -27,6 +27,17 @@ from __future__ import annotations
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
# Snapshotted BEFORE load_dotenv, and that ordering is the entire point.
|
||||
# load_dotenv() is called without override=True, so a variable already in the
|
||||
# process environment silently beats the .env file and keeps beating it no
|
||||
# matter how many times the file is corrected. That is not hypothetical here:
|
||||
# the deployment platform injects its Environment tab into the container, so a
|
||||
# stale value left in that tab overrides the credentials baked into the image
|
||||
# (backend/Dockerfile copies .env.production to /app/.env) and the only symptom
|
||||
# is a 401 that nothing explains. Comparing a name against this set answers
|
||||
# "which of the two won?" - see config_source() below.
|
||||
_PREEXISTING_ENV = frozenset(os.environ)
|
||||
|
||||
try:
|
||||
from dotenv import load_dotenv
|
||||
|
||||
@@ -39,6 +50,45 @@ except ImportError:
|
||||
pass
|
||||
|
||||
|
||||
# Names whose raw value arrived wrapped in quotes or padded with whitespace.
|
||||
# Recorded rather than merely fixed: stripping keeps the login working, but the
|
||||
# only place the original shape is still visible is right here, before the value
|
||||
# is normalised. A quoted hash is the signature of a value pasted into a web
|
||||
# form, so surfacing it at startup is what stops the next person rediscovering
|
||||
# it from a 401. See DB_PASSWORD in .env.production for the counter-case where
|
||||
# the quotes ARE part of the secret - which is why this warns, and does not fail.
|
||||
_ENV_NEEDED_CLEANUP = set()
|
||||
|
||||
|
||||
def _clean(name: str, raw: str) -> str:
|
||||
"""Strip surrounding quotes/whitespace off an env value, remembering if it mattered."""
|
||||
cleaned = raw.strip().strip("'\"")
|
||||
if cleaned != raw:
|
||||
_ENV_NEEDED_CLEANUP.add(name)
|
||||
return cleaned
|
||||
|
||||
|
||||
def cleaned_env_names() -> list:
|
||||
"""Which settings needed quote/whitespace stripping. Reported at startup."""
|
||||
return sorted(_ENV_NEEDED_CLEANUP)
|
||||
|
||||
|
||||
def config_source(name: str) -> str:
|
||||
"""
|
||||
Where a setting's value actually came from: the process environment, the
|
||||
.env file, or this module's own default.
|
||||
|
||||
Reported at startup for the AUTH_* values (see app/main.py) so that an
|
||||
override arriving from outside the image is visible in the logs instead of
|
||||
being inferred from a failing login.
|
||||
"""
|
||||
if name in _PREEXISTING_ENV:
|
||||
return "process-env"
|
||||
if name in os.environ:
|
||||
return "env-file"
|
||||
return "default"
|
||||
|
||||
|
||||
def _bool(name: str, default: str) -> bool:
|
||||
return os.getenv(name, default).strip().lower() in {"1", "true", "yes"}
|
||||
|
||||
@@ -290,20 +340,29 @@ AUTH_TOKEN_TTL_MINUTES = int(os.getenv("AUTH_TOKEN_TTL_MINUTES", "720"))
|
||||
# `make_auth_secrets.py` prints the lines ready to paste.
|
||||
#
|
||||
# `admin` is required whenever auth is on: without it nobody could sign in.
|
||||
AUTH_ADMIN_USERNAME = os.getenv("AUTH_ADMIN_USERNAME", "admin").strip().strip("'\"")
|
||||
AUTH_ADMIN_PASSWORD_HASH = (
|
||||
_require("AUTH_ADMIN_PASSWORD_HASH", feature_flag="AUTH_ENABLED")
|
||||
if AUTH_ENABLED
|
||||
else os.getenv("AUTH_ADMIN_PASSWORD_HASH", "")
|
||||
).strip().strip("'\"")
|
||||
AUTH_ADMIN_USERNAME = _clean(
|
||||
"AUTH_ADMIN_USERNAME", os.getenv("AUTH_ADMIN_USERNAME", "admin")
|
||||
)
|
||||
AUTH_ADMIN_PASSWORD_HASH = _clean(
|
||||
"AUTH_ADMIN_PASSWORD_HASH",
|
||||
(
|
||||
_require("AUTH_ADMIN_PASSWORD_HASH", feature_flag="AUTH_ENABLED")
|
||||
if AUTH_ENABLED
|
||||
else os.getenv("AUTH_ADMIN_PASSWORD_HASH", "")
|
||||
),
|
||||
)
|
||||
|
||||
# The second `user` account is OPTIONAL, and left unset in this deployment.
|
||||
# An empty hash is how the account is switched off: auth.py builds its account
|
||||
# table from these values and omits any entry whose hash is blank, so there is
|
||||
# nothing to sign in to. Setting the hash again re-enables it with no code
|
||||
# change - which is exactly what the test suite does in tests/conftest.py.
|
||||
AUTH_USER_USERNAME = os.getenv("AUTH_USER_USERNAME", "user").strip().strip("'\"")
|
||||
AUTH_USER_PASSWORD_HASH = os.getenv("AUTH_USER_PASSWORD_HASH", "").strip().strip("'\"")
|
||||
AUTH_USER_USERNAME = _clean(
|
||||
"AUTH_USER_USERNAME", os.getenv("AUTH_USER_USERNAME", "user")
|
||||
)
|
||||
AUTH_USER_PASSWORD_HASH = _clean(
|
||||
"AUTH_USER_PASSWORD_HASH", os.getenv("AUTH_USER_PASSWORD_HASH", "")
|
||||
)
|
||||
|
||||
# Failed-login throttle, applied per username+client-IP. Prevents an exposed
|
||||
# login endpoint from being a free password oracle.
|
||||
|
||||
Reference in New Issue
Block a user