#!/usr/bin/env python3 """Emit go-api/internal/domain/resources_gen.go from the live PostgreSQL schema. Column names, types, enum values and nullability are read out of information_schema so they can never drift from the migrations. The per-resource metadata below (path, default sort, default limit, operations, required fields) comes from docs/api-contract.md and is the only hand-maintained part. Usage: make gen-resources """ import collections import json import os import subprocess import sys DB = os.environ.get("DATABASE_NAME", "Krow-force") HOST = os.environ.get("DATABASE_HOST", "127.0.0.1") PORT = os.environ.get("DATABASE_PORT", "5432") USER = os.environ.get("DATABASE_USER", "postgres") def q(sql): out = subprocess.run( ["psql", "-h", HOST, "-p", PORT, "-U", USER, "-d", DB, "-tAF", "\x1f", "-c", sql], capture_output=True, text=True) if out.returncode: sys.exit(out.stderr) return [l.split("\x1f") for l in out.stdout.strip().split("\n") if l.strip()] # ops='' means the resource has a table and is seeded, but serves no HTTP # endpoint (see api-contract.md §2). It still needs a descriptor so the seeder # can write it. META = { 'job_postings': dict(name='JobPosting', path='job-postings', sort='-created_date', limit=100, ops='List|Get|Create|Update', req=['title']), 'job_applications':dict(name='JobApplication',path='job-applications',sort='-ai_score', limit=200, ops='List|Create|Update|Delete', req=['job_posting_id','applicant_name','email']), 'ai_interviews': dict(name='AIInterview', path='ai-interviews', sort='-created_date', limit=100, ops='List|Create', req=['application_id','job_posting_id']), 'staff': dict(name='Staff', path='staff', sort='-created_date', limit=100, ops='List|Create|Update', req=['name','email','hire_date']), 'worker_profiles': dict(name='WorkerProfile', path='worker-profiles', sort='-krow_score', limit=500, ops='List|Create|Update', req=['full_name','email']), 'courses': dict(name='Course', path='courses', sort='-created_date', limit=200, ops='List|Get|Create|Update', req=['title'], orgnull=True), 'learning_paths': dict(name='LearningPath', path='learning-paths', sort='-created_date', limit=100, ops='List', req=['name'], orgnull=True), 'role_categories': dict(name='RoleCategory', path='role-categories', sort='-created_date', limit=100, ops='List|Create', req=['name']), 'certifications': dict(name='Certification', path='certifications', sort='-created_date', limit=200, ops='List|Create|Delete', req=['name']), 'user_activity': dict(name='UserActivity', path='user-activity', sort='-created_date', limit=500, ops='List|Create', req=['event_type']), 'evidence': dict(name='Evidence', path='evidence', sort='-created_date', limit=200, ops='List|Create|Update', req=['type','worker_email']), 'assignments': dict(name='Assignment', path='assignments', sort='-created_date', limit=500, ops='List|Create', req=['job_posting_id','worker_email','starts_at']), 'shift_records': dict(name='ShiftRecord', path='shift-records', sort='-created_date', limit=500, ops='List', req=[]), 'badges': dict(name='Badge', path='badges', sort='-created_date', limit=200, ops='', req=['name']), } ORDER = list(META) # Columns the API never accepts from a request body, on every table that has # them. The row's identity, its tenant and its timestamps are the server's. READONLY = {'id', 'org_id', 'created_date', 'updated_date', 'legacy_id'} # Columns that are server-owned on ONE table only. # # These name a *person*, and before Phase 3D a client could set them freely — # which meant any authorization rule written on top of them could be defeated by # the same request the rule was meant to constrain. A caller could reassign a # worker profile to somebody else, or write an audit-log entry attributed to # anyone in the organization. # # They are read-only here and filled in from the authenticated session instead; # see the Derived rules in internal/domain/policy.go for which value each one # receives and when. SERVER_OWNED = { 'worker_profiles': {'user_id'}, 'user_activity': {'user_id', 'user_email', 'user_name', 'account_type'}, 'job_postings': {'created_by'}, } def kind(dt, udt, enums): """Classify a column. USER-DEFINED covers both enums and extension types: citext reports as USER-DEFINED too. Enum-ness is decided by whether pg_enum actually has labels for the type, not by data_type alone — classifying citext as an enum with no permitted values rejects every email the API is sent. """ if udt == 'uuid': return 'KindUUID', 'uuid' if dt == 'ARRAY': return 'KindTextArray', 'text[]' if dt == 'jsonb': return 'KindJSON', 'jsonb' if dt in ('integer', 'smallint'): return 'KindInt', 'int' if dt == 'bigint': return 'KindInt', 'bigint' if dt == 'numeric': return 'KindFloat', 'numeric' if dt == 'boolean': return 'KindBool', 'boolean' if dt == 'timestamp with time zone': return 'KindTimestamp', 'timestamptz' if dt == 'date': return 'KindDate', 'date' if dt == 'USER-DEFINED' and enums.get(udt): return 'KindEnum', udt if udt == 'citext': return 'KindString', 'citext' return 'KindString', 'text' def main(): cols = q("""SELECT table_name, column_name, data_type, udt_name, is_nullable FROM information_schema.columns WHERE table_schema='public' AND table_name<>'schema_migrations' ORDER BY table_name, ordinal_position""") enums = collections.defaultdict(list) for t, v in q("""SELECT t.typname, e.enumlabel FROM pg_type t JOIN pg_enum e ON e.enumtypid=t.oid JOIN pg_namespace n ON n.oid=t.typnamespace WHERE n.nspname='public' ORDER BY t.typname, e.enumsortorder"""): enums[t].append(v) by_table = collections.defaultdict(list) for t, c, dt, udt, nul in cols: by_table[t].append((c, dt, udt, nul == 'YES')) o = [] o.append('// Code generated by scripts/gen_resources.py. DO NOT EDIT BY HAND.') o.append('// Regenerate with: make gen-resources') o.append('//') o.append('// Column names, types, enum values and nullability are read out of') o.append('// information_schema so they cannot drift from the migrations. The') o.append('// per-resource metadata (path, default sort, default limit, supported') o.append('// operations, required fields) comes from docs/api-contract.md.') o.append('') o.append('package domain') o.append('') o.append('// AllResources is every resource the API serves.') o.append('var AllResources = []*Resource{') for tbl in ORDER: m = META[tbl] if m['ops']: ops = ' | '.join('Op' + x for x in m['ops'].split('|')) else: # No operations means no routes are registered for this resource. o.append(f'\t// {m["name"]} serves NO endpoint: useBadges has zero consumers and every') o.append('\t// badge the UI renders comes from worker_profiles.earned_badges. The') o.append('\t// descriptor exists so the seeder can write the table. api-contract.md §2.') ops = '0' o.append('\t{') o.append(f'\t\tName: {json.dumps(m["name"])}, Path: {json.dumps(m["path"])}, Table: {json.dumps(tbl)},') o.append(f'\t\tDefaultSort: {json.dumps(m["sort"])}, DefaultLimit: {m["limit"]},') o.append(f'\t\tOps: {ops},') if m.get('orgnull'): o.append('\t\tOrgNullable: true,') o.append('\t\tColumns: []Column{') for c, dt, udt, nullable in by_table[tbl]: k, cast = kind(dt, udt, enums) p = [f'Name: {json.dumps(c)}', f'Kind: {k}', f'PGType: {json.dumps(cast)}'] if not nullable: p.append('NotNull: true') if c in READONLY or c in SERVER_OWNED.get(tbl, ()): p.append('ReadOnly: true') if c in m['req']: p.append('Required: true') if k == 'KindEnum': p.append('Enum: []string{' + ', '.join(json.dumps(v) for v in enums[udt]) + '}') o.append('\t\t\t{' + ', '.join(p) + '},') o.append('\t\t},') o.append('\t},') o.append('}') sys.stdout.write('\n'.join(o) + '\n') if __name__ == '__main__': main()