Files
landing_page/refer_landing_page/scripts/extract/verify_counts.py
T
Godopu dadd14feb1 test(e2e,verify): enhance live API verification and dynamic port binding in E2E runner
- Update verify_counts.py to validate SQLite database records and live Go REST API endpoints directly
- Harden E2E test runner with dynamic free port binding (get_free_port) to prevent port collisions
- Add exact regex matching for dynamic (ƒ) and static (○) route table allocations
- Ignore root build binary artifacts in backend/.gitignore
2026-08-24 20:31:42 +09:00

283 lines
14 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
Data Verification Gate (AC-07 / AC-08 / AC-10 / AC-11 / AC-30 / Live Database & API Pipeline Integrity)
Verifies:
1. Full data extraction from legacy HTML sources (16 members, 60 alumni, 68 intl, 80 dom, 75 pat, 45 sem, 6 bodies, 44 docs, 3 projects).
2. SQLite Backend Database (backend/anl.db) tables & row counts matching baseline.
3. Live Go REST API endpoints (http://localhost:8080/api/v1) data losslessness & schema consistency if server is active.
4. ISO 8601 strict date formatting across all entities.
5. Publication volume cross-check and keyword substring assertions.
"""
import sys
import os
import re
import sqlite3
import json
import urllib.request
BASE_DIR = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
sys.path.insert(0, os.path.join(BASE_DIR, "refer_landing_page", "scripts", "extract"))
ROOT_PROJECT_DIR = os.path.dirname(BASE_DIR)
DB_PATH = os.path.join(ROOT_PROJECT_DIR, "backend", "anl.db")
if not os.path.exists(DB_PATH):
DB_PATH = os.path.join(BASE_DIR, "backend", "anl.db")
API_BASE_URL = os.environ.get("API_BASE_URL", "http://localhost:8080/api/v1")
MONTH_MAP = {
"january": "01", "jan": "01", "february": "02", "feb": "02", "march": "03", "mar": "03",
"april": "04", "apr": "04", "may": "05", "june": "06", "jun": "06", "july": "07", "jul": "07",
"august": "08", "aug": "08", "september": "09", "sep": "09", "sept": "09", "october": "10",
"oct": "10", "november": "11", "nov": "11", "december": "12", "dec": "12"
}
def verify():
from generate_all import parse_members, parse_publications, parse_lectures, parse_standards, parse_projects
active_m, alumni_m = parse_members()
intl_p, dom_p, pat_p = parse_publications()
semesters = parse_lectures()
bodies = parse_standards()
projects = parse_projects()
total_pubs = len(intl_p) + len(dom_p) + len(pat_p)
std_docs_count = sum(len(p["documents"]) for b in bodies for p in b["projects"])
print("=== DATA VERIFICATION GATE (AC-07 / AC-08 / AC-10 / AC-11 / AC-30 / LIVE PIPELINE INTEGRITY) ===")
print(f"[HTML Source Extraction Baseline]")
print(f" Active Members: {len(active_m)} (Expected: 16)")
print(f" Alumni: {len(alumni_m)} (Expected: 60)")
print(f" International Papers: {len(intl_p)} (Expected: 68)")
print(f" Domestic Papers: {len(dom_p)} (Expected: 80)")
print(f" Patents: {len(pat_p)} (Expected: 75)")
print(f" Total Publications: {total_pubs} (Expected: 223)")
print(f" Lectures Semesters: {len(semesters)} (Expected: 45)")
print(f" Standard Bodies: {len(bodies)} (Expected: 6)")
print(f" Standard Documents: {std_docs_count} (Expected: 44)")
print(f" Research Projects: {len(projects)} (Expected: 3)")
errors = []
# 1. Source Extraction Integrity Checks
if len(active_m) != 16:
errors.append(f"Active Members count mismatch: got {len(active_m)}, expected 16")
if len(alumni_m) != 60:
errors.append(f"Alumni count mismatch: got {len(alumni_m)}, expected 60")
if len(intl_p) != 68:
errors.append(f"International Papers count mismatch: got {len(intl_p)}, expected 68")
if len(dom_p) != 80:
errors.append(f"Domestic Papers count mismatch: got {len(dom_p)}, expected 80")
if len(pat_p) != 75:
errors.append(f"Patents count mismatch: got {len(pat_p)}, expected 75")
if total_pubs != 223:
errors.append(f"Total Publications count mismatch: got {total_pubs}, expected 223")
if len(semesters) != 45:
errors.append(f"Semesters count mismatch: got {len(semesters)}, expected 45")
if len(bodies) != 6:
errors.append(f"Standard Bodies count mismatch: got {len(bodies)}, expected 6")
if std_docs_count != 44:
errors.append(f"Standard Documents count mismatch: got {std_docs_count}, expected 44")
if len(projects) != 3:
errors.append(f"Research Projects count mismatch: got {len(projects)}, expected 3")
# 2. SQLite Backend Database Verification
if os.path.exists(DB_PATH):
print(f"\n[SQLite Database Verification ({DB_PATH})]")
try:
conn = sqlite3.connect(DB_PATH)
cur = conn.cursor()
def get_count(table, where=""):
query = f"SELECT count(*) FROM {table} {where}"
cur.execute(query)
return cur.fetchone()[0]
db_counts = {
"members": get_count("members"),
"alumni": get_count("alumni"),
"intl_pubs": get_count("publications", "WHERE category='intl-journal-conf'"),
"dom_pubs": get_count("publications", "WHERE category='domestic-journal-conf'"),
"patents": get_count("patents"),
"semesters": get_count("semesters"),
"standards_bodies": get_count("standards_bodies"),
"standard_documents": get_count("standard_documents"),
"research_projects": get_count("research_projects"),
}
for k, v in db_counts.items():
print(f" DB {k}: {v}")
if db_counts["members"] != 16:
errors.append(f"SQLite members count: {db_counts['members']}, expected 16")
if db_counts["alumni"] != 60:
errors.append(f"SQLite alumni count: {db_counts['alumni']}, expected 60")
if db_counts["intl_pubs"] != 68:
errors.append(f"SQLite intl publications count: {db_counts['intl_pubs']}, expected 68")
if db_counts["dom_pubs"] != 80:
errors.append(f"SQLite dom publications count: {db_counts['dom_pubs']}, expected 80")
if db_counts["patents"] != 75:
errors.append(f"SQLite patents count: {db_counts['patents']}, expected 75")
if db_counts["semesters"] != 45:
errors.append(f"SQLite semesters count: {db_counts['semesters']}, expected 45")
if db_counts["standards_bodies"] != 6:
errors.append(f"SQLite standards bodies count: {db_counts['standards_bodies']}, expected 6")
if db_counts["standard_documents"] != 44:
errors.append(f"SQLite standard documents count: {db_counts['standard_documents']}, expected 44")
if db_counts["research_projects"] != 3:
errors.append(f"SQLite research projects count: {db_counts['research_projects']}, expected 3")
conn.close()
print(" ✅ SQLite Database records match expected baseline exactly")
except Exception as e:
errors.append(f"SQLite DB verification failed: {e}")
else:
print(f"\n⚠️ SQLite database not found at {DB_PATH}, skipping direct DB query")
# 3. Live Go REST API Endpoint Verification (if backend is active)
is_api_live = False
try:
req = urllib.request.Request(f"{API_BASE_URL}/health", headers={"User-Agent": "DataVerifier"})
with urllib.request.urlopen(req, timeout=1) as resp:
if resp.status == 200:
is_api_live = True
except Exception:
is_api_live = False
if is_api_live:
print(f"\n[Live Go REST API Verification ({API_BASE_URL})]")
try:
def fetch_json(endpoint):
req = urllib.request.Request(f"{API_BASE_URL}{endpoint}", headers={"User-Agent": "DataVerifier"})
with urllib.request.urlopen(req, timeout=3) as resp:
data = json.loads(resp.read().decode("utf-8"))
return data.get("data", [])
stats = fetch_json("/stats/summary")
api_members = fetch_json("/members")
api_alumni = fetch_json("/alumni")
api_intl = fetch_json("/publications?category=intl-journal-conf")
api_dom = fetch_json("/publications?category=domestic-journal-conf")
api_patents = fetch_json("/patents")
api_semesters = fetch_json("/semesters")
api_standards = fetch_json("/standards-bodies")
api_projects = fetch_json("/research-projects")
api_std_docs_count = sum(len(p.get("documents", [])) for b in api_standards for p in b.get("projects", []))
print(f" API stats summary: {stats}")
print(f" API members count: {len(api_members)}")
print(f" API alumni count: {len(api_alumni)}")
print(f" API publications (intl): {len(api_intl)}, (dom): {len(api_dom)}")
print(f" API patents count: {len(api_patents)}")
print(f" API semesters count: {len(api_semesters)}")
print(f" API standards bodies: {len(api_standards)} (docs: {api_std_docs_count})")
print(f" API projects count: {len(api_projects)}")
if stats.get("intl_publications") != 68:
errors.append(f"API stats intl_publications mismatch: {stats.get('intl_publications')}, expected 68")
if stats.get("standardization_docs") != 44:
errors.append(f"API stats standardization_docs mismatch: {stats.get('standardization_docs')}, expected 44")
if stats.get("patents") != 75:
errors.append(f"API stats patents mismatch: {stats.get('patents')}, expected 75")
if len(api_members) != 16:
errors.append(f"API members count: {len(api_members)}, expected 16")
if len(api_alumni) != 60:
errors.append(f"API alumni count: {len(api_alumni)}, expected 60")
if len(api_intl) != 68:
errors.append(f"API intl publications count: {len(api_intl)}, expected 68")
if len(api_dom) != 80:
errors.append(f"API dom publications count: {len(api_dom)}, expected 80")
if len(api_patents) != 75:
errors.append(f"API patents count: {len(api_patents)}, expected 75")
if len(api_semesters) != 45:
errors.append(f"API semesters count: {len(api_semesters)}, expected 45")
if len(api_standards) != 6:
errors.append(f"API standards bodies count: {len(api_standards)}, expected 6")
if api_std_docs_count != 44:
errors.append(f"API standard documents count: {api_std_docs_count}, expected 44")
if len(api_projects) != 3:
errors.append(f"API research projects count: {len(api_projects)}, expected 3")
print(" ✅ Live REST API endpoints serve 100% losslessly verified data")
except Exception as e:
errors.append(f"Live API verification error: {e}")
else:
print(f"\n️ Live Go API server not responding at {API_BASE_URL} (tested offline DB mode)")
# 4. AC-30 Keyword Substring Verification & AC-32 Deliverable WG Attribution
for proj in projects:
abstract = proj.get("abstract", "")
for kw in proj.get("keywords", []):
if kw not in abstract:
errors.append(f"AC-30 Keyword Violation in project '{proj.get('slug')}': keyword '{kw}' not in abstract")
if proj["slug"] == "ccis":
for d in proj.get("deliverables", []):
if "63246" not in d["ref"]:
errors.append(f"AC-32 Violation: CCIS project deliverable '{d['ref']}' does not belong to CCIS WG (IEC 63246)")
# 5. Title artifact check (No trailing quotes or commas in publication titles)
for p in intl_p + dom_p + pat_p:
t = p.get("title", "")
if re.search(r'["“”]|,$', t):
errors.append(f"N1 Title Artifact Violation in publication '{p.get('id')}': title '{t}' contains quotes or trailing comma")
# 6. Strict ISO 8601 Date and BOM Verification
content_dir = os.path.join(BASE_DIR, "content")
date_regex = re.compile(r'^(19|20)\d{2}(-(0[1-9]|1[0-2]))?(-(0[1-9]|[12]\d|3[01]))?$')
for fname in os.listdir(content_dir) if os.path.exists(content_dir) else []:
if fname.endswith(".ts"):
fpath = os.path.join(content_dir, fname)
with open(fpath, "r", encoding="utf-8") as f:
text = f.read()
if text.startswith('\ufeff'):
errors.append(f"BOM (U+FEFF) found in {fname}")
for p in intl_p + dom_p + pat_p:
for field in ["publishedAt", "applicationAt", "registrationAt"]:
val = p.get(field)
if val and not date_regex.match(val):
errors.append(f"Invalid ISO 8601 date in publication '{p.get('title')}': {field}='{val}'")
for m in alumni_m:
val = m.get("graduatedAt")
if val and not date_regex.match(val):
errors.append(f"Invalid ISO 8601 date in alumnus '{m.get('ko')}': graduatedAt='{val}'")
# 7. Volume Cross-Check for Publications
MON_PAT = r'(?:January|February|March|April|May|June|July|August|September|October|November|December|Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)'
for pub in intl_p + dom_p:
vol = pub.get("volume", "")
pub_date = pub.get("publishedAt", "")
if not pub_date:
errors.append(f"Missing publishedAt for paper: {pub.get('title')}")
continue
cands = []
for m in re.finditer(rf'\b({MON_PAT})\.?\s*,?\s*((?:19|20)\d{{2}})', vol, re.I):
mo_str = m.group(1).lower().rstrip('.')
if mo_str in MONTH_MAP:
cands.append(m.group(2))
for m in re.finditer(r'((?:19|20)\d{2})\s*년\s*(\d{1,2})\s*월', vol):
cands.append(m.group(1))
if cands:
if not any(pub_date.startswith(yr) for yr in cands):
errors.append(f"Volume date mismatch for '{pub.get('title')}': volume candidates={cands}, publishedAt={pub_date}")
if errors:
print(f"\n❌ DATA VERIFICATION FAILED ({len(errors)} errors):")
for err in errors:
print(f" - {err}")
sys.exit(1)
else:
print("\n✅ ALL DATA VERIFICATION GATES PASSED CLEAN (Exit Code 0)")
sys.exit(0)
if __name__ == "__main__":
verify()