136 lines
6.1 KiB
Python
136 lines
6.1 KiB
Python
#!/usr/bin/env python3
|
|
import sys
|
|
import os
|
|
import re
|
|
|
|
BASE_DIR = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
|
sys.path.insert(0, os.path.join(BASE_DIR, "refer_landing_page", "scripts", "extract"))
|
|
|
|
MONTH_MAP = {
|
|
"january": "01", "jan": "01", "february": "02", "feb": "02", "march": "03", "mar": "03",
|
|
"april": "04", "apr": "04", "may": "05", "june": "06", "jun": "06", "july": "07", "jul": "07",
|
|
"august": "08", "aug": "08", "september": "09", "sep": "09", "sept": "09", "october": "10",
|
|
"oct": "10", "november": "11", "nov": "11", "december": "12", "dec": "12"
|
|
}
|
|
|
|
def verify():
|
|
from generate_all import parse_members, parse_publications, parse_lectures, parse_standards, parse_projects
|
|
|
|
active_m, alumni_m = parse_members()
|
|
intl_p, dom_p, pat_p = parse_publications()
|
|
semesters = parse_lectures()
|
|
bodies = parse_standards()
|
|
projects = parse_projects()
|
|
|
|
total_pubs = len(intl_p) + len(dom_p) + len(pat_p)
|
|
std_docs_count = sum(len(p["documents"]) for b in bodies for p in b["projects"])
|
|
|
|
print("=== DATA VERIFICATION GATE (AC-07 / AC-08 / AC-10 / AC-11 / AC-30 / VOLUME CROSS-CHECK) ===")
|
|
print(f"Active Members: {len(active_m)} (Expected: 16)")
|
|
print(f"Alumni: {len(alumni_m)} (Expected: 60)")
|
|
print(f"International Papers: {len(intl_p)} (Expected: 68)")
|
|
print(f"Domestic Papers: {len(dom_p)} (Expected: 80)")
|
|
print(f"Patents: {len(pat_p)} (Expected: 75)")
|
|
print(f"Total Publications: {total_pubs} (Expected: 223)")
|
|
print(f"Lectures Semesters: {len(semesters)} (Expected: 45)")
|
|
print(f"Standard Bodies: {len(bodies)} (Expected: 6)")
|
|
print(f"Standard Documents: {std_docs_count} (Expected: 44)")
|
|
print(f"Research Projects: {len(projects)} (Expected: 3)")
|
|
|
|
errors = []
|
|
|
|
if len(projects) != 3:
|
|
errors.append(f"Research Projects count mismatch: got {len(projects)}, expected 3")
|
|
|
|
# AC-30: Keyword Substring Verification & AC-32 Deliverable WG Attribution
|
|
for proj in projects:
|
|
abstract = proj.get("abstract", "")
|
|
for kw in proj.get("keywords", []):
|
|
if kw not in abstract:
|
|
errors.append(f"AC-30 Keyword Violation in project '{proj.get('slug')}': keyword '{kw}' not in abstract")
|
|
|
|
if proj["slug"] == "ccis":
|
|
for d in proj.get("deliverables", []):
|
|
if "63246" not in d["ref"]:
|
|
errors.append(f"AC-32 Violation: CCIS project deliverable '{d['ref']}' does not belong to CCIS WG (IEC 63246)")
|
|
|
|
# N1: Title artifact check (No trailing quotes or commas in publication titles)
|
|
for p in intl_p + dom_p + pat_p:
|
|
t = p.get("title", "")
|
|
if re.search(r'["“”]|,$', t):
|
|
errors.append(f"N1 Title Artifact Violation in publication '{p.get('id')}': title '{t}' contains quotes or trailing comma")
|
|
|
|
# 1. Count checks
|
|
if len(active_m) != 16:
|
|
errors.append(f"Active Members count mismatch: got {len(active_m)}, expected 16")
|
|
if len(alumni_m) != 60:
|
|
errors.append(f"Alumni count mismatch: got {len(alumni_m)}, expected 60")
|
|
if len(intl_p) != 68:
|
|
errors.append(f"International Papers count mismatch: got {len(intl_p)}, expected 68")
|
|
if len(dom_p) != 80:
|
|
errors.append(f"Domestic Papers count mismatch: got {len(dom_p)}, expected 80")
|
|
if len(pat_p) != 75:
|
|
errors.append(f"Patents count mismatch: got {len(pat_p)}, expected 75")
|
|
if total_pubs != 223:
|
|
errors.append(f"Total Publications count mismatch: got {total_pubs}, expected 223")
|
|
if len(semesters) != 45:
|
|
errors.append(f"Semesters count mismatch: got {len(semesters)}, expected 45")
|
|
if len(bodies) != 6:
|
|
errors.append(f"Standard Bodies count mismatch: got {len(bodies)}, expected 6")
|
|
if std_docs_count != 44:
|
|
errors.append(f"Standard Documents count mismatch: got {std_docs_count}, expected 44")
|
|
|
|
# 2. File checks (BOM, Strict ISO 8601, and Semantic Date Ranges)
|
|
content_dir = os.path.join(BASE_DIR, "content")
|
|
date_regex = re.compile(r'^(19|20)\d{2}(-(0[1-9]|1[0-2]))?(-(0[1-9]|[12]\d|3[01]))?$')
|
|
|
|
for fname in ["members.ts", "publications.ts", "lectures.ts", "standards.ts"]:
|
|
fpath = os.path.join(content_dir, fname)
|
|
if not os.path.exists(fpath):
|
|
errors.append(f"Missing content file: {fname}")
|
|
continue
|
|
with open(fpath, "r", encoding="utf-8") as f:
|
|
text = f.read()
|
|
|
|
if text.startswith('\ufeff'):
|
|
errors.append(f"BOM (U+FEFF) found in {fname}")
|
|
|
|
dates = re.findall(r'"(publishedAt|applicationAt|registrationAt|graduatedAt)": "([^"]+)"', text)
|
|
for k, d in dates:
|
|
if not date_regex.match(d):
|
|
errors.append(f"Invalid ISO 8601 or out-of-bounds date in {fname}: {k}='{d}'")
|
|
|
|
# 3. Volume Cross-Check for Publications (Ensures volume date candidates match publishedAt)
|
|
MON_PAT = r'(?:January|February|March|April|May|June|July|August|September|October|November|December|Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)'
|
|
|
|
for pub in intl_p + dom_p:
|
|
vol = pub.get("volume", "")
|
|
pub_date = pub.get("publishedAt", "")
|
|
if not pub_date:
|
|
errors.append(f"Missing publishedAt for paper: {pub.get('title')}")
|
|
continue
|
|
|
|
cands = []
|
|
for m in re.finditer(rf'\b({MON_PAT})\.?\s*,?\s*((?:19|20)\d{{2}})', vol, re.I):
|
|
mo_str = m.group(1).lower().rstrip('.')
|
|
if mo_str in MONTH_MAP:
|
|
cands.append(m.group(2))
|
|
for m in re.finditer(r'((?:19|20)\d{2})\s*년\s*(\d{1,2})\s*월', vol):
|
|
cands.append(m.group(1))
|
|
|
|
if cands:
|
|
if not any(pub_date.startswith(yr) for yr in cands):
|
|
errors.append(f"Volume date mismatch for '{pub.get('title')}': volume candidates={cands}, publishedAt={pub_date}")
|
|
|
|
if errors:
|
|
print(f"\n❌ DATA VERIFICATION FAILED ({len(errors)} errors):")
|
|
for err in errors:
|
|
print(f" - {err}")
|
|
sys.exit(1)
|
|
else:
|
|
print("\n✅ ALL DATA VERIFICATION GATES PASSED CLEAN (Exit Code 0)")
|
|
sys.exit(0)
|
|
|
|
if __name__ == "__main__":
|
|
verify()
|