- tools/audit_content.py: per-file vs item-title triage (SUSPECT/SMALL/NO-TEXT/NO-CYR) - sec01: .part removed; 10 files restored .pdf ext (LFS rename); #04/#19/#27 verify notes - sec02 #16: corrupted double-encoded txt -> clean flib fb2 b/844509; #17/#22 doc verified via catdoc - sec03 #11: -b epub = Spanish von Franz (Paidós 1983, forged EN OPF) -> bonus label - sec03 #42: КАРО 2012 'Irish Tales' = EN reader (not RU, not the listed book) -> bonus rename - sec03 #56: Onians b/c/d/e = Cambridge 24-87KB previews -> labeled - sec06 #18: filename 1981->1989 (Perera 'Descent to the Goddess' 1989; 1981 = other book) - sec06 #28: ia Zimmer OCR layer = foreign Devanagari text; images = RKP (p.100 verified) - sec04 #30: buksmart 2020 = RU/FR bilingual w/ VeryPDF watermarks -> note - linter: azw3/mobi ext + underscore in fname regex; manifest cross-check ignores off-disk rows; 8 legacy cards section-order fixed (CR before DL) - AGENTS.md: 'Content audit (2026-09-25)' lessons section - check-md: 0/219
124 lines
5.4 KiB
Python
124 lines
5.4 KiB
Python
#!/usr/bin/env python3
|
|
"""Lint book cards against CANON v3 (see AGENTS.md).
|
|
Usage: python3 tools/check-md.py [section ...] (default: all)
|
|
Exit code: 0 = clean, 1 = issues found."""
|
|
import re, sys, os, glob, importlib.util
|
|
|
|
# single source of truth for MANIFEST parsing: the migration tool
|
|
_spec = importlib.util.spec_from_file_location("_mig", os.path.join(os.path.dirname(__file__), "migrate_cards_v3.py"))
|
|
_mig = importlib.util.module_from_spec(_spec); _spec.loader.exec_module(_mig)
|
|
|
|
ALLOWED_SECTIONS = re.compile(r'^## (Editions — [A-Z]{2}|Downloads|Catalog records|Notes)$')
|
|
BANNED_SECTIONS = ('## Russian editions', '## Libgen', '## RSL records', '## Verdict',
|
|
'## Identity check', '## Search log')
|
|
BANNED_TEXTS = ('// not searched yet', '(not searched yet)', '// not searched',
|
|
'**Publisher (EN):**', '**EN ISBN:**', '**Reading list item(s):**')
|
|
VALID_MARKERS = ('✅', '🔶', '❌', '⬜', '🔎')
|
|
ORDER = ['ED', 'DL', 'CR', 'NT']
|
|
|
|
def sec_of(path):
|
|
return os.path.basename(os.path.dirname(path))
|
|
|
|
def lint_card(path, manifest_files):
|
|
c = open(path, encoding='utf-8').read()
|
|
issues = []
|
|
num = os.path.basename(path)[:2]
|
|
if not c.startswith('# '):
|
|
issues.append('no H1')
|
|
for f in ('**Author(s):**', '**Shelf mark:**', '**Section:**', '**Status:**'):
|
|
if f not in c.split('\n\n')[0] + '\n'.join(c.split('\n')[:8]):
|
|
issues.append(f'header missing {f}')
|
|
sm = re.search(r'\*\*Status:\*\* (.*)', c)
|
|
if sm and not sm.group(1).strip().startswith(VALID_MARKERS):
|
|
issues.append(f'status marker invalid: {sm.group(1)[:30]!r}')
|
|
for t in BANNED_TEXTS:
|
|
if t in c:
|
|
issues.append(f'banned text: {t}')
|
|
# sections
|
|
seen = []
|
|
for sm2 in re.finditer(r'^## (.+)$', c, re.M):
|
|
name = sm2.group(1).strip()
|
|
line = f'## {name}'
|
|
if not ALLOWED_SECTIONS.match(line):
|
|
issues.append(f'non-canonical section: {line}')
|
|
key = ('ED' if name.startswith('Editions —') else
|
|
'DL' if name == 'Downloads' else
|
|
'CR' if name == 'Catalog records' else 'NT')
|
|
seen.append(key)
|
|
# Notes exactly once (for migrated cards; tolerate 0 for ⬜ cards)
|
|
if seen.count('NT') > 1:
|
|
issues.append(f'Notes x{seen.count("NT")}')
|
|
# order: all ED before DL before CR before NT
|
|
pos = {k: [i for i, s in enumerate(seen) if s == k] for k in ORDER}
|
|
last = -1
|
|
for k in ORDER:
|
|
if pos[k]:
|
|
if max(pos[k]) < last:
|
|
issues.append(f'section order broken at {k}')
|
|
break
|
|
last = max(pos[k])
|
|
# empty sections
|
|
if re.search(r'^## [A-Za-z—& ]+\n\n\n## ', c, re.M):
|
|
issues.append('empty section body')
|
|
# Downloads table cross-check
|
|
dm = re.search(r'^## Downloads\n(.*?)(?=^## |\Z)', c, re.M | re.S)
|
|
if dm:
|
|
rows = re.findall(r'^\|\s*(EN|RU\??|\?)\s*\|\s*([0-9]{2}-[a-zA-Z0-9_\.\-]+)\s*\|', dm.group(1), re.M)
|
|
disk = set(os.path.basename(f) for f in glob.glob(f"downloads/{sec_of(path)}/*"))
|
|
manifest = set(f for f, _, _ in manifest_files.get(num, []) if f in disk) # skip legacy/renamed rows
|
|
for lang, fname in rows:
|
|
if fname not in disk:
|
|
issues.append(f'Downloads file not on disk: {fname}')
|
|
if manifest and fname not in manifest:
|
|
issues.append(f'Downloads file not in MANIFEST: {fname}')
|
|
if manifest and set(f for _, f in rows) != set(manifest):
|
|
extra = set(manifest) - set(f for _, f in rows)
|
|
missing = set(f for _, f in rows) - manifest
|
|
if extra: issues.append(f'MANIFEST files missing from card: {sorted(extra)}')
|
|
if missing: issues.append(f'card files not in MANIFEST: {sorted(missing)}')
|
|
return issues
|
|
|
|
def lint_index_status(section_dir):
|
|
"""INDEX.md status column must match card Status."""
|
|
idx = os.path.join(section_dir, 'INDEX.md')
|
|
if not os.path.exists(idx):
|
|
return ['no INDEX.md']
|
|
issues = []
|
|
idx_text = open(idx, encoding='utf-8').read()
|
|
for p in sorted(glob.glob(f'{section_dir}/[0-9][0-9]-*.md')):
|
|
num = os.path.basename(p)[:2]
|
|
card_status = re.search(r'\*\*Status:\*\* (.)', open(p).read())
|
|
card_m = card_status.group(1) if card_status else None
|
|
row = re.search(rf'^\|\s*{num}\b.*?((?:✅|🔶|❌|⬜|🔎)[^\n|]*)', idx_text, re.M)
|
|
if row:
|
|
idx_m = row.group(1).strip()[:1]
|
|
if card_m and idx_m != card_m:
|
|
issues.append(f'INDEX {num} status {idx_m} ≠ card {card_m}')
|
|
return issues
|
|
|
|
def main():
|
|
secs = sys.argv[1:] or sorted(os.path.basename(p) for p in glob.glob('sections/0*'))
|
|
total = 0
|
|
for sec in secs:
|
|
secdir = f'sections/{sec}'
|
|
if not os.path.isdir(secdir):
|
|
continue
|
|
mfiles = _mig.parse_manifest_files(f'downloads/{sec}/MANIFEST.md')
|
|
ncards = 0
|
|
for p in sorted(glob.glob(f'{secdir}/[0-9][0-9]-*.md')):
|
|
ncards += 1
|
|
for i in lint_card(p, mfiles):
|
|
total += 1
|
|
print(f'{p}: {i}')
|
|
if ncards == 0:
|
|
print(f'{sec}: 0 cards (skipped)')
|
|
continue
|
|
for i in lint_index_status(secdir):
|
|
total += 1
|
|
print(f'{secdir}/INDEX: {i}')
|
|
print(f'{sec}: {ncards} cards linted')
|
|
print(f'\nTOTAL issues: {total}')
|
|
sys.exit(1 if total else 0)
|
|
|
|
if __name__ == '__main__':
|
|
main()
|