v3 migration: sec04 #22 — ES section + Original field (Spanish original)

This commit is contained in:
Dmitry Kokorin 2026-09-21 11:09:57 +03:00
parent 54b01b8002
commit 4f5372ab56
3 changed files with 353 additions and 1 deletions

123
tools/check-md.py Normal file
View file

@ -0,0 +1,123 @@
#!/usr/bin/env python3
"""Lint book cards against CANON v3 (see AGENTS.md).
Usage: python3 tools/check-md.py [section ...] (default: all)
Exit code: 0 = clean, 1 = issues found."""
import re, sys, os, glob
ALLOWED_SECTIONS = re.compile(r'^## (Editions — [A-Z]{2}|Downloads|Catalog records|Notes)$')
BANNED_SECTIONS = ('## Russian editions', '## Libgen', '## RSL records', '## Verdict',
'## Identity check', '## Search log')
BANNED_TEXTS = ('// not searched yet', '(not searched yet)', '// not searched',
'**Publisher (EN):**', '**EN ISBN:**', '**Reading list item(s):**')
VALID_MARKERS = ('✅', '🔶', '❌', '⬜', '🔎')
ORDER = ['ED', 'DL', 'CR', 'NT']
def sec_of(path):
return os.path.basename(os.path.dirname(path))
def lint_card(path, manifest_files):
c = open(path, encoding='utf-8').read()
issues = []
num = os.path.basename(path)[:2]
if not c.startswith('# '):
issues.append('no H1')
for f in ('**Author(s):**', '**Shelf mark:**', '**Section:**', '**Status:**'):
if f not in c.split('\n\n')[0] + '\n'.join(c.split('\n')[:8]):
issues.append(f'header missing {f}')
sm = re.search(r'\*\*Status:\*\* (.*)', c)
if sm and not sm.group(1).strip().startswith(VALID_MARKERS):
issues.append(f'status marker invalid: {sm.group(1)[:30]!r}')
for t in BANNED_TEXTS:
if t in c:
issues.append(f'banned text: {t}')
# sections
seen = []
for sm2 in re.finditer(r'^## (.+)$', c, re.M):
name = sm2.group(1).strip()
line = f'## {name}'
if not ALLOWED_SECTIONS.match(line):
issues.append(f'non-canonical section: {line}')
key = ('ED' if name.startswith('Editions —') else
'DL' if name == 'Downloads' else
'CR' if name == 'Catalog records' else 'NT')
seen.append(key)
# Notes exactly once (for migrated cards; tolerate 0 for ⬜ cards)
if seen.count('NT') > 1:
issues.append(f'Notes x{seen.count("NT")}')
# order: all ED before DL before CR before NT
pos = {k: [i for i, s in enumerate(seen) if s == k] for k in ORDER}
last = -1
for k in ORDER:
if pos[k]:
if max(pos[k]) < last:
issues.append(f'section order broken at {k}')
break
last = max(pos[k])
# empty sections
if re.search(r'^## [A-Za-z—& ]+\n\n\n## ', c, re.M):
issues.append('empty section body')
# Downloads table cross-check
dm = re.search(r'^## Downloads\n(.*?)(?=^## |\Z)', c, re.M | re.S)
if dm:
rows = re.findall(r'^\|\s*(EN|RU\??|\?)\s*\|\s*([0-9]{2}-[a-zA-Z0-9\.\-]+)\s*\|', dm.group(1), re.M)
disk = set(os.path.basename(f) for f in glob.glob(f"downloads/{sec_of(path)}/*"))
manifest = set(f for f in manifest_files.get(num, []))
for lang, fname in rows:
if fname not in disk:
issues.append(f'Downloads file not on disk: {fname}')
if manifest and fname not in manifest:
issues.append(f'Downloads file not in MANIFEST: {fname}')
if manifest and set(f for _, f in rows) != manifest:
extra = manifest - set(f for _, f in rows)
missing = set(f for _, f in rows) - manifest
if extra: issues.append(f'MANIFEST files missing from card: {sorted(extra)}')
if missing: issues.append(f'card files not in MANIFEST: {sorted(missing)}')
return issues
def lint_index_status(section_dir):
"""INDEX.md status column must match card Status."""
idx = os.path.join(section_dir, 'INDEX.md')
if not os.path.exists(idx):
return ['no INDEX.md']
issues = []
idx_text = open(idx, encoding='utf-8').read()
for p in sorted(glob.glob(f'{section_dir}/[0-9][0-9]-*.md')):
num = os.path.basename(p)[:2]
card_status = re.search(r'\*\*Status:\*\* (.)', open(p).read())
card_m = card_status.group(1) if card_status else None
row = re.search(rf'^\|\s*{num}\b.*?((?:✅|🔶|❌|⬜|🔎)[^\n|]*)', idx_text, re.M)
if row:
idx_m = row.group(1).strip()[:1]
if card_m and idx_m != card_m:
issues.append(f'INDEX {num} status {idx_m} ≠ card {card_m}')
return issues
def main():
secs = sys.argv[1:] or sorted(os.path.basename(p) for p in glob.glob('sections/0*'))
total = 0
for sec in secs:
secdir = f'sections/{sec}'
if not os.path.isdir(secdir):
continue
# manifest file sets
mfiles = {}
mp = f'downloads/{sec}/MANIFEST.md'
if os.path.exists(mp):
mt = open(mp).read()
for f in re.findall(r'([0-9]{2}-[a-zA-Z0-9\.\-]+\.(?:pdf|fb2|epub|doc|docx|chm|djvu|txt|zip|azw3))', mt):
mfiles.setdefault(f[:2], set()).add(f)
ncards = 0
for p in sorted(glob.glob(f'{secdir}/[0-9][0-9]-*.md')):
ncards += 1
for i in lint_card(p, mfiles):
total += 1
print(f'{p}: {i}')
for i in lint_index_status(secdir):
total += 1
print(f'{secdir}/INDEX: {i}')
print(f'{sec}: {ncards} cards linted')
print(f'\nTOTAL issues: {total}')
sys.exit(1 if total else 0)
if __name__ == '__main__':
main()

228
tools/migrate_cards_v3.py Normal file
View file

@ -0,0 +1,228 @@
#!/usr/bin/env python3
"""Migrate book cards to CANON v3 (see AGENTS.md). Lossless where possible.
Usage: python3 tools/migrate_cards_v3.py --section 01-fundamentals [--dry-run NN ...]"""
import re, sys, os, glob, argparse
BOILER_LIBGEN = re.compile(r'^[-*]?\s*\((none|not searched|not searched yet|not checked|not checked yet|trade-only|empty|TBC|нет|—)\)?\s*$', re.I)
SEC_DATES = { # section completion / verification dates (for gate lines)
'01-fundamentals': '2026-07-17', '02-dreams': '2026-07-18',
'03-myths-fairy-tales': '2026-09-20', '04-pictures': '2026-07-15',
}
SEC_NAMES = {
'01-fundamentals': '01 Fundamentals', '02-dreams': '02 Dreams',
'03-myths-fairy-tales': '03 Myths & Fairy Tales', '04-pictures': '04 Pictures',
}
def parse_manifest_files(manifest_path):
"""Return {item_num: [(file, size_str, source), ...]} for a section MANIFEST."""
items = {}
if not os.path.exists(manifest_path):
return items
c = open(manifest_path, encoding='utf-8').read()
fname_re = r'([0-9]{2}-[a-zA-Z0-9\-]+(?:\.[a-zA-Z0-9]+)*\.(?:pdf|fb2|epub|doc|docx|chm|djvu|txt|zip))'
# sec01/02: table rows | file | size | source | note |
for m in re.finditer(r'^\|\s*' + fname_re + r'\s*\|\s*([^|]+)\|\s*([^|]+)\|', c, re.M):
f, size, src = m.group(1), m.group(2).strip(), m.group(3).strip()
if re.search(r'-(ru|en)[.\-]', f):
items.setdefault(f[:2], []).append((f, size, src))
# sec03: per-item subsections with bullets
for m in re.finditer(r'^### (\d{2}) —.*?\n(.*?)(?=\n### |\n## |\Z)', c, re.M | re.S):
num, body = m.group(1), m.group(2)
for b in re.finditer(r'^- `([^`]+)` \(([^)]+)\)', body, re.M):
fname, size = b.group(1), b.group(2)
items.setdefault(fname[:2], []).append((fname, size, 'MANIFEST'))
# sec04: per-item tables | file | bytes | ed/f | verify |
for m in re.finditer(r'^## (\d{2}) —.*?\n(.*?)(?=\n## |\Z)', c, re.M | re.S):
num, body = m.group(1), m.group(2)
for row in re.finditer(r'^\|\s*' + fname_re + r'\s*\|\s*(\d[\d ]*?)\s*\|\s*([^|]+)\|', body, re.M):
fname, sz, ids = row.group(1), f"{int(row.group(2).replace(' ', ''))//1048576} MB", row.group(3).strip()
fids = re.search(r'/(\d+)$', ids)
src = f'lg f/{fids.group(1)}' if fids else ids
items.setdefault(fname[:2], []).append((fname, sz, src))
# dedupe by filename; prefer human-readable source (lg f/N, flib b/N, ia, MANIFEST)
good = re.compile(r'^(lg f|flib b|ia |MANIFEST|CarlJung)')
for num, lst in items.items():
seen = {}
for f, s, r in lst:
if f not in seen or (good.match(r) and not good.match(seen[f][2])):
seen[f] = (f, s, r)
items[num] = list(seen.values())
return items
def h1_title(c):
m = re.match(r'^# (.*)$', c, re.M)
return m.group(1).strip() if m else ''
def is_dash(v):
if v is None:
return True
v = v.strip()
return v in ('—', '-', '— (Phase 0: to resolve)', '') or v.startswith('— (Phase 0')
def human_size(s):
s = s.strip()
m = re.match(r'^([\d\s]+)$', s)
if m:
b = int(m.group(1).replace(' ', ''))
return f'{b//1048576} MB' if b >= 1048576 else f'{b//1024} KB'
return s
def migrate_card(path, files_by_item, sec, dry=False):
c = open(path, encoding='utf-8').read()
num = os.path.basename(path)[:2]
# split header (before first ##) and sections
m = re.search(r'\n## ', c)
header, rest = c[:m.start()], c[m.start()+1:]
def fld(name):
mm = re.search(rf'\*\*{re.escape(name)}:\*\* (.*)', header)
return mm.group(1).strip() if mm else None
title = h1_title(c)
author = fld('Author(s)') or '—'
shelf = fld('Shelf mark')
section = fld('Section')
status = fld('Status') or '⬜'
pub_en = fld('Publisher (EN)')
isbn_en = fld('EN ISBN')
rli = fld('Reading list item(s)')
notes_extra = []
if rli:
notes_extra.append(f'- Reading list item(s): {rli}')
# --- EN editions table
en_block = ''
if not is_dash(pub_en) or (isbn_en and not is_dash(isbn_en)):
pub = pub_en or '—'
year = '—'
mm = re.match(r'^(.*),\s*(\d{4}[a-z]?)\s*(\(.*\))?$', pub)
if mm:
pub, year = mm.group(1).strip(), mm.group(2)
isbn = '—' if is_dash(isbn_en) else isbn_en
en_block = (
'\n## Editions — EN\n\n'
'| Title / edition | Publisher | Year | Pages | ISBN | Notes |\n'
'|-----------------|-----------|------|-------|------|-------|\n'
f'| {title} | {pub} | {year} | — | {isbn} | |\n')
# --- gather sections
secs = {}
for sm in re.finditer(r'^## (.+?)\n(.*?)(?=^## |\Z)', rest, re.M | re.S):
secs[sm.group(1).strip()] = sm.group(2).strip('\n')
ru_body = secs.get('Russian editions')
libgen_body = secs.get('Libgen')
rsl_body = secs.get('RSL records')
notes_body = secs.get('Notes', '')
# --- Editions — RU
ru_block = ''
if ru_body is not None:
tb = re.search(r'\| RU title \|.*?(?=\n\n|\Z)', ru_body, re.S)
if tb:
tbl = tb.group(0).rstrip()
data_rows = [l for l in tbl.split('\n') if l.startswith('|') and 'RU title' not in l
and not re.match(r'^\|[-\s|]+\|$', l) and '(not searched' not in l
and not re.match(r'^\|\s*\(?\s*(none|n/a|—|-)\s*\)?\s*\|', l)]
if data_rows:
ru_block = '\n## Editions — RU\n\n' + tbl + '\n'
else:
vdate = re.search(r'verified (\d{4}-\d{2}-\d{2})', status)
vdate = vdate.group(1) if vdate else SEC_DATES.get(sec, '')
ru_block = f'\n## Editions — RU\n\n— нет (verified {vdate}, gate: MISSING.md §{num})\n'
else:
vdate = re.search(r'verified (\d{4}-\d{2}-\d{2})', status)
vdate = vdate.group(1) if vdate else SEC_DATES.get(sec, '')
ru_block = f'\n## Editions — RU\n\n— нет (verified {vdate}, gate: MISSING.md §{num})\n'
# --- Downloads
dl_block = ''
files = files_by_item.get(num, [])
if files:
rows = []
for fname, size, src in files:
lang = 'RU' if re.search(r'-ru[.\-]', fname) else ('EN' if re.search(r'-en[.\-]', fname) else '?')
rows.append(f'| {lang} | {fname} | {human_size(size)} | {src} |')
dl_block = ('## Downloads\n\n| lang | file | size | source |\n'
'|------|------|------|--------|\n' + '\n'.join(rows) + '\n')
# libgen body residue → Notes (ids not covered by the table, or no table at all)
if libgen_body:
src_ids = set(re.findall(r'(?:flib b|lg f)/\d+', ' '.join(r for _, _, r in files)))
keep = []
for l in libgen_body.split('\n'):
s = l.strip()
if not s or BOILER_LIBGEN.match(s):
continue
if re.match(r'^(?:- )?(RU|EN) (file|files):', s):
ids = set(re.findall(r'(?:flib b|lg f)/\d+', s))
if not ids or ids <= src_ids:
continue # no info beyond the table
keep.append(l)
if keep:
notes_extra.append('- (was Libgen): ' + ' '.join(k.lstrip('- ').strip() for k in keep))
# --- Catalog records
cat_block = ''
if rsl_body is not None:
body = rsl_body.strip()
inner = re.sub(r'^```json\n?|```$', '', body).strip()
if not inner or inner == '// not searched yet' or inner == '// not searched':
sweep = f'data/sweeps/{sec}/{num}-{os.path.basename(path)[:-3]}.md'
if os.path.exists(sweep):
inner = f'РГБ: (свип: {sweep})'
else:
vdate = re.search(r'verified (\d{4}-\d{2}-\d{2})', status)
inner = f'no records (verified {vdate.group(1) if vdate else SEC_DATES.get(sec, "")})'
else:
inner = re.sub(r'^//\s*', 'РГБ: ', inner, count=1)
if not inner.startswith(('РГБ', 'no records')):
inner = 'РГБ: ' + inner
cat_block = f'## Catalog records\n\n```\n{inner}\n```\n'
# --- Notes (merge extra sections)
extra_secs = []
for name in list(secs):
if name in ('Russian editions', 'Libgen', 'RSL records', 'Notes'):
continue
extra_secs.append(f'{name}: {secs[name].strip()}')
notes_lines = [l for l in notes_body.split('\n')
if l.strip() != '- (empty)' and not re.match(r'^-?.*not searched yet\.?\s*$', l.strip())]
notes = '\n'.join(notes_lines + notes_extra + [f'- (was {e})' for e in extra_secs]).strip()
# --- assemble
out = f'# {title}\n\n'
out += f'**Author(s):** {author}\n'
out += f'**Shelf mark:** {shelf if shelf is not None else "—"}\n'
out += f'**Section:** {section if section is not None else SEC_NAMES.get(sec, sec)}\n'
out += f'**Status:** {status}\n'
out += en_block
out += ru_block
if dl_block:
out += '\n' + dl_block
if cat_block:
out += '\n' + cat_block
if notes:
out += f'\n## Notes\n\n{notes}\n'
if dry:
print('=' * 70)
print(f'DRY-RUN {path}')
print(out)
return
open(path, 'w', encoding='utf-8').write(out)
def main():
ap = argparse.ArgumentParser()
ap.add_argument('--section', required=True)
ap.add_argument('--dry-run', nargs='*', default=None)
a = ap.parse_args()
secdir = f'sections/{a.section}'
man = f'downloads/{a.section}/MANIFEST.md'
files = parse_manifest_files(man)
cards = sorted(glob.glob(f'{secdir}/[0-9][0-9]-*.md'))
for p in cards:
if a.dry_run is not None and os.path.basename(p)[:2] not in a.dry_run:
continue
migrate_card(p, files, a.section, dry=(a.dry_run is not None))
if not a.dry_run:
print(f'migrated {len(cards)} cards in {secdir}')
if __name__ == '__main__':
main()