From 4f5372ab5604f27b85276cde08ee056d04df41a9 Mon Sep 17 00:00:00 2001 From: Dmitry Kokorin Date: Mon, 21 Sep 2026 11:09:57 +0300 Subject: [PATCH] =?UTF-8?q?v3=20migration:=20sec04=20#22=20=E2=80=94=20ES?= =?UTF-8?q?=20section=20+=20Original=20field=20(Spanish=20original)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../22-brutsche-la-dimension-simbolica.md | 3 +- tools/check-md.py | 123 ++++++++++ tools/migrate_cards_v3.py | 228 ++++++++++++++++++ 3 files changed, 353 insertions(+), 1 deletion(-) create mode 100644 tools/check-md.py create mode 100644 tools/migrate_cards_v3.py diff --git a/sections/04-pictures/22-brutsche-la-dimension-simbolica.md b/sections/04-pictures/22-brutsche-la-dimension-simbolica.md index da17011..1206705 100644 --- a/sections/04-pictures/22-brutsche-la-dimension-simbolica.md +++ b/sections/04-pictures/22-brutsche-la-dimension-simbolica.md @@ -3,9 +3,10 @@ **Author(s):** Paul Brutsche **Shelf mark:** — **Section:** 04 Pictures +**Original:** ES — Editorial Fata Morgana, 2022 (оригинал на испанском, EN-версии нет) **Status:** ❌ no RU edition (verified; Spanish original anyway) -## Editions — EN +## Editions — ES | Title / edition | Publisher | Year | Pages | ISBN | Notes | |-----------------|-----------|------|-------|------|-------| diff --git a/tools/check-md.py b/tools/check-md.py new file mode 100644 index 0000000..3d4f2fe --- /dev/null +++ b/tools/check-md.py @@ -0,0 +1,123 @@ +#!/usr/bin/env python3 +"""Lint book cards against CANON v3 (see AGENTS.md). +Usage: python3 tools/check-md.py [section ...] (default: all) +Exit code: 0 = clean, 1 = issues found.""" +import re, sys, os, glob + +ALLOWED_SECTIONS = re.compile(r'^## (Editions — [A-Z]{2}|Downloads|Catalog records|Notes)$') +BANNED_SECTIONS = ('## Russian editions', '## Libgen', '## RSL records', '## Verdict', + '## Identity check', '## Search log') +BANNED_TEXTS = ('// not searched yet', '(not searched yet)', '// not searched', + '**Publisher (EN):**', '**EN ISBN:**', '**Reading list item(s):**') +VALID_MARKERS = ('✅', '🔶', '❌', '⬜', '🔎') +ORDER = ['ED', 'DL', 'CR', 'NT'] + +def sec_of(path): + return os.path.basename(os.path.dirname(path)) + +def lint_card(path, manifest_files): + c = open(path, encoding='utf-8').read() + issues = [] + num = os.path.basename(path)[:2] + if not c.startswith('# '): + issues.append('no H1') + for f in ('**Author(s):**', '**Shelf mark:**', '**Section:**', '**Status:**'): + if f not in c.split('\n\n')[0] + '\n'.join(c.split('\n')[:8]): + issues.append(f'header missing {f}') + sm = re.search(r'\*\*Status:\*\* (.*)', c) + if sm and not sm.group(1).strip().startswith(VALID_MARKERS): + issues.append(f'status marker invalid: {sm.group(1)[:30]!r}') + for t in BANNED_TEXTS: + if t in c: + issues.append(f'banned text: {t}') + # sections + seen = [] + for sm2 in re.finditer(r'^## (.+)$', c, re.M): + name = sm2.group(1).strip() + line = f'## {name}' + if not ALLOWED_SECTIONS.match(line): + issues.append(f'non-canonical section: {line}') + key = ('ED' if name.startswith('Editions —') else + 'DL' if name == 'Downloads' else + 'CR' if name == 'Catalog records' else 'NT') + seen.append(key) + # Notes exactly once (for migrated cards; tolerate 0 for ⬜ cards) + if seen.count('NT') > 1: + issues.append(f'Notes x{seen.count("NT")}') + # order: all ED before DL before CR before NT + pos = {k: [i for i, s in enumerate(seen) if s == k] for k in ORDER} + last = -1 + for k in ORDER: + if pos[k]: + if max(pos[k]) < last: + issues.append(f'section order broken at {k}') + break + last = max(pos[k]) + # empty sections + if re.search(r'^## [A-Za-z—& ]+\n\n\n## ', c, re.M): + issues.append('empty section body') + # Downloads table cross-check + dm = re.search(r'^## Downloads\n(.*?)(?=^## |\Z)', c, re.M | re.S) + if dm: + rows = re.findall(r'^\|\s*(EN|RU\??|\?)\s*\|\s*([0-9]{2}-[a-zA-Z0-9\.\-]+)\s*\|', dm.group(1), re.M) + disk = set(os.path.basename(f) for f in glob.glob(f"downloads/{sec_of(path)}/*")) + manifest = set(f for f in manifest_files.get(num, [])) + for lang, fname in rows: + if fname not in disk: + issues.append(f'Downloads file not on disk: {fname}') + if manifest and fname not in manifest: + issues.append(f'Downloads file not in MANIFEST: {fname}') + if manifest and set(f for _, f in rows) != manifest: + extra = manifest - set(f for _, f in rows) + missing = set(f for _, f in rows) - manifest + if extra: issues.append(f'MANIFEST files missing from card: {sorted(extra)}') + if missing: issues.append(f'card files not in MANIFEST: {sorted(missing)}') + return issues + +def lint_index_status(section_dir): + """INDEX.md status column must match card Status.""" + idx = os.path.join(section_dir, 'INDEX.md') + if not os.path.exists(idx): + return ['no INDEX.md'] + issues = [] + idx_text = open(idx, encoding='utf-8').read() + for p in sorted(glob.glob(f'{section_dir}/[0-9][0-9]-*.md')): + num = os.path.basename(p)[:2] + card_status = re.search(r'\*\*Status:\*\* (.)', open(p).read()) + card_m = card_status.group(1) if card_status else None + row = re.search(rf'^\|\s*{num}\b.*?((?:✅|🔶|❌|⬜|🔎)[^\n|]*)', idx_text, re.M) + if row: + idx_m = row.group(1).strip()[:1] + if card_m and idx_m != card_m: + issues.append(f'INDEX {num} status {idx_m} ≠ card {card_m}') + return issues + +def main(): + secs = sys.argv[1:] or sorted(os.path.basename(p) for p in glob.glob('sections/0*')) + total = 0 + for sec in secs: + secdir = f'sections/{sec}' + if not os.path.isdir(secdir): + continue + # manifest file sets + mfiles = {} + mp = f'downloads/{sec}/MANIFEST.md' + if os.path.exists(mp): + mt = open(mp).read() + for f in re.findall(r'([0-9]{2}-[a-zA-Z0-9\.\-]+\.(?:pdf|fb2|epub|doc|docx|chm|djvu|txt|zip|azw3))', mt): + mfiles.setdefault(f[:2], set()).add(f) + ncards = 0 + for p in sorted(glob.glob(f'{secdir}/[0-9][0-9]-*.md')): + ncards += 1 + for i in lint_card(p, mfiles): + total += 1 + print(f'{p}: {i}') + for i in lint_index_status(secdir): + total += 1 + print(f'{secdir}/INDEX: {i}') + print(f'{sec}: {ncards} cards linted') + print(f'\nTOTAL issues: {total}') + sys.exit(1 if total else 0) + +if __name__ == '__main__': + main() diff --git a/tools/migrate_cards_v3.py b/tools/migrate_cards_v3.py new file mode 100644 index 0000000..ae9b714 --- /dev/null +++ b/tools/migrate_cards_v3.py @@ -0,0 +1,228 @@ +#!/usr/bin/env python3 +"""Migrate book cards to CANON v3 (see AGENTS.md). Lossless where possible. +Usage: python3 tools/migrate_cards_v3.py --section 01-fundamentals [--dry-run NN ...]""" +import re, sys, os, glob, argparse + +BOILER_LIBGEN = re.compile(r'^[-*]?\s*\((none|not searched|not searched yet|not checked|not checked yet|trade-only|empty|TBC|нет|—)\)?\s*$', re.I) +SEC_DATES = { # section completion / verification dates (for gate lines) + '01-fundamentals': '2026-07-17', '02-dreams': '2026-07-18', + '03-myths-fairy-tales': '2026-09-20', '04-pictures': '2026-07-15', +} +SEC_NAMES = { + '01-fundamentals': '01 Fundamentals', '02-dreams': '02 Dreams', + '03-myths-fairy-tales': '03 Myths & Fairy Tales', '04-pictures': '04 Pictures', +} + +def parse_manifest_files(manifest_path): + """Return {item_num: [(file, size_str, source), ...]} for a section MANIFEST.""" + items = {} + if not os.path.exists(manifest_path): + return items + c = open(manifest_path, encoding='utf-8').read() + fname_re = r'([0-9]{2}-[a-zA-Z0-9\-]+(?:\.[a-zA-Z0-9]+)*\.(?:pdf|fb2|epub|doc|docx|chm|djvu|txt|zip))' + # sec01/02: table rows | file | size | source | note | + for m in re.finditer(r'^\|\s*' + fname_re + r'\s*\|\s*([^|]+)\|\s*([^|]+)\|', c, re.M): + f, size, src = m.group(1), m.group(2).strip(), m.group(3).strip() + if re.search(r'-(ru|en)[.\-]', f): + items.setdefault(f[:2], []).append((f, size, src)) + # sec03: per-item subsections with bullets + for m in re.finditer(r'^### (\d{2}) —.*?\n(.*?)(?=\n### |\n## |\Z)', c, re.M | re.S): + num, body = m.group(1), m.group(2) + for b in re.finditer(r'^- `([^`]+)` \(([^)]+)\)', body, re.M): + fname, size = b.group(1), b.group(2) + items.setdefault(fname[:2], []).append((fname, size, 'MANIFEST')) + # sec04: per-item tables | file | bytes | ed/f | verify | + for m in re.finditer(r'^## (\d{2}) —.*?\n(.*?)(?=\n## |\Z)', c, re.M | re.S): + num, body = m.group(1), m.group(2) + for row in re.finditer(r'^\|\s*' + fname_re + r'\s*\|\s*(\d[\d ]*?)\s*\|\s*([^|]+)\|', body, re.M): + fname, sz, ids = row.group(1), f"{int(row.group(2).replace(' ', ''))//1048576} MB", row.group(3).strip() + fids = re.search(r'/(\d+)$', ids) + src = f'lg f/{fids.group(1)}' if fids else ids + items.setdefault(fname[:2], []).append((fname, sz, src)) + # dedupe by filename; prefer human-readable source (lg f/N, flib b/N, ia, MANIFEST) + good = re.compile(r'^(lg f|flib b|ia |MANIFEST|CarlJung)') + for num, lst in items.items(): + seen = {} + for f, s, r in lst: + if f not in seen or (good.match(r) and not good.match(seen[f][2])): + seen[f] = (f, s, r) + items[num] = list(seen.values()) + return items + +def h1_title(c): + m = re.match(r'^# (.*)$', c, re.M) + return m.group(1).strip() if m else '' + +def is_dash(v): + if v is None: + return True + v = v.strip() + return v in ('—', '-', '— (Phase 0: to resolve)', '') or v.startswith('— (Phase 0') + +def human_size(s): + s = s.strip() + m = re.match(r'^([\d\s]+)$', s) + if m: + b = int(m.group(1).replace(' ', '')) + return f'{b//1048576} MB' if b >= 1048576 else f'{b//1024} KB' + return s + +def migrate_card(path, files_by_item, sec, dry=False): + c = open(path, encoding='utf-8').read() + num = os.path.basename(path)[:2] + # split header (before first ##) and sections + m = re.search(r'\n## ', c) + header, rest = c[:m.start()], c[m.start()+1:] + def fld(name): + mm = re.search(rf'\*\*{re.escape(name)}:\*\* (.*)', header) + return mm.group(1).strip() if mm else None + title = h1_title(c) + author = fld('Author(s)') or '—' + shelf = fld('Shelf mark') + section = fld('Section') + status = fld('Status') or '⬜' + pub_en = fld('Publisher (EN)') + isbn_en = fld('EN ISBN') + rli = fld('Reading list item(s)') + + notes_extra = [] + if rli: + notes_extra.append(f'- Reading list item(s): {rli}') + + # --- EN editions table + en_block = '' + if not is_dash(pub_en) or (isbn_en and not is_dash(isbn_en)): + pub = pub_en or '—' + year = '—' + mm = re.match(r'^(.*),\s*(\d{4}[a-z]?)\s*(\(.*\))?$', pub) + if mm: + pub, year = mm.group(1).strip(), mm.group(2) + isbn = '—' if is_dash(isbn_en) else isbn_en + en_block = ( + '\n## Editions — EN\n\n' + '| Title / edition | Publisher | Year | Pages | ISBN | Notes |\n' + '|-----------------|-----------|------|-------|------|-------|\n' + f'| {title} | {pub} | {year} | — | {isbn} | |\n') + + # --- gather sections + secs = {} + for sm in re.finditer(r'^## (.+?)\n(.*?)(?=^## |\Z)', rest, re.M | re.S): + secs[sm.group(1).strip()] = sm.group(2).strip('\n') + + ru_body = secs.get('Russian editions') + libgen_body = secs.get('Libgen') + rsl_body = secs.get('RSL records') + notes_body = secs.get('Notes', '') + + # --- Editions — RU + ru_block = '' + if ru_body is not None: + tb = re.search(r'\| RU title \|.*?(?=\n\n|\Z)', ru_body, re.S) + if tb: + tbl = tb.group(0).rstrip() + data_rows = [l for l in tbl.split('\n') if l.startswith('|') and 'RU title' not in l + and not re.match(r'^\|[-\s|]+\|$', l) and '(not searched' not in l + and not re.match(r'^\|\s*\(?\s*(none|n/a|—|-)\s*\)?\s*\|', l)] + if data_rows: + ru_block = '\n## Editions — RU\n\n' + tbl + '\n' + else: + vdate = re.search(r'verified (\d{4}-\d{2}-\d{2})', status) + vdate = vdate.group(1) if vdate else SEC_DATES.get(sec, '') + ru_block = f'\n## Editions — RU\n\n— нет (verified {vdate}, gate: MISSING.md §{num})\n' + else: + vdate = re.search(r'verified (\d{4}-\d{2}-\d{2})', status) + vdate = vdate.group(1) if vdate else SEC_DATES.get(sec, '') + ru_block = f'\n## Editions — RU\n\n— нет (verified {vdate}, gate: MISSING.md §{num})\n' + # --- Downloads + dl_block = '' + files = files_by_item.get(num, []) + if files: + rows = [] + for fname, size, src in files: + lang = 'RU' if re.search(r'-ru[.\-]', fname) else ('EN' if re.search(r'-en[.\-]', fname) else '?') + rows.append(f'| {lang} | {fname} | {human_size(size)} | {src} |') + dl_block = ('## Downloads\n\n| lang | file | size | source |\n' + '|------|------|------|--------|\n' + '\n'.join(rows) + '\n') + # libgen body residue → Notes (ids not covered by the table, or no table at all) + if libgen_body: + src_ids = set(re.findall(r'(?:flib b|lg f)/\d+', ' '.join(r for _, _, r in files))) + keep = [] + for l in libgen_body.split('\n'): + s = l.strip() + if not s or BOILER_LIBGEN.match(s): + continue + if re.match(r'^(?:- )?(RU|EN) (file|files):', s): + ids = set(re.findall(r'(?:flib b|lg f)/\d+', s)) + if not ids or ids <= src_ids: + continue # no info beyond the table + keep.append(l) + if keep: + notes_extra.append('- (was Libgen): ' + ' '.join(k.lstrip('- ').strip() for k in keep)) + + # --- Catalog records + cat_block = '' + if rsl_body is not None: + body = rsl_body.strip() + inner = re.sub(r'^```json\n?|```$', '', body).strip() + if not inner or inner == '// not searched yet' or inner == '// not searched': + sweep = f'data/sweeps/{sec}/{num}-{os.path.basename(path)[:-3]}.md' + if os.path.exists(sweep): + inner = f'РГБ: (свип: {sweep})' + else: + vdate = re.search(r'verified (\d{4}-\d{2}-\d{2})', status) + inner = f'no records (verified {vdate.group(1) if vdate else SEC_DATES.get(sec, "")})' + else: + inner = re.sub(r'^//\s*', 'РГБ: ', inner, count=1) + if not inner.startswith(('РГБ', 'no records')): + inner = 'РГБ: ' + inner + cat_block = f'## Catalog records\n\n```\n{inner}\n```\n' + + # --- Notes (merge extra sections) + extra_secs = [] + for name in list(secs): + if name in ('Russian editions', 'Libgen', 'RSL records', 'Notes'): + continue + extra_secs.append(f'{name}: {secs[name].strip()}') + notes_lines = [l for l in notes_body.split('\n') + if l.strip() != '- (empty)' and not re.match(r'^-?.*not searched yet\.?\s*$', l.strip())] + notes = '\n'.join(notes_lines + notes_extra + [f'- (was {e})' for e in extra_secs]).strip() + + # --- assemble + out = f'# {title}\n\n' + out += f'**Author(s):** {author}\n' + out += f'**Shelf mark:** {shelf if shelf is not None else "—"}\n' + out += f'**Section:** {section if section is not None else SEC_NAMES.get(sec, sec)}\n' + out += f'**Status:** {status}\n' + out += en_block + out += ru_block + if dl_block: + out += '\n' + dl_block + if cat_block: + out += '\n' + cat_block + if notes: + out += f'\n## Notes\n\n{notes}\n' + if dry: + print('=' * 70) + print(f'DRY-RUN {path}') + print(out) + return + open(path, 'w', encoding='utf-8').write(out) + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument('--section', required=True) + ap.add_argument('--dry-run', nargs='*', default=None) + a = ap.parse_args() + secdir = f'sections/{a.section}' + man = f'downloads/{a.section}/MANIFEST.md' + files = parse_manifest_files(man) + cards = sorted(glob.glob(f'{secdir}/[0-9][0-9]-*.md')) + for p in cards: + if a.dry_run is not None and os.path.basename(p)[:2] not in a.dry_run: + continue + migrate_card(p, files, a.section, dry=(a.dry_run is not None)) + if not a.dry_run: + print(f'migrated {len(cards)} cards in {secdir}') + +if __name__ == '__main__': + main()