jung/tools/migrate_cards_v3.py
Dmitry Kokorin f5cb34195e content audit sec01-06: fixes (audit_content.py + 12 file/card corrections)
- tools/audit_content.py: per-file vs item-title triage (SUSPECT/SMALL/NO-TEXT/NO-CYR)
- sec01: .part removed; 10 files restored .pdf ext (LFS rename); #04/#19/#27 verify notes
- sec02 #16: corrupted double-encoded txt -> clean flib fb2 b/844509; #17/#22 doc verified via catdoc
- sec03 #11: -b epub = Spanish von Franz (Paidós 1983, forged EN OPF) -> bonus label
- sec03 #42: КАРО 2012 'Irish Tales' = EN reader (not RU, not the listed book) -> bonus rename
- sec03 #56: Onians b/c/d/e = Cambridge 24-87KB previews -> labeled
- sec06 #18: filename 1981->1989 (Perera 'Descent to the Goddess' 1989; 1981 = other book)
- sec06 #28: ia Zimmer OCR layer = foreign Devanagari text; images = RKP (p.100 verified)
- sec04 #30: buksmart 2020 = RU/FR bilingual w/ VeryPDF watermarks -> note
- linter: azw3/mobi ext + underscore in fname regex; manifest cross-check ignores off-disk rows;
  8 legacy cards section-order fixed (CR before DL)
- AGENTS.md: 'Content audit (2026-09-25)' lessons section
- check-md: 0/219
2026-09-26 01:32:02 +03:00

228 lines
9.8 KiB
Python

#!/usr/bin/env python3
"""Migrate book cards to CANON v3 (see AGENTS.md). Lossless where possible.
Usage: python3 tools/migrate_cards_v3.py --section 01-fundamentals [--dry-run NN ...]"""
import re, sys, os, glob, argparse
BOILER_LIBGEN = re.compile(r'^[-*]?\s*\((none|not searched|not searched yet|not checked|not checked yet|trade-only|empty|TBC|нет|—)\)?\s*$', re.I)
SEC_DATES = { # section completion / verification dates (for gate lines)
'01-fundamentals': '2026-07-17', '02-dreams': '2026-07-18',
'03-myths-fairy-tales': '2026-09-20', '04-pictures': '2026-07-15',
}
SEC_NAMES = {
'01-fundamentals': '01 Fundamentals', '02-dreams': '02 Dreams',
'03-myths-fairy-tales': '03 Myths & Fairy Tales', '04-pictures': '04 Pictures',
}
def parse_manifest_files(manifest_path):
"""Return {item_num: [(file, size_str, source), ...]} for a section MANIFEST."""
items = {}
if not os.path.exists(manifest_path):
return items
c = open(manifest_path, encoding='utf-8').read()
fname_re = r'([0-9]{2}-[a-zA-Z0-9_\-]+(?:\.[a-zA-Z0-9]+)*\.(?:pdf|fb2|epub|doc|docx|chm|djvu|txt|zip|mobi|azw3))'
# sec01/02: table rows | file | size | source | note |
for m in re.finditer(r'^\|\s*' + fname_re + r'\s*\|\s*([^|]+)\|\s*([^|]+)\|', c, re.M):
f, size, src = m.group(1), m.group(2).strip(), m.group(3).strip()
if re.search(r'-(ru|en)[.\-]', f):
items.setdefault(f[:2], []).append((f, size, src))
# sec03: per-item subsections with bullets
for m in re.finditer(r'^### (\d{2}) —.*?\n(.*?)(?=\n### |\n## |\Z)', c, re.M | re.S):
num, body = m.group(1), m.group(2)
for b in re.finditer(r'^- `([^`]+)` \(([^)]+)\)', body, re.M):
fname, size = b.group(1), b.group(2)
items.setdefault(fname[:2], []).append((fname, size, 'MANIFEST'))
# sec04: per-item tables | file | bytes | ed/f | verify |
for m in re.finditer(r'^## (\d{2}) —.*?\n(.*?)(?=\n## |\Z)', c, re.M | re.S):
num, body = m.group(1), m.group(2)
for row in re.finditer(r'^\|\s*' + fname_re + r'\s*\|\s*(\d[\d ]*?)\s*\|\s*([^|]+)\|', body, re.M):
fname, sz, ids = row.group(1), f"{int(row.group(2).replace(' ', ''))//1048576} MB", row.group(3).strip()
fids = re.search(r'/(\d+)$', ids)
src = f'lg f/{fids.group(1)}' if fids else ids
items.setdefault(fname[:2], []).append((fname, sz, src))
# dedupe by filename; prefer human-readable source (lg f/N, flib b/N, ia, MANIFEST)
good = re.compile(r'^(lg f|flib b|ia |MANIFEST|CarlJung)')
for num, lst in items.items():
seen = {}
for f, s, r in lst:
if f not in seen or (good.match(r) and not good.match(seen[f][2])):
seen[f] = (f, s, r)
items[num] = list(seen.values())
return items
def h1_title(c):
m = re.match(r'^# (.*)$', c, re.M)
return m.group(1).strip() if m else ''
def is_dash(v):
if v is None:
return True
v = v.strip()
return v in ('—', '-', '— (Phase 0: to resolve)', '') or v.startswith('— (Phase 0')
def human_size(s):
s = s.strip()
m = re.match(r'^([\d\s]+)$', s)
if m:
b = int(m.group(1).replace(' ', ''))
return f'{b//1048576} MB' if b >= 1048576 else f'{b//1024} KB'
return s
def migrate_card(path, files_by_item, sec, dry=False):
c = open(path, encoding='utf-8').read()
num = os.path.basename(path)[:2]
# split header (before first ##) and sections
m = re.search(r'\n## ', c)
header, rest = c[:m.start()], c[m.start()+1:]
def fld(name):
mm = re.search(rf'\*\*{re.escape(name)}:\*\* (.*)', header)
return mm.group(1).strip() if mm else None
title = h1_title(c)
author = fld('Author(s)') or '—'
shelf = fld('Shelf mark')
section = fld('Section')
status = fld('Status') or '⬜'
pub_en = fld('Publisher (EN)')
isbn_en = fld('EN ISBN')
rli = fld('Reading list item(s)')
notes_extra = []
if rli:
notes_extra.append(f'- Reading list item(s): {rli}')
# --- EN editions table
en_block = ''
if not is_dash(pub_en) or (isbn_en and not is_dash(isbn_en)):
pub = pub_en or '—'
year = '—'
mm = re.match(r'^(.*),\s*(\d{4}[a-z]?)\s*(\(.*\))?$', pub)
if mm:
pub, year = mm.group(1).strip(), mm.group(2)
isbn = '—' if is_dash(isbn_en) else isbn_en
en_block = (
'\n## Editions — EN\n\n'
'| Title / edition | Publisher | Year | Pages | ISBN | Notes |\n'
'|-----------------|-----------|------|-------|------|-------|\n'
f'| {title} | {pub} | {year} | — | {isbn} | |\n')
# --- gather sections
secs = {}
for sm in re.finditer(r'^## (.+?)\n(.*?)(?=^## |\Z)', rest, re.M | re.S):
secs[sm.group(1).strip()] = sm.group(2).strip('\n')
ru_body = secs.get('Russian editions')
libgen_body = secs.get('Libgen')
rsl_body = secs.get('RSL records')
notes_body = secs.get('Notes', '')
# --- Editions — RU
ru_block = ''
if ru_body is not None:
tb = re.search(r'\| RU title \|.*?(?=\n\n|\Z)', ru_body, re.S)
if tb:
tbl = tb.group(0).rstrip()
data_rows = [l for l in tbl.split('\n') if l.startswith('|') and 'RU title' not in l
and not re.match(r'^\|[-\s|]+\|$', l) and '(not searched' not in l
and not re.match(r'^\|\s*\(?\s*(none|n/a|—|-)\s*\)?\s*\|', l)]
if data_rows:
ru_block = '\n## Editions — RU\n\n' + tbl + '\n'
else:
vdate = re.search(r'verified (\d{4}-\d{2}-\d{2})', status)
vdate = vdate.group(1) if vdate else SEC_DATES.get(sec, '')
ru_block = f'\n## Editions — RU\n\n— нет (verified {vdate}, gate: MISSING.md §{num})\n'
else:
vdate = re.search(r'verified (\d{4}-\d{2}-\d{2})', status)
vdate = vdate.group(1) if vdate else SEC_DATES.get(sec, '')
ru_block = f'\n## Editions — RU\n\n— нет (verified {vdate}, gate: MISSING.md §{num})\n'
# --- Downloads
dl_block = ''
files = files_by_item.get(num, [])
if files:
rows = []
for fname, size, src in files:
lang = 'RU' if re.search(r'-ru[.\-]', fname) else ('EN' if re.search(r'-en[.\-]', fname) else '?')
rows.append(f'| {lang} | {fname} | {human_size(size)} | {src} |')
dl_block = ('## Downloads\n\n| lang | file | size | source |\n'
'|------|------|------|--------|\n' + '\n'.join(rows) + '\n')
# libgen body residue → Notes (ids not covered by the table, or no table at all)
if libgen_body:
src_ids = set(re.findall(r'(?:flib b|lg f)/\d+', ' '.join(r for _, _, r in files)))
keep = []
for l in libgen_body.split('\n'):
s = l.strip()
if not s or BOILER_LIBGEN.match(s):
continue
if re.match(r'^(?:- )?(RU|EN) (file|files):', s):
ids = set(re.findall(r'(?:flib b|lg f)/\d+', s))
if not ids or ids <= src_ids:
continue # no info beyond the table
keep.append(l)
if keep:
notes_extra.append('- (was Libgen): ' + ' '.join(k.lstrip('- ').strip() for k in keep))
# --- Catalog records
cat_block = ''
if rsl_body is not None:
body = rsl_body.strip()
inner = re.sub(r'^```json\n?|```$', '', body).strip()
if not inner or inner == '// not searched yet' or inner == '// not searched':
sweep = f'data/sweeps/{sec}/{num}-{os.path.basename(path)[:-3]}.md'
if os.path.exists(sweep):
inner = f'РГБ: (свип: {sweep})'
else:
vdate = re.search(r'verified (\d{4}-\d{2}-\d{2})', status)
inner = f'no records (verified {vdate.group(1) if vdate else SEC_DATES.get(sec, "")})'
else:
inner = re.sub(r'^//\s*', 'РГБ: ', inner, count=1)
if not inner.startswith(('РГБ', 'no records')):
inner = 'РГБ: ' + inner
cat_block = f'## Catalog records\n\n```\n{inner}\n```\n'
# --- Notes (merge extra sections)
extra_secs = []
for name in list(secs):
if name in ('Russian editions', 'Libgen', 'RSL records', 'Notes'):
continue
extra_secs.append(f'{name}: {secs[name].strip()}')
notes_lines = [l for l in notes_body.split('\n')
if l.strip() != '- (empty)' and not re.match(r'^-?.*not searched yet\.?\s*$', l.strip())]
notes = '\n'.join(notes_lines + notes_extra + [f'- (was {e})' for e in extra_secs]).strip()
# --- assemble
out = f'# {title}\n\n'
out += f'**Author(s):** {author}\n'
out += f'**Shelf mark:** {shelf if shelf is not None else "—"}\n'
out += f'**Section:** {section if section is not None else SEC_NAMES.get(sec, sec)}\n'
out += f'**Status:** {status}\n'
out += en_block
out += ru_block
if dl_block:
out += '\n' + dl_block
if cat_block:
out += '\n' + cat_block
if notes:
out += f'\n## Notes\n\n{notes}\n'
if dry:
print('=' * 70)
print(f'DRY-RUN {path}')
print(out)
return
open(path, 'w', encoding='utf-8').write(out)
def main():
ap = argparse.ArgumentParser()
ap.add_argument('--section', required=True)
ap.add_argument('--dry-run', nargs='*', default=None)
a = ap.parse_args()
secdir = f'sections/{a.section}'
man = f'downloads/{a.section}/MANIFEST.md'
files = parse_manifest_files(man)
cards = sorted(glob.glob(f'{secdir}/[0-9][0-9]-*.md'))
for p in cards:
if a.dry_run is not None and os.path.basename(p)[:2] not in a.dry_run:
continue
migrate_card(p, files, a.section, dry=(a.dry_run is not None))
if not a.dry_run:
print(f'migrated {len(cards)} cards in {secdir}')
if __name__ == '__main__':
main()