- tools/audit_content.py: per-file vs item-title triage (SUSPECT/SMALL/NO-TEXT/NO-CYR) - sec01: .part removed; 10 files restored .pdf ext (LFS rename); #04/#19/#27 verify notes - sec02 #16: corrupted double-encoded txt -> clean flib fb2 b/844509; #17/#22 doc verified via catdoc - sec03 #11: -b epub = Spanish von Franz (Paidós 1983, forged EN OPF) -> bonus label - sec03 #42: КАРО 2012 'Irish Tales' = EN reader (not RU, not the listed book) -> bonus rename - sec03 #56: Onians b/c/d/e = Cambridge 24-87KB previews -> labeled - sec06 #18: filename 1981->1989 (Perera 'Descent to the Goddess' 1989; 1981 = other book) - sec06 #28: ia Zimmer OCR layer = foreign Devanagari text; images = RKP (p.100 verified) - sec04 #30: buksmart 2020 = RU/FR bilingual w/ VeryPDF watermarks -> note - linter: azw3/mobi ext + underscore in fname regex; manifest cross-check ignores off-disk rows; 8 legacy cards section-order fixed (CR before DL) - AGENTS.md: 'Content audit (2026-09-25)' lessons section - check-md: 0/219
228 lines
9.8 KiB
Python
228 lines
9.8 KiB
Python
#!/usr/bin/env python3
|
|
"""Migrate book cards to CANON v3 (see AGENTS.md). Lossless where possible.
|
|
Usage: python3 tools/migrate_cards_v3.py --section 01-fundamentals [--dry-run NN ...]"""
|
|
import re, sys, os, glob, argparse
|
|
|
|
BOILER_LIBGEN = re.compile(r'^[-*]?\s*\((none|not searched|not searched yet|not checked|not checked yet|trade-only|empty|TBC|нет|—)\)?\s*$', re.I)
|
|
SEC_DATES = { # section completion / verification dates (for gate lines)
|
|
'01-fundamentals': '2026-07-17', '02-dreams': '2026-07-18',
|
|
'03-myths-fairy-tales': '2026-09-20', '04-pictures': '2026-07-15',
|
|
}
|
|
SEC_NAMES = {
|
|
'01-fundamentals': '01 Fundamentals', '02-dreams': '02 Dreams',
|
|
'03-myths-fairy-tales': '03 Myths & Fairy Tales', '04-pictures': '04 Pictures',
|
|
}
|
|
|
|
def parse_manifest_files(manifest_path):
|
|
"""Return {item_num: [(file, size_str, source), ...]} for a section MANIFEST."""
|
|
items = {}
|
|
if not os.path.exists(manifest_path):
|
|
return items
|
|
c = open(manifest_path, encoding='utf-8').read()
|
|
fname_re = r'([0-9]{2}-[a-zA-Z0-9_\-]+(?:\.[a-zA-Z0-9]+)*\.(?:pdf|fb2|epub|doc|docx|chm|djvu|txt|zip|mobi|azw3))'
|
|
# sec01/02: table rows | file | size | source | note |
|
|
for m in re.finditer(r'^\|\s*' + fname_re + r'\s*\|\s*([^|]+)\|\s*([^|]+)\|', c, re.M):
|
|
f, size, src = m.group(1), m.group(2).strip(), m.group(3).strip()
|
|
if re.search(r'-(ru|en)[.\-]', f):
|
|
items.setdefault(f[:2], []).append((f, size, src))
|
|
# sec03: per-item subsections with bullets
|
|
for m in re.finditer(r'^### (\d{2}) —.*?\n(.*?)(?=\n### |\n## |\Z)', c, re.M | re.S):
|
|
num, body = m.group(1), m.group(2)
|
|
for b in re.finditer(r'^- `([^`]+)` \(([^)]+)\)', body, re.M):
|
|
fname, size = b.group(1), b.group(2)
|
|
items.setdefault(fname[:2], []).append((fname, size, 'MANIFEST'))
|
|
# sec04: per-item tables | file | bytes | ed/f | verify |
|
|
for m in re.finditer(r'^## (\d{2}) —.*?\n(.*?)(?=\n## |\Z)', c, re.M | re.S):
|
|
num, body = m.group(1), m.group(2)
|
|
for row in re.finditer(r'^\|\s*' + fname_re + r'\s*\|\s*(\d[\d ]*?)\s*\|\s*([^|]+)\|', body, re.M):
|
|
fname, sz, ids = row.group(1), f"{int(row.group(2).replace(' ', ''))//1048576} MB", row.group(3).strip()
|
|
fids = re.search(r'/(\d+)$', ids)
|
|
src = f'lg f/{fids.group(1)}' if fids else ids
|
|
items.setdefault(fname[:2], []).append((fname, sz, src))
|
|
# dedupe by filename; prefer human-readable source (lg f/N, flib b/N, ia, MANIFEST)
|
|
good = re.compile(r'^(lg f|flib b|ia |MANIFEST|CarlJung)')
|
|
for num, lst in items.items():
|
|
seen = {}
|
|
for f, s, r in lst:
|
|
if f not in seen or (good.match(r) and not good.match(seen[f][2])):
|
|
seen[f] = (f, s, r)
|
|
items[num] = list(seen.values())
|
|
return items
|
|
|
|
def h1_title(c):
|
|
m = re.match(r'^# (.*)$', c, re.M)
|
|
return m.group(1).strip() if m else ''
|
|
|
|
def is_dash(v):
|
|
if v is None:
|
|
return True
|
|
v = v.strip()
|
|
return v in ('—', '-', '— (Phase 0: to resolve)', '') or v.startswith('— (Phase 0')
|
|
|
|
def human_size(s):
|
|
s = s.strip()
|
|
m = re.match(r'^([\d\s]+)$', s)
|
|
if m:
|
|
b = int(m.group(1).replace(' ', ''))
|
|
return f'{b//1048576} MB' if b >= 1048576 else f'{b//1024} KB'
|
|
return s
|
|
|
|
def migrate_card(path, files_by_item, sec, dry=False):
|
|
c = open(path, encoding='utf-8').read()
|
|
num = os.path.basename(path)[:2]
|
|
# split header (before first ##) and sections
|
|
m = re.search(r'\n## ', c)
|
|
header, rest = c[:m.start()], c[m.start()+1:]
|
|
def fld(name):
|
|
mm = re.search(rf'\*\*{re.escape(name)}:\*\* (.*)', header)
|
|
return mm.group(1).strip() if mm else None
|
|
title = h1_title(c)
|
|
author = fld('Author(s)') or '—'
|
|
shelf = fld('Shelf mark')
|
|
section = fld('Section')
|
|
status = fld('Status') or '⬜'
|
|
pub_en = fld('Publisher (EN)')
|
|
isbn_en = fld('EN ISBN')
|
|
rli = fld('Reading list item(s)')
|
|
|
|
notes_extra = []
|
|
if rli:
|
|
notes_extra.append(f'- Reading list item(s): {rli}')
|
|
|
|
# --- EN editions table
|
|
en_block = ''
|
|
if not is_dash(pub_en) or (isbn_en and not is_dash(isbn_en)):
|
|
pub = pub_en or '—'
|
|
year = '—'
|
|
mm = re.match(r'^(.*),\s*(\d{4}[a-z]?)\s*(\(.*\))?$', pub)
|
|
if mm:
|
|
pub, year = mm.group(1).strip(), mm.group(2)
|
|
isbn = '—' if is_dash(isbn_en) else isbn_en
|
|
en_block = (
|
|
'\n## Editions — EN\n\n'
|
|
'| Title / edition | Publisher | Year | Pages | ISBN | Notes |\n'
|
|
'|-----------------|-----------|------|-------|------|-------|\n'
|
|
f'| {title} | {pub} | {year} | — | {isbn} | |\n')
|
|
|
|
# --- gather sections
|
|
secs = {}
|
|
for sm in re.finditer(r'^## (.+?)\n(.*?)(?=^## |\Z)', rest, re.M | re.S):
|
|
secs[sm.group(1).strip()] = sm.group(2).strip('\n')
|
|
|
|
ru_body = secs.get('Russian editions')
|
|
libgen_body = secs.get('Libgen')
|
|
rsl_body = secs.get('RSL records')
|
|
notes_body = secs.get('Notes', '')
|
|
|
|
# --- Editions — RU
|
|
ru_block = ''
|
|
if ru_body is not None:
|
|
tb = re.search(r'\| RU title \|.*?(?=\n\n|\Z)', ru_body, re.S)
|
|
if tb:
|
|
tbl = tb.group(0).rstrip()
|
|
data_rows = [l for l in tbl.split('\n') if l.startswith('|') and 'RU title' not in l
|
|
and not re.match(r'^\|[-\s|]+\|$', l) and '(not searched' not in l
|
|
and not re.match(r'^\|\s*\(?\s*(none|n/a|—|-)\s*\)?\s*\|', l)]
|
|
if data_rows:
|
|
ru_block = '\n## Editions — RU\n\n' + tbl + '\n'
|
|
else:
|
|
vdate = re.search(r'verified (\d{4}-\d{2}-\d{2})', status)
|
|
vdate = vdate.group(1) if vdate else SEC_DATES.get(sec, '')
|
|
ru_block = f'\n## Editions — RU\n\n— нет (verified {vdate}, gate: MISSING.md §{num})\n'
|
|
else:
|
|
vdate = re.search(r'verified (\d{4}-\d{2}-\d{2})', status)
|
|
vdate = vdate.group(1) if vdate else SEC_DATES.get(sec, '')
|
|
ru_block = f'\n## Editions — RU\n\n— нет (verified {vdate}, gate: MISSING.md §{num})\n'
|
|
# --- Downloads
|
|
dl_block = ''
|
|
files = files_by_item.get(num, [])
|
|
if files:
|
|
rows = []
|
|
for fname, size, src in files:
|
|
lang = 'RU' if re.search(r'-ru[.\-]', fname) else ('EN' if re.search(r'-en[.\-]', fname) else '?')
|
|
rows.append(f'| {lang} | {fname} | {human_size(size)} | {src} |')
|
|
dl_block = ('## Downloads\n\n| lang | file | size | source |\n'
|
|
'|------|------|------|--------|\n' + '\n'.join(rows) + '\n')
|
|
# libgen body residue → Notes (ids not covered by the table, or no table at all)
|
|
if libgen_body:
|
|
src_ids = set(re.findall(r'(?:flib b|lg f)/\d+', ' '.join(r for _, _, r in files)))
|
|
keep = []
|
|
for l in libgen_body.split('\n'):
|
|
s = l.strip()
|
|
if not s or BOILER_LIBGEN.match(s):
|
|
continue
|
|
if re.match(r'^(?:- )?(RU|EN) (file|files):', s):
|
|
ids = set(re.findall(r'(?:flib b|lg f)/\d+', s))
|
|
if not ids or ids <= src_ids:
|
|
continue # no info beyond the table
|
|
keep.append(l)
|
|
if keep:
|
|
notes_extra.append('- (was Libgen): ' + ' '.join(k.lstrip('- ').strip() for k in keep))
|
|
|
|
# --- Catalog records
|
|
cat_block = ''
|
|
if rsl_body is not None:
|
|
body = rsl_body.strip()
|
|
inner = re.sub(r'^```json\n?|```$', '', body).strip()
|
|
if not inner or inner == '// not searched yet' or inner == '// not searched':
|
|
sweep = f'data/sweeps/{sec}/{num}-{os.path.basename(path)[:-3]}.md'
|
|
if os.path.exists(sweep):
|
|
inner = f'РГБ: (свип: {sweep})'
|
|
else:
|
|
vdate = re.search(r'verified (\d{4}-\d{2}-\d{2})', status)
|
|
inner = f'no records (verified {vdate.group(1) if vdate else SEC_DATES.get(sec, "")})'
|
|
else:
|
|
inner = re.sub(r'^//\s*', 'РГБ: ', inner, count=1)
|
|
if not inner.startswith(('РГБ', 'no records')):
|
|
inner = 'РГБ: ' + inner
|
|
cat_block = f'## Catalog records\n\n```\n{inner}\n```\n'
|
|
|
|
# --- Notes (merge extra sections)
|
|
extra_secs = []
|
|
for name in list(secs):
|
|
if name in ('Russian editions', 'Libgen', 'RSL records', 'Notes'):
|
|
continue
|
|
extra_secs.append(f'{name}: {secs[name].strip()}')
|
|
notes_lines = [l for l in notes_body.split('\n')
|
|
if l.strip() != '- (empty)' and not re.match(r'^-?.*not searched yet\.?\s*$', l.strip())]
|
|
notes = '\n'.join(notes_lines + notes_extra + [f'- (was {e})' for e in extra_secs]).strip()
|
|
|
|
# --- assemble
|
|
out = f'# {title}\n\n'
|
|
out += f'**Author(s):** {author}\n'
|
|
out += f'**Shelf mark:** {shelf if shelf is not None else "—"}\n'
|
|
out += f'**Section:** {section if section is not None else SEC_NAMES.get(sec, sec)}\n'
|
|
out += f'**Status:** {status}\n'
|
|
out += en_block
|
|
out += ru_block
|
|
if dl_block:
|
|
out += '\n' + dl_block
|
|
if cat_block:
|
|
out += '\n' + cat_block
|
|
if notes:
|
|
out += f'\n## Notes\n\n{notes}\n'
|
|
if dry:
|
|
print('=' * 70)
|
|
print(f'DRY-RUN {path}')
|
|
print(out)
|
|
return
|
|
open(path, 'w', encoding='utf-8').write(out)
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument('--section', required=True)
|
|
ap.add_argument('--dry-run', nargs='*', default=None)
|
|
a = ap.parse_args()
|
|
secdir = f'sections/{a.section}'
|
|
man = f'downloads/{a.section}/MANIFEST.md'
|
|
files = parse_manifest_files(man)
|
|
cards = sorted(glob.glob(f'{secdir}/[0-9][0-9]-*.md'))
|
|
for p in cards:
|
|
if a.dry_run is not None and os.path.basename(p)[:2] not in a.dry_run:
|
|
continue
|
|
migrate_card(p, files, a.section, dry=(a.dry_run is not None))
|
|
if not a.dry_run:
|
|
print(f'migrated {len(cards)} cards in {secdir}')
|
|
|
|
if __name__ == '__main__':
|
|
main()
|