v3 migration: sec04 #22 — ES section + Original field (Spanish original)
This commit is contained in:
parent
54b01b8002
commit
4f5372ab56
3 changed files with 353 additions and 1 deletions
|
|
@ -3,9 +3,10 @@
|
|||
**Author(s):** Paul Brutsche
|
||||
**Shelf mark:** —
|
||||
**Section:** 04 Pictures
|
||||
**Original:** ES — Editorial Fata Morgana, 2022 (оригинал на испанском, EN-версии нет)
|
||||
**Status:** ❌ no RU edition (verified; Spanish original anyway)
|
||||
|
||||
## Editions — EN
|
||||
## Editions — ES
|
||||
|
||||
| Title / edition | Publisher | Year | Pages | ISBN | Notes |
|
||||
|-----------------|-----------|------|-------|------|-------|
|
||||
|
|
|
|||
123
tools/check-md.py
Normal file
123
tools/check-md.py
Normal file
|
|
@ -0,0 +1,123 @@
|
|||
#!/usr/bin/env python3
|
||||
"""Lint book cards against CANON v3 (see AGENTS.md).
|
||||
Usage: python3 tools/check-md.py [section ...] (default: all)
|
||||
Exit code: 0 = clean, 1 = issues found."""
|
||||
import re, sys, os, glob
|
||||
|
||||
ALLOWED_SECTIONS = re.compile(r'^## (Editions — [A-Z]{2}|Downloads|Catalog records|Notes)$')
|
||||
BANNED_SECTIONS = ('## Russian editions', '## Libgen', '## RSL records', '## Verdict',
|
||||
'## Identity check', '## Search log')
|
||||
BANNED_TEXTS = ('// not searched yet', '(not searched yet)', '// not searched',
|
||||
'**Publisher (EN):**', '**EN ISBN:**', '**Reading list item(s):**')
|
||||
VALID_MARKERS = ('✅', '🔶', '❌', '⬜', '🔎')
|
||||
ORDER = ['ED', 'DL', 'CR', 'NT']
|
||||
|
||||
def sec_of(path):
|
||||
return os.path.basename(os.path.dirname(path))
|
||||
|
||||
def lint_card(path, manifest_files):
|
||||
c = open(path, encoding='utf-8').read()
|
||||
issues = []
|
||||
num = os.path.basename(path)[:2]
|
||||
if not c.startswith('# '):
|
||||
issues.append('no H1')
|
||||
for f in ('**Author(s):**', '**Shelf mark:**', '**Section:**', '**Status:**'):
|
||||
if f not in c.split('\n\n')[0] + '\n'.join(c.split('\n')[:8]):
|
||||
issues.append(f'header missing {f}')
|
||||
sm = re.search(r'\*\*Status:\*\* (.*)', c)
|
||||
if sm and not sm.group(1).strip().startswith(VALID_MARKERS):
|
||||
issues.append(f'status marker invalid: {sm.group(1)[:30]!r}')
|
||||
for t in BANNED_TEXTS:
|
||||
if t in c:
|
||||
issues.append(f'banned text: {t}')
|
||||
# sections
|
||||
seen = []
|
||||
for sm2 in re.finditer(r'^## (.+)$', c, re.M):
|
||||
name = sm2.group(1).strip()
|
||||
line = f'## {name}'
|
||||
if not ALLOWED_SECTIONS.match(line):
|
||||
issues.append(f'non-canonical section: {line}')
|
||||
key = ('ED' if name.startswith('Editions —') else
|
||||
'DL' if name == 'Downloads' else
|
||||
'CR' if name == 'Catalog records' else 'NT')
|
||||
seen.append(key)
|
||||
# Notes exactly once (for migrated cards; tolerate 0 for ⬜ cards)
|
||||
if seen.count('NT') > 1:
|
||||
issues.append(f'Notes x{seen.count("NT")}')
|
||||
# order: all ED before DL before CR before NT
|
||||
pos = {k: [i for i, s in enumerate(seen) if s == k] for k in ORDER}
|
||||
last = -1
|
||||
for k in ORDER:
|
||||
if pos[k]:
|
||||
if max(pos[k]) < last:
|
||||
issues.append(f'section order broken at {k}')
|
||||
break
|
||||
last = max(pos[k])
|
||||
# empty sections
|
||||
if re.search(r'^## [A-Za-z—& ]+\n\n\n## ', c, re.M):
|
||||
issues.append('empty section body')
|
||||
# Downloads table cross-check
|
||||
dm = re.search(r'^## Downloads\n(.*?)(?=^## |\Z)', c, re.M | re.S)
|
||||
if dm:
|
||||
rows = re.findall(r'^\|\s*(EN|RU\??|\?)\s*\|\s*([0-9]{2}-[a-zA-Z0-9\.\-]+)\s*\|', dm.group(1), re.M)
|
||||
disk = set(os.path.basename(f) for f in glob.glob(f"downloads/{sec_of(path)}/*"))
|
||||
manifest = set(f for f in manifest_files.get(num, []))
|
||||
for lang, fname in rows:
|
||||
if fname not in disk:
|
||||
issues.append(f'Downloads file not on disk: {fname}')
|
||||
if manifest and fname not in manifest:
|
||||
issues.append(f'Downloads file not in MANIFEST: {fname}')
|
||||
if manifest and set(f for _, f in rows) != manifest:
|
||||
extra = manifest - set(f for _, f in rows)
|
||||
missing = set(f for _, f in rows) - manifest
|
||||
if extra: issues.append(f'MANIFEST files missing from card: {sorted(extra)}')
|
||||
if missing: issues.append(f'card files not in MANIFEST: {sorted(missing)}')
|
||||
return issues
|
||||
|
||||
def lint_index_status(section_dir):
|
||||
"""INDEX.md status column must match card Status."""
|
||||
idx = os.path.join(section_dir, 'INDEX.md')
|
||||
if not os.path.exists(idx):
|
||||
return ['no INDEX.md']
|
||||
issues = []
|
||||
idx_text = open(idx, encoding='utf-8').read()
|
||||
for p in sorted(glob.glob(f'{section_dir}/[0-9][0-9]-*.md')):
|
||||
num = os.path.basename(p)[:2]
|
||||
card_status = re.search(r'\*\*Status:\*\* (.)', open(p).read())
|
||||
card_m = card_status.group(1) if card_status else None
|
||||
row = re.search(rf'^\|\s*{num}\b.*?((?:✅|🔶|❌|⬜|🔎)[^\n|]*)', idx_text, re.M)
|
||||
if row:
|
||||
idx_m = row.group(1).strip()[:1]
|
||||
if card_m and idx_m != card_m:
|
||||
issues.append(f'INDEX {num} status {idx_m} ≠ card {card_m}')
|
||||
return issues
|
||||
|
||||
def main():
|
||||
secs = sys.argv[1:] or sorted(os.path.basename(p) for p in glob.glob('sections/0*'))
|
||||
total = 0
|
||||
for sec in secs:
|
||||
secdir = f'sections/{sec}'
|
||||
if not os.path.isdir(secdir):
|
||||
continue
|
||||
# manifest file sets
|
||||
mfiles = {}
|
||||
mp = f'downloads/{sec}/MANIFEST.md'
|
||||
if os.path.exists(mp):
|
||||
mt = open(mp).read()
|
||||
for f in re.findall(r'([0-9]{2}-[a-zA-Z0-9\.\-]+\.(?:pdf|fb2|epub|doc|docx|chm|djvu|txt|zip|azw3))', mt):
|
||||
mfiles.setdefault(f[:2], set()).add(f)
|
||||
ncards = 0
|
||||
for p in sorted(glob.glob(f'{secdir}/[0-9][0-9]-*.md')):
|
||||
ncards += 1
|
||||
for i in lint_card(p, mfiles):
|
||||
total += 1
|
||||
print(f'{p}: {i}')
|
||||
for i in lint_index_status(secdir):
|
||||
total += 1
|
||||
print(f'{secdir}/INDEX: {i}')
|
||||
print(f'{sec}: {ncards} cards linted')
|
||||
print(f'\nTOTAL issues: {total}')
|
||||
sys.exit(1 if total else 0)
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
228
tools/migrate_cards_v3.py
Normal file
228
tools/migrate_cards_v3.py
Normal file
|
|
@ -0,0 +1,228 @@
|
|||
#!/usr/bin/env python3
|
||||
"""Migrate book cards to CANON v3 (see AGENTS.md). Lossless where possible.
|
||||
Usage: python3 tools/migrate_cards_v3.py --section 01-fundamentals [--dry-run NN ...]"""
|
||||
import re, sys, os, glob, argparse
|
||||
|
||||
BOILER_LIBGEN = re.compile(r'^[-*]?\s*\((none|not searched|not searched yet|not checked|not checked yet|trade-only|empty|TBC|нет|—)\)?\s*$', re.I)
|
||||
SEC_DATES = { # section completion / verification dates (for gate lines)
|
||||
'01-fundamentals': '2026-07-17', '02-dreams': '2026-07-18',
|
||||
'03-myths-fairy-tales': '2026-09-20', '04-pictures': '2026-07-15',
|
||||
}
|
||||
SEC_NAMES = {
|
||||
'01-fundamentals': '01 Fundamentals', '02-dreams': '02 Dreams',
|
||||
'03-myths-fairy-tales': '03 Myths & Fairy Tales', '04-pictures': '04 Pictures',
|
||||
}
|
||||
|
||||
def parse_manifest_files(manifest_path):
|
||||
"""Return {item_num: [(file, size_str, source), ...]} for a section MANIFEST."""
|
||||
items = {}
|
||||
if not os.path.exists(manifest_path):
|
||||
return items
|
||||
c = open(manifest_path, encoding='utf-8').read()
|
||||
fname_re = r'([0-9]{2}-[a-zA-Z0-9\-]+(?:\.[a-zA-Z0-9]+)*\.(?:pdf|fb2|epub|doc|docx|chm|djvu|txt|zip))'
|
||||
# sec01/02: table rows | file | size | source | note |
|
||||
for m in re.finditer(r'^\|\s*' + fname_re + r'\s*\|\s*([^|]+)\|\s*([^|]+)\|', c, re.M):
|
||||
f, size, src = m.group(1), m.group(2).strip(), m.group(3).strip()
|
||||
if re.search(r'-(ru|en)[.\-]', f):
|
||||
items.setdefault(f[:2], []).append((f, size, src))
|
||||
# sec03: per-item subsections with bullets
|
||||
for m in re.finditer(r'^### (\d{2}) —.*?\n(.*?)(?=\n### |\n## |\Z)', c, re.M | re.S):
|
||||
num, body = m.group(1), m.group(2)
|
||||
for b in re.finditer(r'^- `([^`]+)` \(([^)]+)\)', body, re.M):
|
||||
fname, size = b.group(1), b.group(2)
|
||||
items.setdefault(fname[:2], []).append((fname, size, 'MANIFEST'))
|
||||
# sec04: per-item tables | file | bytes | ed/f | verify |
|
||||
for m in re.finditer(r'^## (\d{2}) —.*?\n(.*?)(?=\n## |\Z)', c, re.M | re.S):
|
||||
num, body = m.group(1), m.group(2)
|
||||
for row in re.finditer(r'^\|\s*' + fname_re + r'\s*\|\s*(\d[\d ]*?)\s*\|\s*([^|]+)\|', body, re.M):
|
||||
fname, sz, ids = row.group(1), f"{int(row.group(2).replace(' ', ''))//1048576} MB", row.group(3).strip()
|
||||
fids = re.search(r'/(\d+)$', ids)
|
||||
src = f'lg f/{fids.group(1)}' if fids else ids
|
||||
items.setdefault(fname[:2], []).append((fname, sz, src))
|
||||
# dedupe by filename; prefer human-readable source (lg f/N, flib b/N, ia, MANIFEST)
|
||||
good = re.compile(r'^(lg f|flib b|ia |MANIFEST|CarlJung)')
|
||||
for num, lst in items.items():
|
||||
seen = {}
|
||||
for f, s, r in lst:
|
||||
if f not in seen or (good.match(r) and not good.match(seen[f][2])):
|
||||
seen[f] = (f, s, r)
|
||||
items[num] = list(seen.values())
|
||||
return items
|
||||
|
||||
def h1_title(c):
|
||||
m = re.match(r'^# (.*)$', c, re.M)
|
||||
return m.group(1).strip() if m else ''
|
||||
|
||||
def is_dash(v):
|
||||
if v is None:
|
||||
return True
|
||||
v = v.strip()
|
||||
return v in ('—', '-', '— (Phase 0: to resolve)', '') or v.startswith('— (Phase 0')
|
||||
|
||||
def human_size(s):
|
||||
s = s.strip()
|
||||
m = re.match(r'^([\d\s]+)$', s)
|
||||
if m:
|
||||
b = int(m.group(1).replace(' ', ''))
|
||||
return f'{b//1048576} MB' if b >= 1048576 else f'{b//1024} KB'
|
||||
return s
|
||||
|
||||
def migrate_card(path, files_by_item, sec, dry=False):
|
||||
c = open(path, encoding='utf-8').read()
|
||||
num = os.path.basename(path)[:2]
|
||||
# split header (before first ##) and sections
|
||||
m = re.search(r'\n## ', c)
|
||||
header, rest = c[:m.start()], c[m.start()+1:]
|
||||
def fld(name):
|
||||
mm = re.search(rf'\*\*{re.escape(name)}:\*\* (.*)', header)
|
||||
return mm.group(1).strip() if mm else None
|
||||
title = h1_title(c)
|
||||
author = fld('Author(s)') or '—'
|
||||
shelf = fld('Shelf mark')
|
||||
section = fld('Section')
|
||||
status = fld('Status') or '⬜'
|
||||
pub_en = fld('Publisher (EN)')
|
||||
isbn_en = fld('EN ISBN')
|
||||
rli = fld('Reading list item(s)')
|
||||
|
||||
notes_extra = []
|
||||
if rli:
|
||||
notes_extra.append(f'- Reading list item(s): {rli}')
|
||||
|
||||
# --- EN editions table
|
||||
en_block = ''
|
||||
if not is_dash(pub_en) or (isbn_en and not is_dash(isbn_en)):
|
||||
pub = pub_en or '—'
|
||||
year = '—'
|
||||
mm = re.match(r'^(.*),\s*(\d{4}[a-z]?)\s*(\(.*\))?$', pub)
|
||||
if mm:
|
||||
pub, year = mm.group(1).strip(), mm.group(2)
|
||||
isbn = '—' if is_dash(isbn_en) else isbn_en
|
||||
en_block = (
|
||||
'\n## Editions — EN\n\n'
|
||||
'| Title / edition | Publisher | Year | Pages | ISBN | Notes |\n'
|
||||
'|-----------------|-----------|------|-------|------|-------|\n'
|
||||
f'| {title} | {pub} | {year} | — | {isbn} | |\n')
|
||||
|
||||
# --- gather sections
|
||||
secs = {}
|
||||
for sm in re.finditer(r'^## (.+?)\n(.*?)(?=^## |\Z)', rest, re.M | re.S):
|
||||
secs[sm.group(1).strip()] = sm.group(2).strip('\n')
|
||||
|
||||
ru_body = secs.get('Russian editions')
|
||||
libgen_body = secs.get('Libgen')
|
||||
rsl_body = secs.get('RSL records')
|
||||
notes_body = secs.get('Notes', '')
|
||||
|
||||
# --- Editions — RU
|
||||
ru_block = ''
|
||||
if ru_body is not None:
|
||||
tb = re.search(r'\| RU title \|.*?(?=\n\n|\Z)', ru_body, re.S)
|
||||
if tb:
|
||||
tbl = tb.group(0).rstrip()
|
||||
data_rows = [l for l in tbl.split('\n') if l.startswith('|') and 'RU title' not in l
|
||||
and not re.match(r'^\|[-\s|]+\|$', l) and '(not searched' not in l
|
||||
and not re.match(r'^\|\s*\(?\s*(none|n/a|—|-)\s*\)?\s*\|', l)]
|
||||
if data_rows:
|
||||
ru_block = '\n## Editions — RU\n\n' + tbl + '\n'
|
||||
else:
|
||||
vdate = re.search(r'verified (\d{4}-\d{2}-\d{2})', status)
|
||||
vdate = vdate.group(1) if vdate else SEC_DATES.get(sec, '')
|
||||
ru_block = f'\n## Editions — RU\n\n— нет (verified {vdate}, gate: MISSING.md §{num})\n'
|
||||
else:
|
||||
vdate = re.search(r'verified (\d{4}-\d{2}-\d{2})', status)
|
||||
vdate = vdate.group(1) if vdate else SEC_DATES.get(sec, '')
|
||||
ru_block = f'\n## Editions — RU\n\n— нет (verified {vdate}, gate: MISSING.md §{num})\n'
|
||||
# --- Downloads
|
||||
dl_block = ''
|
||||
files = files_by_item.get(num, [])
|
||||
if files:
|
||||
rows = []
|
||||
for fname, size, src in files:
|
||||
lang = 'RU' if re.search(r'-ru[.\-]', fname) else ('EN' if re.search(r'-en[.\-]', fname) else '?')
|
||||
rows.append(f'| {lang} | {fname} | {human_size(size)} | {src} |')
|
||||
dl_block = ('## Downloads\n\n| lang | file | size | source |\n'
|
||||
'|------|------|------|--------|\n' + '\n'.join(rows) + '\n')
|
||||
# libgen body residue → Notes (ids not covered by the table, or no table at all)
|
||||
if libgen_body:
|
||||
src_ids = set(re.findall(r'(?:flib b|lg f)/\d+', ' '.join(r for _, _, r in files)))
|
||||
keep = []
|
||||
for l in libgen_body.split('\n'):
|
||||
s = l.strip()
|
||||
if not s or BOILER_LIBGEN.match(s):
|
||||
continue
|
||||
if re.match(r'^(?:- )?(RU|EN) (file|files):', s):
|
||||
ids = set(re.findall(r'(?:flib b|lg f)/\d+', s))
|
||||
if not ids or ids <= src_ids:
|
||||
continue # no info beyond the table
|
||||
keep.append(l)
|
||||
if keep:
|
||||
notes_extra.append('- (was Libgen): ' + ' '.join(k.lstrip('- ').strip() for k in keep))
|
||||
|
||||
# --- Catalog records
|
||||
cat_block = ''
|
||||
if rsl_body is not None:
|
||||
body = rsl_body.strip()
|
||||
inner = re.sub(r'^```json\n?|```$', '', body).strip()
|
||||
if not inner or inner == '// not searched yet' or inner == '// not searched':
|
||||
sweep = f'data/sweeps/{sec}/{num}-{os.path.basename(path)[:-3]}.md'
|
||||
if os.path.exists(sweep):
|
||||
inner = f'РГБ: (свип: {sweep})'
|
||||
else:
|
||||
vdate = re.search(r'verified (\d{4}-\d{2}-\d{2})', status)
|
||||
inner = f'no records (verified {vdate.group(1) if vdate else SEC_DATES.get(sec, "")})'
|
||||
else:
|
||||
inner = re.sub(r'^//\s*', 'РГБ: ', inner, count=1)
|
||||
if not inner.startswith(('РГБ', 'no records')):
|
||||
inner = 'РГБ: ' + inner
|
||||
cat_block = f'## Catalog records\n\n```\n{inner}\n```\n'
|
||||
|
||||
# --- Notes (merge extra sections)
|
||||
extra_secs = []
|
||||
for name in list(secs):
|
||||
if name in ('Russian editions', 'Libgen', 'RSL records', 'Notes'):
|
||||
continue
|
||||
extra_secs.append(f'{name}: {secs[name].strip()}')
|
||||
notes_lines = [l for l in notes_body.split('\n')
|
||||
if l.strip() != '- (empty)' and not re.match(r'^-?.*not searched yet\.?\s*$', l.strip())]
|
||||
notes = '\n'.join(notes_lines + notes_extra + [f'- (was {e})' for e in extra_secs]).strip()
|
||||
|
||||
# --- assemble
|
||||
out = f'# {title}\n\n'
|
||||
out += f'**Author(s):** {author}\n'
|
||||
out += f'**Shelf mark:** {shelf if shelf is not None else "—"}\n'
|
||||
out += f'**Section:** {section if section is not None else SEC_NAMES.get(sec, sec)}\n'
|
||||
out += f'**Status:** {status}\n'
|
||||
out += en_block
|
||||
out += ru_block
|
||||
if dl_block:
|
||||
out += '\n' + dl_block
|
||||
if cat_block:
|
||||
out += '\n' + cat_block
|
||||
if notes:
|
||||
out += f'\n## Notes\n\n{notes}\n'
|
||||
if dry:
|
||||
print('=' * 70)
|
||||
print(f'DRY-RUN {path}')
|
||||
print(out)
|
||||
return
|
||||
open(path, 'w', encoding='utf-8').write(out)
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument('--section', required=True)
|
||||
ap.add_argument('--dry-run', nargs='*', default=None)
|
||||
a = ap.parse_args()
|
||||
secdir = f'sections/{a.section}'
|
||||
man = f'downloads/{a.section}/MANIFEST.md'
|
||||
files = parse_manifest_files(man)
|
||||
cards = sorted(glob.glob(f'{secdir}/[0-9][0-9]-*.md'))
|
||||
for p in cards:
|
||||
if a.dry_run is not None and os.path.basename(p)[:2] not in a.dry_run:
|
||||
continue
|
||||
migrate_card(p, files, a.section, dry=(a.dry_run is not None))
|
||||
if not a.dry_run:
|
||||
print(f'migrated {len(cards)} cards in {secdir}')
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Loading…
Add table
Add a link
Reference in a new issue