jung/tools/migrate_manifests_canonical.py
Dmitry Kokorin 2847f1bcea manifests 01-12: canonical v1 full migration (tools/migrate_manifests_canonical.py)
- per-item blocks for EVERY INDEX item (01: 49->55, 02: 28->57, 03: 39->68,
  04: 12->39, 05: 23->29); xref blocks: sec02#21->sec01#44, sec12#03/06/13/18/21
- heading normalization (no status/✔/prose tails), INDEX order everywhere (sec12
  27/28/31 repositioned)
- tail blocks in canonical order: Not downloadable -> Cross-references -> Notes
  (sec04 ND block built from MISSING.md + #16 row; sec03 + #13/#28 rows)
- 2 orphan file rows added: sec06 #26 scholem en-b.epub, sec07 #02 cw3-ru fb2
- Totals recounted from disk (all 12 sections verified: files = mentions)
- 0 file rows lost (git old-vs-new diff check)
- xref.py stale-line check: clean

Prep step for the repo-root READING-LIST.md generator (per user).
2026-09-28 23:50:09 +03:00

282 lines
13 KiB
Python

#!/usr/bin/env python3
"""tools/migrate_manifests_canonical.py — normalize ALL manifests to CANON v1 (2026-09-28).
CANON (per AGENTS.md + user cross-ref rule 2026-09-26):
H1: # Downloads — Section NN <Name> — FINAL <date>
## Totals — N files, X MB — RU a / EN b (source breakdown)
## Files (per item) — ONE block per INDEX item, in INDEX order:
### NN — <title> — N files (local file rows)
### NN — <title> — files in secSS (xref-only items)
4-col table: | file | size | source | verify |
xref row: | — (cross-ref secSS #MM: …) | | secSS | |
## Not downloadable (verified <date>) — | item | reason |
## Cross-references — summary index (NOT a substitute for per-item blocks)
## Notes
Guarantees: markdown-only, NO file moves/deletes. Prints a change report + REVIEW list.
Usage: python3 tools/migrate_manifests_canonical.py [--dry] [--sec 12]
"""
import os, re, sys, datetime
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import xref as X
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
DRY = "--dry" in sys.argv
SEC_ONLY = None
if "--sec" in sys.argv:
SEC_ONLY = sys.argv[sys.argv.index("--sec") + 1].zfill(2)
SECS = ["01-fundamentals", "02-dreams", "03-myths-fairy-tales", "04-pictures",
"05-ethnology", "06-religion", "07-complexes", "08-developmental",
"09-comparison-of-psychodynamic-concepts", "10-psychopathology-psychiatry",
"11-individuation", "12-practical-case"]
DL_DIR = {"09-comparison-of-psychodynamic-concepts": "09-comparison",
"10-psychopathology-psychiatry": "10-psychopathology"}
def human_mb(n):
return f"{n/1024/1024:.1f} MB" if n >= 1024*1024 else f"{n/1024:.0f} KB"
def parse_index(sec):
"""items: num -> (title, author, status, card) in INDEX order"""
p = os.path.join(BASE, "sections", sec, "INDEX.md")
items = {}
for line in open(p, encoding="utf-8"):
m = re.match(r'^\|\s*(\d{2})\s*\|\s*([^|]*)\|\s*([^|]*)\|\s*([^|]*)\|\s*([^|]+?)\s*\|\s*([^|]+)\|', line)
if m:
num, grp, title, author, status, card = m.groups()
em = re.search(r'[✅🔶❌⬜]', status)
items[num] = dict(title=title.strip(), author=author.strip(),
status=(em.group(0) if em else status.strip()),
status_raw=status.strip(), card=card.strip())
return items
def parse_manifest(sec):
"""returns: header lines, blocks {num: [lines]}, tail sections {name: lines}, xref table rows"""
p = os.path.join(BASE, "downloads", DL_DIR.get(sec, sec), "MANIFEST.md")
lines = open(p, encoding="utf-8").read().splitlines()
header, region, tails = [], [], {}
mode, cur = "header", None
for ln in lines:
if re.match(r'^### \d{2} —', ln):
mode = "block"
cur = ln.split(" — ", 1)[0].replace("### ", "").strip()
region.append(ln)
continue
if ln.startswith("## "):
if mode == "region_start":
pass
mode = "tail"
m = re.match(r'^## (Not downloadable.*|Cross-references.*|Cross-section.*|Notes.*)$', ln)
name = (m.group(1).split(" (")[0] if m else "other")
tails.setdefault(name, []).append(ln)
cur = name
continue
if ln.startswith("## Files (per item)"):
mode = "region_start"
header.append(ln)
continue
if mode == "header":
header.append(ln)
elif mode in ("region_start", "block"):
region.append(ln)
else:
tails[cur].append(ln)
# split region into per-item blocks
blocks, cur = {}, None
for ln in region:
if re.match(r'^### \d{2} —', ln):
cur = ln.split(" — ", 1)[0].replace("### ", "").strip()
blocks[cur] = []
elif cur is not None:
blocks[cur].append(ln)
return header, blocks, tails
def file_rows(block_lines):
"""local file rows: first cell is a filename (not a cross-ref '—' row)"""
out = []
for ln in block_lines:
if not ln.startswith("|"):
continue
cells = [c.strip() for c in ln.strip("|").split("|")]
first = cells[0]
if first.startswith("—") or first in ("file", "") or first.startswith(":"):
continue
if re.search(r'\.\w{2,4}$', first):
out.append(ln)
return out
def xref_rows(block_lines):
return [ln for ln in block_lines if ln.startswith("|") and ln.strip("|").split("|")[0].strip().startswith("—")]
def norm_heading(num, title, block_lines):
fx = xref_rows(block_lines)
fl = file_rows(block_lines)
if fx and not fl:
secs = sorted(set(re.findall(r'sec(\d\d)', " ".join(fx))))
tgt = "+".join(secs) if secs else "??"
return f"### {num} — {title} — files in sec{tgt}"
n = len(fl)
return f"### {num} — {title} — {n} files"
def clean_title(rest):
"""strip trailing annotations from a legacy heading"""
rest = re.sub(r'\s*—\s*(\d+\s*files?.*)$', '', rest)
rest = re.sub(r'\s*—\s*(files in sec.*)$', '', rest)
rest = re.sub(r'\s*—\s*✔.*$', '', rest)
rest = re.sub(r'\s*—\s*EN.*$', '', rest)
rest = re.sub(r'\s*—\s*RU.*$', '', rest)
return rest.strip() or rest
def main():
report = []
# cross-section card matcher (xref.py)
try:
cards = X.card_entries()
except Exception as ex:
print("card_entries failed:", ex); cards = []
for sec in SECS:
if SEC_ONLY and sec[:2] != SEC_ONLY:
continue
items = parse_index(sec)
header, blocks, tails = parse_manifest(sec)
ddir = os.path.join(BASE, "downloads", DL_DIR.get(sec, sec))
dfiles = sorted(f for f in os.listdir(ddir) if f != "MANIFEST.md" and (re.match(r'^\d\d-', f) or f.startswith("bonus")))
# xref map from trailing Cross-references table: item -> "secSS #MM"
xmap = {}
xref_tail = []
for name in ("Cross-references", "Cross-section", "other"):
if name in tails:
xref_tail = tails[name]
for ln in tails[name]:
if not ln.startswith("|") or ln.startswith("| item") or ln.startswith("|---"):
continue
cells = [c.strip() for c in ln.strip("|").split("|")]
if cells and re.match(r'^\d{2}', cells[0]):
num_x = cells[0][:2]
m_t = re.search(r'sec(\d{2})|(?:(\d{2}) #)', ln)
if m_t:
xmap[num_x] = ("sec" + (m_t.group(1) or m_t.group(2)), ln)
break # first found section-name wins
report.append(f"== {sec}: {len(items)} items, {len(blocks)} blocks before")
# ---- build new per-item region
new_region, seen = [], set()
all_old_lines = "\n".join(l for bl in blocks.values() for l in bl)
review = []
for num in sorted(items):
seen.add(num)
title = items[num]["title"]
if num in blocks:
bl = blocks[num]
t = clean_title(title)
new_region.append(norm_heading(num, t, bl))
new_region.extend(bl)
else:
local = [f for f in dfiles if f.startswith(num + "-")]
if local:
# find rows for these files wherever they are
rows = []
for f in local:
m = [ln for ln in all_old_lines.splitlines() if f in ln and ln.startswith("|")]
rows.extend(m[:1] if m else [f"| {f} | {human_mb(os.path.getsize(os.path.join(ddir,f)))} | UNKNOWN | REVIEW |"])
new_region.append(f"### {num} — {clean_title(title)} — {len(local)} files")
new_region.extend(rows)
review.append(f" {num} {title[:40]}: block ADDED ({len(local)} local files, rows moved/rebuilt)")
elif num in xmap or re.match(r'^xref\s+sec\d+', items[num]["status_raw"]) or (not local and cards):
sr = re.search(r'xref\s+sec(\d{2})(?:\s+#(\d{2}))?', items[num]["status_raw"])
if sr:
tgt_s, mm = "sec" + sr.group(1), sr.group(2)
elif num in xmap:
tgt_s, mm = xmap[num][0], None
else:
# cross-section card match (earlier sections only)
best, best_sc = None, 0.0
e_tok = X.norm(clean_title(title))
for c in cards:
if c['sec'] == sec[:2]:
continue
sc = X.score_pair(e_tok, None, items[num]['author'], c['tok'], c['cw'], c['author'])
if sc > best_sc:
best, best_sc = c, sc
ok = False
if best and int(best['sec']) < int(sec[:2]):
if best_sc >= 0.55:
ok = True
elif best_sc >= 0.5:
# lenient: title subset + shared surname token (e.g. Cambray sec12#06 -> sec01#09)
def surnames(a):
return {w.lower().strip(".(), ") for w in (a or "").replace("&", " ").split() if w[:1].isupper()}
subset = e_tok and best['tok'] and (e_tok <= best['tok'] or best['tok'] <= e_tok)
ok = subset and bool(surnames(items[num]['author']) & surnames(best['author']))
if ok:
tgt_s, mm = "sec" + best['sec'], best['n']
else:
tgt_s, mm = None, None
if tgt_s is None:
pass # fall through to 0 files (below)
else:
new_region.append(f"### {num} — {clean_title(title)} — files in {tgt_s}")
new_region.append(f"| — (cross-ref {tgt_s}{' #' + mm if mm else ''}: see Cross-references) | | {tgt_s[3:]} | |")
review.append(f" {num} {title[:40]}: xref block ADDED ({tgt_s}{' #'+mm if mm else ''})")
continue
# fall through: no xref resolved
new_region.append(f"### {num} — {clean_title(title)} — 0 files")
if items[num]["status"] in "✅🔶":
review.append(f" {num} {title[:40]}: 0 files but status {items[num]['status']} — check Not-downloadable block!")
# legacy blocks for items not in INDEX
for num in blocks:
if num not in seen:
review.append(f" BLOCK {num} has no INDEX item — DROPPED from region: {blocks[num][0] if blocks[num] else ''}")
# ---- tails in canonical order
nd = tails.get("Not downloadable", [])
if not nd:
# build from MISSING.md ❌ items (compact)
mp = os.path.join(BASE, "sections", sec, "MISSING.md")
nd = ["## Not downloadable (verified 2026-09-28)", "", "| item | reason |", "|------|--------|"]
if os.path.exists(mp):
ms = open(mp, encoding="utf-8").read()
for m2 in re.finditer(r'^### (\d{2}) — (.+?)\n((?:^.*\n){0,3})', ms, re.M):
i2, ttl, body = m2.group(1), m2.group(2).strip(), m2.group(3)
reason = re.sub(r'\s+', ' ', body).strip()[:140]
nd.append(f"| {i2} {ttl[:50]} | {reason} |")
# notes tail
notes = tails.get("Notes", [])
# ---- totals (recount)
n_files = len(dfiles)
tot_bytes = sum(os.path.getsize(os.path.join(ddir, f)) for f in dfiles)
n_ru = sum(1 for f in dfiles if re.search(r'-ru[.\-]', f))
n_en = sum(1 for f in dfiles if re.search(r'-en[.\-]', f))
allrows = [l for l in new_region if l.startswith("|") and not l.strip("|").split("|")[0].strip().startswith("—")]
def cnt(pat): return sum(1 for l in allrows if pat in l)
src = f"lg {cnt('lg f/')} + ia {cnt('ia ')} + flib {cnt('flib')}"
# keep old header but ensure H1 format
h1 = [l for l in header if l.startswith("# ")]
h1 = [h1[0]] if h1 else [f"# Downloads — Section {sec[:2]} {sec[3:]} — FINAL 2026-09-28"]
out = list(h1) + ["", "## Totals", "",
f"{n_files} files, {human_mb(tot_bytes)} — RU {n_ru} / EN {n_en} / other {n_files-n_ru-n_en} ({src})",
"", "## Files (per item)", ""]
out += new_region
out += [""] + (nd if nd else [])
if xref_tail:
out += ["", "## Cross-references"] + [l for l in xref_tail if not l.startswith("## ")]
if notes:
out += [""] + notes
# trim trailing blank lines
while out and not out[-1].strip():
out.pop()
out.append("")
text = "\n".join(out)
if not DRY:
open(os.path.join(BASE, "downloads", DL_DIR.get(sec, sec), "MANIFEST.md"), "w", encoding="utf-8").write(text)
report.append(f" -> {len(items)} blocks after; {n_files} files, {human_mb(tot_bytes)}")
report.extend(review)
print("\n".join(report))
if __name__ == "__main__":
main()