- per-item blocks for EVERY INDEX item (01: 49->55, 02: 28->57, 03: 39->68, 04: 12->39, 05: 23->29); xref blocks: sec02#21->sec01#44, sec12#03/06/13/18/21 - heading normalization (no status/✔/prose tails), INDEX order everywhere (sec12 27/28/31 repositioned) - tail blocks in canonical order: Not downloadable -> Cross-references -> Notes (sec04 ND block built from MISSING.md + #16 row; sec03 + #13/#28 rows) - 2 orphan file rows added: sec06 #26 scholem en-b.epub, sec07 #02 cw3-ru fb2 - Totals recounted from disk (all 12 sections verified: files = mentions) - 0 file rows lost (git old-vs-new diff check) - xref.py stale-line check: clean Prep step for the repo-root READING-LIST.md generator (per user).
282 lines
13 KiB
Python
282 lines
13 KiB
Python
#!/usr/bin/env python3
|
|
"""tools/migrate_manifests_canonical.py — normalize ALL manifests to CANON v1 (2026-09-28).
|
|
|
|
CANON (per AGENTS.md + user cross-ref rule 2026-09-26):
|
|
H1: # Downloads — Section NN <Name> — FINAL <date>
|
|
## Totals — N files, X MB — RU a / EN b (source breakdown)
|
|
## Files (per item) — ONE block per INDEX item, in INDEX order:
|
|
### NN — <title> — N files (local file rows)
|
|
### NN — <title> — files in secSS (xref-only items)
|
|
4-col table: | file | size | source | verify |
|
|
xref row: | — (cross-ref secSS #MM: …) | | secSS | |
|
|
## Not downloadable (verified <date>) — | item | reason |
|
|
## Cross-references — summary index (NOT a substitute for per-item blocks)
|
|
## Notes
|
|
|
|
Guarantees: markdown-only, NO file moves/deletes. Prints a change report + REVIEW list.
|
|
Usage: python3 tools/migrate_manifests_canonical.py [--dry] [--sec 12]
|
|
"""
|
|
import os, re, sys, datetime
|
|
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
|
import xref as X
|
|
|
|
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
DRY = "--dry" in sys.argv
|
|
SEC_ONLY = None
|
|
if "--sec" in sys.argv:
|
|
SEC_ONLY = sys.argv[sys.argv.index("--sec") + 1].zfill(2)
|
|
|
|
SECS = ["01-fundamentals", "02-dreams", "03-myths-fairy-tales", "04-pictures",
|
|
"05-ethnology", "06-religion", "07-complexes", "08-developmental",
|
|
"09-comparison-of-psychodynamic-concepts", "10-psychopathology-psychiatry",
|
|
"11-individuation", "12-practical-case"]
|
|
|
|
DL_DIR = {"09-comparison-of-psychodynamic-concepts": "09-comparison",
|
|
"10-psychopathology-psychiatry": "10-psychopathology"}
|
|
|
|
def human_mb(n):
|
|
return f"{n/1024/1024:.1f} MB" if n >= 1024*1024 else f"{n/1024:.0f} KB"
|
|
|
|
def parse_index(sec):
|
|
"""items: num -> (title, author, status, card) in INDEX order"""
|
|
p = os.path.join(BASE, "sections", sec, "INDEX.md")
|
|
items = {}
|
|
for line in open(p, encoding="utf-8"):
|
|
m = re.match(r'^\|\s*(\d{2})\s*\|\s*([^|]*)\|\s*([^|]*)\|\s*([^|]*)\|\s*([^|]+?)\s*\|\s*([^|]+)\|', line)
|
|
if m:
|
|
num, grp, title, author, status, card = m.groups()
|
|
em = re.search(r'[✅🔶❌⬜]', status)
|
|
items[num] = dict(title=title.strip(), author=author.strip(),
|
|
status=(em.group(0) if em else status.strip()),
|
|
status_raw=status.strip(), card=card.strip())
|
|
return items
|
|
|
|
def parse_manifest(sec):
|
|
"""returns: header lines, blocks {num: [lines]}, tail sections {name: lines}, xref table rows"""
|
|
p = os.path.join(BASE, "downloads", DL_DIR.get(sec, sec), "MANIFEST.md")
|
|
lines = open(p, encoding="utf-8").read().splitlines()
|
|
header, region, tails = [], [], {}
|
|
mode, cur = "header", None
|
|
for ln in lines:
|
|
if re.match(r'^### \d{2} —', ln):
|
|
mode = "block"
|
|
cur = ln.split(" — ", 1)[0].replace("### ", "").strip()
|
|
region.append(ln)
|
|
continue
|
|
if ln.startswith("## "):
|
|
if mode == "region_start":
|
|
pass
|
|
mode = "tail"
|
|
m = re.match(r'^## (Not downloadable.*|Cross-references.*|Cross-section.*|Notes.*)$', ln)
|
|
name = (m.group(1).split(" (")[0] if m else "other")
|
|
tails.setdefault(name, []).append(ln)
|
|
cur = name
|
|
continue
|
|
if ln.startswith("## Files (per item)"):
|
|
mode = "region_start"
|
|
header.append(ln)
|
|
continue
|
|
if mode == "header":
|
|
header.append(ln)
|
|
elif mode in ("region_start", "block"):
|
|
region.append(ln)
|
|
else:
|
|
tails[cur].append(ln)
|
|
# split region into per-item blocks
|
|
blocks, cur = {}, None
|
|
for ln in region:
|
|
if re.match(r'^### \d{2} —', ln):
|
|
cur = ln.split(" — ", 1)[0].replace("### ", "").strip()
|
|
blocks[cur] = []
|
|
elif cur is not None:
|
|
blocks[cur].append(ln)
|
|
return header, blocks, tails
|
|
|
|
def file_rows(block_lines):
|
|
"""local file rows: first cell is a filename (not a cross-ref '—' row)"""
|
|
out = []
|
|
for ln in block_lines:
|
|
if not ln.startswith("|"):
|
|
continue
|
|
cells = [c.strip() for c in ln.strip("|").split("|")]
|
|
first = cells[0]
|
|
if first.startswith("—") or first in ("file", "") or first.startswith(":"):
|
|
continue
|
|
if re.search(r'\.\w{2,4}$', first):
|
|
out.append(ln)
|
|
return out
|
|
|
|
def xref_rows(block_lines):
|
|
return [ln for ln in block_lines if ln.startswith("|") and ln.strip("|").split("|")[0].strip().startswith("—")]
|
|
|
|
def norm_heading(num, title, block_lines):
|
|
fx = xref_rows(block_lines)
|
|
fl = file_rows(block_lines)
|
|
if fx and not fl:
|
|
secs = sorted(set(re.findall(r'sec(\d\d)', " ".join(fx))))
|
|
tgt = "+".join(secs) if secs else "??"
|
|
return f"### {num} — {title} — files in sec{tgt}"
|
|
n = len(fl)
|
|
return f"### {num} — {title} — {n} files"
|
|
|
|
def clean_title(rest):
|
|
"""strip trailing annotations from a legacy heading"""
|
|
rest = re.sub(r'\s*—\s*(\d+\s*files?.*)$', '', rest)
|
|
rest = re.sub(r'\s*—\s*(files in sec.*)$', '', rest)
|
|
rest = re.sub(r'\s*—\s*✔.*$', '', rest)
|
|
rest = re.sub(r'\s*—\s*EN.*$', '', rest)
|
|
rest = re.sub(r'\s*—\s*RU.*$', '', rest)
|
|
return rest.strip() or rest
|
|
|
|
def main():
|
|
report = []
|
|
# cross-section card matcher (xref.py)
|
|
try:
|
|
cards = X.card_entries()
|
|
except Exception as ex:
|
|
print("card_entries failed:", ex); cards = []
|
|
for sec in SECS:
|
|
if SEC_ONLY and sec[:2] != SEC_ONLY:
|
|
continue
|
|
items = parse_index(sec)
|
|
header, blocks, tails = parse_manifest(sec)
|
|
ddir = os.path.join(BASE, "downloads", DL_DIR.get(sec, sec))
|
|
dfiles = sorted(f for f in os.listdir(ddir) if f != "MANIFEST.md" and (re.match(r'^\d\d-', f) or f.startswith("bonus")))
|
|
|
|
# xref map from trailing Cross-references table: item -> "secSS #MM"
|
|
xmap = {}
|
|
xref_tail = []
|
|
for name in ("Cross-references", "Cross-section", "other"):
|
|
if name in tails:
|
|
xref_tail = tails[name]
|
|
for ln in tails[name]:
|
|
if not ln.startswith("|") or ln.startswith("| item") or ln.startswith("|---"):
|
|
continue
|
|
cells = [c.strip() for c in ln.strip("|").split("|")]
|
|
if cells and re.match(r'^\d{2}', cells[0]):
|
|
num_x = cells[0][:2]
|
|
m_t = re.search(r'sec(\d{2})|(?:(\d{2}) #)', ln)
|
|
if m_t:
|
|
xmap[num_x] = ("sec" + (m_t.group(1) or m_t.group(2)), ln)
|
|
break # first found section-name wins
|
|
report.append(f"== {sec}: {len(items)} items, {len(blocks)} blocks before")
|
|
|
|
# ---- build new per-item region
|
|
new_region, seen = [], set()
|
|
all_old_lines = "\n".join(l for bl in blocks.values() for l in bl)
|
|
review = []
|
|
for num in sorted(items):
|
|
seen.add(num)
|
|
title = items[num]["title"]
|
|
if num in blocks:
|
|
bl = blocks[num]
|
|
t = clean_title(title)
|
|
new_region.append(norm_heading(num, t, bl))
|
|
new_region.extend(bl)
|
|
else:
|
|
local = [f for f in dfiles if f.startswith(num + "-")]
|
|
if local:
|
|
# find rows for these files wherever they are
|
|
rows = []
|
|
for f in local:
|
|
m = [ln for ln in all_old_lines.splitlines() if f in ln and ln.startswith("|")]
|
|
rows.extend(m[:1] if m else [f"| {f} | {human_mb(os.path.getsize(os.path.join(ddir,f)))} | UNKNOWN | REVIEW |"])
|
|
new_region.append(f"### {num} — {clean_title(title)} — {len(local)} files")
|
|
new_region.extend(rows)
|
|
review.append(f" {num} {title[:40]}: block ADDED ({len(local)} local files, rows moved/rebuilt)")
|
|
elif num in xmap or re.match(r'^xref\s+sec\d+', items[num]["status_raw"]) or (not local and cards):
|
|
sr = re.search(r'xref\s+sec(\d{2})(?:\s+#(\d{2}))?', items[num]["status_raw"])
|
|
if sr:
|
|
tgt_s, mm = "sec" + sr.group(1), sr.group(2)
|
|
elif num in xmap:
|
|
tgt_s, mm = xmap[num][0], None
|
|
else:
|
|
# cross-section card match (earlier sections only)
|
|
best, best_sc = None, 0.0
|
|
e_tok = X.norm(clean_title(title))
|
|
for c in cards:
|
|
if c['sec'] == sec[:2]:
|
|
continue
|
|
sc = X.score_pair(e_tok, None, items[num]['author'], c['tok'], c['cw'], c['author'])
|
|
if sc > best_sc:
|
|
best, best_sc = c, sc
|
|
ok = False
|
|
if best and int(best['sec']) < int(sec[:2]):
|
|
if best_sc >= 0.55:
|
|
ok = True
|
|
elif best_sc >= 0.5:
|
|
# lenient: title subset + shared surname token (e.g. Cambray sec12#06 -> sec01#09)
|
|
def surnames(a):
|
|
return {w.lower().strip(".(), ") for w in (a or "").replace("&", " ").split() if w[:1].isupper()}
|
|
subset = e_tok and best['tok'] and (e_tok <= best['tok'] or best['tok'] <= e_tok)
|
|
ok = subset and bool(surnames(items[num]['author']) & surnames(best['author']))
|
|
if ok:
|
|
tgt_s, mm = "sec" + best['sec'], best['n']
|
|
else:
|
|
tgt_s, mm = None, None
|
|
if tgt_s is None:
|
|
pass # fall through to 0 files (below)
|
|
else:
|
|
new_region.append(f"### {num} — {clean_title(title)} — files in {tgt_s}")
|
|
new_region.append(f"| — (cross-ref {tgt_s}{' #' + mm if mm else ''}: see Cross-references) | | {tgt_s[3:]} | |")
|
|
review.append(f" {num} {title[:40]}: xref block ADDED ({tgt_s}{' #'+mm if mm else ''})")
|
|
continue
|
|
# fall through: no xref resolved
|
|
new_region.append(f"### {num} — {clean_title(title)} — 0 files")
|
|
if items[num]["status"] in "✅🔶":
|
|
review.append(f" {num} {title[:40]}: 0 files but status {items[num]['status']} — check Not-downloadable block!")
|
|
# legacy blocks for items not in INDEX
|
|
for num in blocks:
|
|
if num not in seen:
|
|
review.append(f" BLOCK {num} has no INDEX item — DROPPED from region: {blocks[num][0] if blocks[num] else ''}")
|
|
|
|
# ---- tails in canonical order
|
|
nd = tails.get("Not downloadable", [])
|
|
if not nd:
|
|
# build from MISSING.md ❌ items (compact)
|
|
mp = os.path.join(BASE, "sections", sec, "MISSING.md")
|
|
nd = ["## Not downloadable (verified 2026-09-28)", "", "| item | reason |", "|------|--------|"]
|
|
if os.path.exists(mp):
|
|
ms = open(mp, encoding="utf-8").read()
|
|
for m2 in re.finditer(r'^### (\d{2}) — (.+?)\n((?:^.*\n){0,3})', ms, re.M):
|
|
i2, ttl, body = m2.group(1), m2.group(2).strip(), m2.group(3)
|
|
reason = re.sub(r'\s+', ' ', body).strip()[:140]
|
|
nd.append(f"| {i2} {ttl[:50]} | {reason} |")
|
|
|
|
# notes tail
|
|
notes = tails.get("Notes", [])
|
|
|
|
# ---- totals (recount)
|
|
n_files = len(dfiles)
|
|
tot_bytes = sum(os.path.getsize(os.path.join(ddir, f)) for f in dfiles)
|
|
n_ru = sum(1 for f in dfiles if re.search(r'-ru[.\-]', f))
|
|
n_en = sum(1 for f in dfiles if re.search(r'-en[.\-]', f))
|
|
allrows = [l for l in new_region if l.startswith("|") and not l.strip("|").split("|")[0].strip().startswith("—")]
|
|
def cnt(pat): return sum(1 for l in allrows if pat in l)
|
|
src = f"lg {cnt('lg f/')} + ia {cnt('ia ')} + flib {cnt('flib')}"
|
|
# keep old header but ensure H1 format
|
|
h1 = [l for l in header if l.startswith("# ")]
|
|
h1 = [h1[0]] if h1 else [f"# Downloads — Section {sec[:2]} {sec[3:]} — FINAL 2026-09-28"]
|
|
out = list(h1) + ["", "## Totals", "",
|
|
f"{n_files} files, {human_mb(tot_bytes)} — RU {n_ru} / EN {n_en} / other {n_files-n_ru-n_en} ({src})",
|
|
"", "## Files (per item)", ""]
|
|
out += new_region
|
|
out += [""] + (nd if nd else [])
|
|
if xref_tail:
|
|
out += ["", "## Cross-references"] + [l for l in xref_tail if not l.startswith("## ")]
|
|
if notes:
|
|
out += [""] + notes
|
|
# trim trailing blank lines
|
|
while out and not out[-1].strip():
|
|
out.pop()
|
|
out.append("")
|
|
|
|
text = "\n".join(out)
|
|
if not DRY:
|
|
open(os.path.join(BASE, "downloads", DL_DIR.get(sec, sec), "MANIFEST.md"), "w", encoding="utf-8").write(text)
|
|
report.append(f" -> {len(items)} blocks after; {n_files} files, {human_mb(tot_bytes)}")
|
|
report.extend(review)
|
|
print("\n".join(report))
|
|
|
|
if __name__ == "__main__":
|
|
main()
|