manifests 01-12: canonical v1 full migration (tools/migrate_manifests_canonical.py)
- per-item blocks for EVERY INDEX item (01: 49->55, 02: 28->57, 03: 39->68, 04: 12->39, 05: 23->29); xref blocks: sec02#21->sec01#44, sec12#03/06/13/18/21 - heading normalization (no status/✔/prose tails), INDEX order everywhere (sec12 27/28/31 repositioned) - tail blocks in canonical order: Not downloadable -> Cross-references -> Notes (sec04 ND block built from MISSING.md + #16 row; sec03 + #13/#28 rows) - 2 orphan file rows added: sec06 #26 scholem en-b.epub, sec07 #02 cw3-ru fb2 - Totals recounted from disk (all 12 sections verified: files = mentions) - 0 file rows lost (git old-vs-new diff check) - xref.py stale-line check: clean Prep step for the repo-root READING-LIST.md generator (per user).
This commit is contained in:
parent
a39f2c1ec6
commit
2847f1bcea
13 changed files with 945 additions and 387 deletions
282
tools/migrate_manifests_canonical.py
Normal file
282
tools/migrate_manifests_canonical.py
Normal file
|
|
@ -0,0 +1,282 @@
|
|||
#!/usr/bin/env python3
|
||||
"""tools/migrate_manifests_canonical.py — normalize ALL manifests to CANON v1 (2026-09-28).
|
||||
|
||||
CANON (per AGENTS.md + user cross-ref rule 2026-09-26):
|
||||
H1: # Downloads — Section NN <Name> — FINAL <date>
|
||||
## Totals — N files, X MB — RU a / EN b (source breakdown)
|
||||
## Files (per item) — ONE block per INDEX item, in INDEX order:
|
||||
### NN — <title> — N files (local file rows)
|
||||
### NN — <title> — files in secSS (xref-only items)
|
||||
4-col table: | file | size | source | verify |
|
||||
xref row: | — (cross-ref secSS #MM: …) | | secSS | |
|
||||
## Not downloadable (verified <date>) — | item | reason |
|
||||
## Cross-references — summary index (NOT a substitute for per-item blocks)
|
||||
## Notes
|
||||
|
||||
Guarantees: markdown-only, NO file moves/deletes. Prints a change report + REVIEW list.
|
||||
Usage: python3 tools/migrate_manifests_canonical.py [--dry] [--sec 12]
|
||||
"""
|
||||
import os, re, sys, datetime
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
import xref as X
|
||||
|
||||
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
DRY = "--dry" in sys.argv
|
||||
SEC_ONLY = None
|
||||
if "--sec" in sys.argv:
|
||||
SEC_ONLY = sys.argv[sys.argv.index("--sec") + 1].zfill(2)
|
||||
|
||||
SECS = ["01-fundamentals", "02-dreams", "03-myths-fairy-tales", "04-pictures",
|
||||
"05-ethnology", "06-religion", "07-complexes", "08-developmental",
|
||||
"09-comparison-of-psychodynamic-concepts", "10-psychopathology-psychiatry",
|
||||
"11-individuation", "12-practical-case"]
|
||||
|
||||
DL_DIR = {"09-comparison-of-psychodynamic-concepts": "09-comparison",
|
||||
"10-psychopathology-psychiatry": "10-psychopathology"}
|
||||
|
||||
def human_mb(n):
|
||||
return f"{n/1024/1024:.1f} MB" if n >= 1024*1024 else f"{n/1024:.0f} KB"
|
||||
|
||||
def parse_index(sec):
|
||||
"""items: num -> (title, author, status, card) in INDEX order"""
|
||||
p = os.path.join(BASE, "sections", sec, "INDEX.md")
|
||||
items = {}
|
||||
for line in open(p, encoding="utf-8"):
|
||||
m = re.match(r'^\|\s*(\d{2})\s*\|\s*([^|]*)\|\s*([^|]*)\|\s*([^|]*)\|\s*([^|]+?)\s*\|\s*([^|]+)\|', line)
|
||||
if m:
|
||||
num, grp, title, author, status, card = m.groups()
|
||||
em = re.search(r'[✅🔶❌⬜]', status)
|
||||
items[num] = dict(title=title.strip(), author=author.strip(),
|
||||
status=(em.group(0) if em else status.strip()),
|
||||
status_raw=status.strip(), card=card.strip())
|
||||
return items
|
||||
|
||||
def parse_manifest(sec):
|
||||
"""returns: header lines, blocks {num: [lines]}, tail sections {name: lines}, xref table rows"""
|
||||
p = os.path.join(BASE, "downloads", DL_DIR.get(sec, sec), "MANIFEST.md")
|
||||
lines = open(p, encoding="utf-8").read().splitlines()
|
||||
header, region, tails = [], [], {}
|
||||
mode, cur = "header", None
|
||||
for ln in lines:
|
||||
if re.match(r'^### \d{2} —', ln):
|
||||
mode = "block"
|
||||
cur = ln.split(" — ", 1)[0].replace("### ", "").strip()
|
||||
region.append(ln)
|
||||
continue
|
||||
if ln.startswith("## "):
|
||||
if mode == "region_start":
|
||||
pass
|
||||
mode = "tail"
|
||||
m = re.match(r'^## (Not downloadable.*|Cross-references.*|Cross-section.*|Notes.*)$', ln)
|
||||
name = (m.group(1).split(" (")[0] if m else "other")
|
||||
tails.setdefault(name, []).append(ln)
|
||||
cur = name
|
||||
continue
|
||||
if ln.startswith("## Files (per item)"):
|
||||
mode = "region_start"
|
||||
header.append(ln)
|
||||
continue
|
||||
if mode == "header":
|
||||
header.append(ln)
|
||||
elif mode in ("region_start", "block"):
|
||||
region.append(ln)
|
||||
else:
|
||||
tails[cur].append(ln)
|
||||
# split region into per-item blocks
|
||||
blocks, cur = {}, None
|
||||
for ln in region:
|
||||
if re.match(r'^### \d{2} —', ln):
|
||||
cur = ln.split(" — ", 1)[0].replace("### ", "").strip()
|
||||
blocks[cur] = []
|
||||
elif cur is not None:
|
||||
blocks[cur].append(ln)
|
||||
return header, blocks, tails
|
||||
|
||||
def file_rows(block_lines):
|
||||
"""local file rows: first cell is a filename (not a cross-ref '—' row)"""
|
||||
out = []
|
||||
for ln in block_lines:
|
||||
if not ln.startswith("|"):
|
||||
continue
|
||||
cells = [c.strip() for c in ln.strip("|").split("|")]
|
||||
first = cells[0]
|
||||
if first.startswith("—") or first in ("file", "") or first.startswith(":"):
|
||||
continue
|
||||
if re.search(r'\.\w{2,4}$', first):
|
||||
out.append(ln)
|
||||
return out
|
||||
|
||||
def xref_rows(block_lines):
|
||||
return [ln for ln in block_lines if ln.startswith("|") and ln.strip("|").split("|")[0].strip().startswith("—")]
|
||||
|
||||
def norm_heading(num, title, block_lines):
|
||||
fx = xref_rows(block_lines)
|
||||
fl = file_rows(block_lines)
|
||||
if fx and not fl:
|
||||
secs = sorted(set(re.findall(r'sec(\d\d)', " ".join(fx))))
|
||||
tgt = "+".join(secs) if secs else "??"
|
||||
return f"### {num} — {title} — files in sec{tgt}"
|
||||
n = len(fl)
|
||||
return f"### {num} — {title} — {n} files"
|
||||
|
||||
def clean_title(rest):
|
||||
"""strip trailing annotations from a legacy heading"""
|
||||
rest = re.sub(r'\s*—\s*(\d+\s*files?.*)$', '', rest)
|
||||
rest = re.sub(r'\s*—\s*(files in sec.*)$', '', rest)
|
||||
rest = re.sub(r'\s*—\s*✔.*$', '', rest)
|
||||
rest = re.sub(r'\s*—\s*EN.*$', '', rest)
|
||||
rest = re.sub(r'\s*—\s*RU.*$', '', rest)
|
||||
return rest.strip() or rest
|
||||
|
||||
def main():
|
||||
report = []
|
||||
# cross-section card matcher (xref.py)
|
||||
try:
|
||||
cards = X.card_entries()
|
||||
except Exception as ex:
|
||||
print("card_entries failed:", ex); cards = []
|
||||
for sec in SECS:
|
||||
if SEC_ONLY and sec[:2] != SEC_ONLY:
|
||||
continue
|
||||
items = parse_index(sec)
|
||||
header, blocks, tails = parse_manifest(sec)
|
||||
ddir = os.path.join(BASE, "downloads", DL_DIR.get(sec, sec))
|
||||
dfiles = sorted(f for f in os.listdir(ddir) if f != "MANIFEST.md" and (re.match(r'^\d\d-', f) or f.startswith("bonus")))
|
||||
|
||||
# xref map from trailing Cross-references table: item -> "secSS #MM"
|
||||
xmap = {}
|
||||
xref_tail = []
|
||||
for name in ("Cross-references", "Cross-section", "other"):
|
||||
if name in tails:
|
||||
xref_tail = tails[name]
|
||||
for ln in tails[name]:
|
||||
if not ln.startswith("|") or ln.startswith("| item") or ln.startswith("|---"):
|
||||
continue
|
||||
cells = [c.strip() for c in ln.strip("|").split("|")]
|
||||
if cells and re.match(r'^\d{2}', cells[0]):
|
||||
num_x = cells[0][:2]
|
||||
m_t = re.search(r'sec(\d{2})|(?:(\d{2}) #)', ln)
|
||||
if m_t:
|
||||
xmap[num_x] = ("sec" + (m_t.group(1) or m_t.group(2)), ln)
|
||||
break # first found section-name wins
|
||||
report.append(f"== {sec}: {len(items)} items, {len(blocks)} blocks before")
|
||||
|
||||
# ---- build new per-item region
|
||||
new_region, seen = [], set()
|
||||
all_old_lines = "\n".join(l for bl in blocks.values() for l in bl)
|
||||
review = []
|
||||
for num in sorted(items):
|
||||
seen.add(num)
|
||||
title = items[num]["title"]
|
||||
if num in blocks:
|
||||
bl = blocks[num]
|
||||
t = clean_title(title)
|
||||
new_region.append(norm_heading(num, t, bl))
|
||||
new_region.extend(bl)
|
||||
else:
|
||||
local = [f for f in dfiles if f.startswith(num + "-")]
|
||||
if local:
|
||||
# find rows for these files wherever they are
|
||||
rows = []
|
||||
for f in local:
|
||||
m = [ln for ln in all_old_lines.splitlines() if f in ln and ln.startswith("|")]
|
||||
rows.extend(m[:1] if m else [f"| {f} | {human_mb(os.path.getsize(os.path.join(ddir,f)))} | UNKNOWN | REVIEW |"])
|
||||
new_region.append(f"### {num} — {clean_title(title)} — {len(local)} files")
|
||||
new_region.extend(rows)
|
||||
review.append(f" {num} {title[:40]}: block ADDED ({len(local)} local files, rows moved/rebuilt)")
|
||||
elif num in xmap or re.match(r'^xref\s+sec\d+', items[num]["status_raw"]) or (not local and cards):
|
||||
sr = re.search(r'xref\s+sec(\d{2})(?:\s+#(\d{2}))?', items[num]["status_raw"])
|
||||
if sr:
|
||||
tgt_s, mm = "sec" + sr.group(1), sr.group(2)
|
||||
elif num in xmap:
|
||||
tgt_s, mm = xmap[num][0], None
|
||||
else:
|
||||
# cross-section card match (earlier sections only)
|
||||
best, best_sc = None, 0.0
|
||||
e_tok = X.norm(clean_title(title))
|
||||
for c in cards:
|
||||
if c['sec'] == sec[:2]:
|
||||
continue
|
||||
sc = X.score_pair(e_tok, None, items[num]['author'], c['tok'], c['cw'], c['author'])
|
||||
if sc > best_sc:
|
||||
best, best_sc = c, sc
|
||||
ok = False
|
||||
if best and int(best['sec']) < int(sec[:2]):
|
||||
if best_sc >= 0.55:
|
||||
ok = True
|
||||
elif best_sc >= 0.5:
|
||||
# lenient: title subset + shared surname token (e.g. Cambray sec12#06 -> sec01#09)
|
||||
def surnames(a):
|
||||
return {w.lower().strip(".(), ") for w in (a or "").replace("&", " ").split() if w[:1].isupper()}
|
||||
subset = e_tok and best['tok'] and (e_tok <= best['tok'] or best['tok'] <= e_tok)
|
||||
ok = subset and bool(surnames(items[num]['author']) & surnames(best['author']))
|
||||
if ok:
|
||||
tgt_s, mm = "sec" + best['sec'], best['n']
|
||||
else:
|
||||
tgt_s, mm = None, None
|
||||
if tgt_s is None:
|
||||
pass # fall through to 0 files (below)
|
||||
else:
|
||||
new_region.append(f"### {num} — {clean_title(title)} — files in {tgt_s}")
|
||||
new_region.append(f"| — (cross-ref {tgt_s}{' #' + mm if mm else ''}: see Cross-references) | | {tgt_s[3:]} | |")
|
||||
review.append(f" {num} {title[:40]}: xref block ADDED ({tgt_s}{' #'+mm if mm else ''})")
|
||||
continue
|
||||
# fall through: no xref resolved
|
||||
new_region.append(f"### {num} — {clean_title(title)} — 0 files")
|
||||
if items[num]["status"] in "✅🔶":
|
||||
review.append(f" {num} {title[:40]}: 0 files but status {items[num]['status']} — check Not-downloadable block!")
|
||||
# legacy blocks for items not in INDEX
|
||||
for num in blocks:
|
||||
if num not in seen:
|
||||
review.append(f" BLOCK {num} has no INDEX item — DROPPED from region: {blocks[num][0] if blocks[num] else ''}")
|
||||
|
||||
# ---- tails in canonical order
|
||||
nd = tails.get("Not downloadable", [])
|
||||
if not nd:
|
||||
# build from MISSING.md ❌ items (compact)
|
||||
mp = os.path.join(BASE, "sections", sec, "MISSING.md")
|
||||
nd = ["## Not downloadable (verified 2026-09-28)", "", "| item | reason |", "|------|--------|"]
|
||||
if os.path.exists(mp):
|
||||
ms = open(mp, encoding="utf-8").read()
|
||||
for m2 in re.finditer(r'^### (\d{2}) — (.+?)\n((?:^.*\n){0,3})', ms, re.M):
|
||||
i2, ttl, body = m2.group(1), m2.group(2).strip(), m2.group(3)
|
||||
reason = re.sub(r'\s+', ' ', body).strip()[:140]
|
||||
nd.append(f"| {i2} {ttl[:50]} | {reason} |")
|
||||
|
||||
# notes tail
|
||||
notes = tails.get("Notes", [])
|
||||
|
||||
# ---- totals (recount)
|
||||
n_files = len(dfiles)
|
||||
tot_bytes = sum(os.path.getsize(os.path.join(ddir, f)) for f in dfiles)
|
||||
n_ru = sum(1 for f in dfiles if re.search(r'-ru[.\-]', f))
|
||||
n_en = sum(1 for f in dfiles if re.search(r'-en[.\-]', f))
|
||||
allrows = [l for l in new_region if l.startswith("|") and not l.strip("|").split("|")[0].strip().startswith("—")]
|
||||
def cnt(pat): return sum(1 for l in allrows if pat in l)
|
||||
src = f"lg {cnt('lg f/')} + ia {cnt('ia ')} + flib {cnt('flib')}"
|
||||
# keep old header but ensure H1 format
|
||||
h1 = [l for l in header if l.startswith("# ")]
|
||||
h1 = [h1[0]] if h1 else [f"# Downloads — Section {sec[:2]} {sec[3:]} — FINAL 2026-09-28"]
|
||||
out = list(h1) + ["", "## Totals", "",
|
||||
f"{n_files} files, {human_mb(tot_bytes)} — RU {n_ru} / EN {n_en} / other {n_files-n_ru-n_en} ({src})",
|
||||
"", "## Files (per item)", ""]
|
||||
out += new_region
|
||||
out += [""] + (nd if nd else [])
|
||||
if xref_tail:
|
||||
out += ["", "## Cross-references"] + [l for l in xref_tail if not l.startswith("## ")]
|
||||
if notes:
|
||||
out += [""] + notes
|
||||
# trim trailing blank lines
|
||||
while out and not out[-1].strip():
|
||||
out.pop()
|
||||
out.append("")
|
||||
|
||||
text = "\n".join(out)
|
||||
if not DRY:
|
||||
open(os.path.join(BASE, "downloads", DL_DIR.get(sec, sec), "MANIFEST.md"), "w", encoding="utf-8").write(text)
|
||||
report.append(f" -> {len(items)} blocks after; {n_files} files, {human_mb(tot_bytes)}")
|
||||
report.extend(review)
|
||||
print("\n".join(report))
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Loading…
Add table
Add a link
Reference in a new issue