#!/usr/bin/env python3 """tools/migrate_manifests_canonical.py — normalize ALL manifests to CANON v1 (2026-09-28). CANON (per AGENTS.md + user cross-ref rule 2026-09-26): H1: # Downloads — Section NN — FINAL ## Totals — N files, X MB — RU a / EN b (source breakdown) ## Files (per item) — ONE block per INDEX item, in INDEX order: ### NN — — N files (local file rows) ### NN — <title> — files in secSS (xref-only items) 4-col table: | file | size | source | verify | xref row: | — (cross-ref secSS #MM: …) | | secSS | | ## Not downloadable (verified <date>) — | item | reason | ## Cross-references — summary index (NOT a substitute for per-item blocks) ## Notes Guarantees: markdown-only, NO file moves/deletes. Prints a change report + REVIEW list. Usage: python3 tools/migrate_manifests_canonical.py [--dry] [--sec 12] """ import os, re, sys, datetime sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) import xref as X BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) DRY = "--dry" in sys.argv SEC_ONLY = None if "--sec" in sys.argv: SEC_ONLY = sys.argv[sys.argv.index("--sec") + 1].zfill(2) SECS = ["01-fundamentals", "02-dreams", "03-myths-fairy-tales", "04-pictures", "05-ethnology", "06-religion", "07-complexes", "08-developmental", "09-comparison-of-psychodynamic-concepts", "10-psychopathology-psychiatry", "11-individuation", "12-practical-case"] DL_DIR = {"09-comparison-of-psychodynamic-concepts": "09-comparison", "10-psychopathology-psychiatry": "10-psychopathology"} def human_mb(n): return f"{n/1024/1024:.1f} MB" if n >= 1024*1024 else f"{n/1024:.0f} KB" def parse_index(sec): """items: num -> (title, author, status, card) in INDEX order""" p = os.path.join(BASE, "sections", sec, "INDEX.md") items = {} for line in open(p, encoding="utf-8"): m = re.match(r'^\|\s*(\d{2})\s*\|\s*([^|]*)\|\s*([^|]*)\|\s*([^|]*)\|\s*([^|]+?)\s*\|\s*([^|]+)\|', line) if m: num, grp, title, author, status, card = m.groups() em = re.search(r'[✅🔶❌⬜]', status) items[num] = dict(title=title.strip(), author=author.strip(), status=(em.group(0) if em else status.strip()), status_raw=status.strip(), card=card.strip()) return items def parse_manifest(sec): """returns: header lines, blocks {num: [lines]}, tail sections {name: lines}, xref table rows""" p = os.path.join(BASE, "downloads", DL_DIR.get(sec, sec), "MANIFEST.md") lines = open(p, encoding="utf-8").read().splitlines() header, region, tails = [], [], {} mode, cur = "header", None for ln in lines: if re.match(r'^### \d{2} —', ln): mode = "block" cur = ln.split(" — ", 1)[0].replace("### ", "").strip() region.append(ln) continue if ln.startswith("## "): if mode == "region_start": pass mode = "tail" m = re.match(r'^## (Not downloadable.*|Cross-references.*|Cross-section.*|Notes.*)$', ln) name = (m.group(1).split(" (")[0] if m else "other") tails.setdefault(name, []).append(ln) cur = name continue if ln.startswith("## Files (per item)"): mode = "region_start" header.append(ln) continue if mode == "header": header.append(ln) elif mode in ("region_start", "block"): region.append(ln) else: tails[cur].append(ln) # split region into per-item blocks blocks, cur = {}, None for ln in region: if re.match(r'^### \d{2} —', ln): cur = ln.split(" — ", 1)[0].replace("### ", "").strip() blocks[cur] = [] elif cur is not None: blocks[cur].append(ln) return header, blocks, tails def file_rows(block_lines): """local file rows: first cell is a filename (not a cross-ref '—' row)""" out = [] for ln in block_lines: if not ln.startswith("|"): continue cells = [c.strip() for c in ln.strip("|").split("|")] first = cells[0] if first.startswith("—") or first in ("file", "") or first.startswith(":"): continue if re.search(r'\.\w{2,4}$', first): out.append(ln) return out def xref_rows(block_lines): return [ln for ln in block_lines if ln.startswith("|") and ln.strip("|").split("|")[0].strip().startswith("—")] def norm_heading(num, title, block_lines): fx = xref_rows(block_lines) fl = file_rows(block_lines) if fx and not fl: secs = sorted(set(re.findall(r'sec(\d\d)', " ".join(fx)))) tgt = "+".join(secs) if secs else "??" return f"### {num} — {title} — files in sec{tgt}" n = len(fl) return f"### {num} — {title} — {n} files" def clean_title(rest): """strip trailing annotations from a legacy heading""" rest = re.sub(r'\s*—\s*(\d+\s*files?.*)$', '', rest) rest = re.sub(r'\s*—\s*(files in sec.*)$', '', rest) rest = re.sub(r'\s*—\s*✔.*$', '', rest) rest = re.sub(r'\s*—\s*EN.*$', '', rest) rest = re.sub(r'\s*—\s*RU.*$', '', rest) return rest.strip() or rest def main(): report = [] # cross-section card matcher (xref.py) try: cards = X.card_entries() except Exception as ex: print("card_entries failed:", ex); cards = [] for sec in SECS: if SEC_ONLY and sec[:2] != SEC_ONLY: continue items = parse_index(sec) header, blocks, tails = parse_manifest(sec) ddir = os.path.join(BASE, "downloads", DL_DIR.get(sec, sec)) dfiles = sorted(f for f in os.listdir(ddir) if f != "MANIFEST.md" and (re.match(r'^\d\d-', f) or f.startswith("bonus"))) # xref map from trailing Cross-references table: item -> "secSS #MM" xmap = {} xref_tail = [] for name in ("Cross-references", "Cross-section", "other"): if name in tails: xref_tail = tails[name] for ln in tails[name]: if not ln.startswith("|") or ln.startswith("| item") or ln.startswith("|---"): continue cells = [c.strip() for c in ln.strip("|").split("|")] if cells and re.match(r'^\d{2}', cells[0]): num_x = cells[0][:2] m_t = re.search(r'sec(\d{2})|(?:(\d{2}) #)', ln) if m_t: xmap[num_x] = ("sec" + (m_t.group(1) or m_t.group(2)), ln) break # first found section-name wins report.append(f"== {sec}: {len(items)} items, {len(blocks)} blocks before") # ---- build new per-item region new_region, seen = [], set() all_old_lines = "\n".join(l for bl in blocks.values() for l in bl) review = [] for num in sorted(items): seen.add(num) title = items[num]["title"] if num in blocks: bl = blocks[num] t = clean_title(title) new_region.append(norm_heading(num, t, bl)) new_region.extend(bl) else: local = [f for f in dfiles if f.startswith(num + "-")] if local: # find rows for these files wherever they are rows = [] for f in local: m = [ln for ln in all_old_lines.splitlines() if f in ln and ln.startswith("|")] rows.extend(m[:1] if m else [f"| {f} | {human_mb(os.path.getsize(os.path.join(ddir,f)))} | UNKNOWN | REVIEW |"]) new_region.append(f"### {num} — {clean_title(title)} — {len(local)} files") new_region.extend(rows) review.append(f" {num} {title[:40]}: block ADDED ({len(local)} local files, rows moved/rebuilt)") elif num in xmap or re.match(r'^xref\s+sec\d+', items[num]["status_raw"]) or (not local and cards): sr = re.search(r'xref\s+sec(\d{2})(?:\s+#(\d{2}))?', items[num]["status_raw"]) if sr: tgt_s, mm = "sec" + sr.group(1), sr.group(2) elif num in xmap: tgt_s, mm = xmap[num][0], None else: # cross-section card match (earlier sections only) best, best_sc = None, 0.0 e_tok = X.norm(clean_title(title)) for c in cards: if c['sec'] == sec[:2]: continue sc = X.score_pair(e_tok, None, items[num]['author'], c['tok'], c['cw'], c['author']) if sc > best_sc: best, best_sc = c, sc ok = False if best and int(best['sec']) < int(sec[:2]): if best_sc >= 0.55: ok = True elif best_sc >= 0.5: # lenient: title subset + shared surname token (e.g. Cambray sec12#06 -> sec01#09) def surnames(a): return {w.lower().strip(".(), ") for w in (a or "").replace("&", " ").split() if w[:1].isupper()} subset = e_tok and best['tok'] and (e_tok <= best['tok'] or best['tok'] <= e_tok) ok = subset and bool(surnames(items[num]['author']) & surnames(best['author'])) if ok: tgt_s, mm = "sec" + best['sec'], best['n'] else: tgt_s, mm = None, None if tgt_s is None: pass # fall through to 0 files (below) else: new_region.append(f"### {num} — {clean_title(title)} — files in {tgt_s}") new_region.append(f"| — (cross-ref {tgt_s}{' #' + mm if mm else ''}: see Cross-references) | | {tgt_s[3:]} | |") review.append(f" {num} {title[:40]}: xref block ADDED ({tgt_s}{' #'+mm if mm else ''})") continue # fall through: no xref resolved new_region.append(f"### {num} — {clean_title(title)} — 0 files") if items[num]["status"] in "✅🔶": review.append(f" {num} {title[:40]}: 0 files but status {items[num]['status']} — check Not-downloadable block!") # legacy blocks for items not in INDEX for num in blocks: if num not in seen: review.append(f" BLOCK {num} has no INDEX item — DROPPED from region: {blocks[num][0] if blocks[num] else ''}") # ---- tails in canonical order nd = tails.get("Not downloadable", []) if not nd: # build from MISSING.md ❌ items (compact) mp = os.path.join(BASE, "sections", sec, "MISSING.md") nd = ["## Not downloadable (verified 2026-09-28)", "", "| item | reason |", "|------|--------|"] if os.path.exists(mp): ms = open(mp, encoding="utf-8").read() for m2 in re.finditer(r'^### (\d{2}) — (.+?)\n((?:^.*\n){0,3})', ms, re.M): i2, ttl, body = m2.group(1), m2.group(2).strip(), m2.group(3) reason = re.sub(r'\s+', ' ', body).strip()[:140] nd.append(f"| {i2} {ttl[:50]} | {reason} |") # notes tail notes = tails.get("Notes", []) # ---- totals (recount) n_files = len(dfiles) tot_bytes = sum(os.path.getsize(os.path.join(ddir, f)) for f in dfiles) n_ru = sum(1 for f in dfiles if re.search(r'-ru[.\-]', f)) n_en = sum(1 for f in dfiles if re.search(r'-en[.\-]', f)) allrows = [l for l in new_region if l.startswith("|") and not l.strip("|").split("|")[0].strip().startswith("—")] def cnt(pat): return sum(1 for l in allrows if pat in l) src = f"lg {cnt('lg f/')} + ia {cnt('ia ')} + flib {cnt('flib')}" # keep old header but ensure H1 format h1 = [l for l in header if l.startswith("# ")] h1 = [h1[0]] if h1 else [f"# Downloads — Section {sec[:2]} {sec[3:]} — FINAL 2026-09-28"] out = list(h1) + ["", "## Totals", "", f"{n_files} files, {human_mb(tot_bytes)} — RU {n_ru} / EN {n_en} / other {n_files-n_ru-n_en} ({src})", "", "## Files (per item)", ""] out += new_region out += [""] + (nd if nd else []) if xref_tail: out += ["", "## Cross-references"] + [l for l in xref_tail if not l.startswith("## ")] if notes: out += [""] + notes # trim trailing blank lines while out and not out[-1].strip(): out.pop() out.append("") text = "\n".join(out) if not DRY: open(os.path.join(BASE, "downloads", DL_DIR.get(sec, sec), "MANIFEST.md"), "w", encoding="utf-8").write(text) report.append(f" -> {len(items)} blocks after; {n_files} files, {human_mb(tot_bytes)}") report.extend(review) print("\n".join(report)) if __name__ == "__main__": main()