#!/usr/bin/env python3 """tools/make_rootlist.py — generate READING-LIST.md at repo root. Mirrors the ISAP Zurich reading-list page: one table per section, every item with its listed title, author, RU edition, card link, and links to ALL downloaded files (cross-section books link to the canonical section's file, tagged «secNN»). Sources: sections/NN/INDEX.md (items) + downloads/NN/MANIFEST.md (files) + sections/NN/NN-*.md cards (RU title). Regenerable, no manual edits. """ import os, re, sys, datetime def is_en(f): return "-en-" in f or re.search(r"-en(\.b)?\.\w{2,4}$", f) is not None def is_ru(f): return "-ru-" in f or re.search(r"-ru(\.b)?\.\w{2,4}$", f) is not None BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) DL_DIR = {"09-comparison-of-psychodynamic-concepts": "09-comparison", "10-psychopathology-psychiatry": "10-psychopathology"} SECS = ["01-fundamentals", "02-dreams", "03-myths-fairy-tales", "04-pictures", "05-ethnology", "06-religion", "07-complexes", "08-developmental", "09-comparison-of-psychodynamic-concepts", "10-psychopathology-psychiatry", "11-individuation", "12-practical-case"] DATE = datetime.date.today().isoformat() def dldir(sec): return DL_DIR.get(sec, sec) SECNUM_TO_DIR = {s[:2]: dldir(s) for s in SECS} SECNUM_TO_FULL = {s[:2]: s for s in SECS} def human(size): if size >= 1024*1024: v = size/1048576 return (f"{v:.1f}M" if v < 10 else f"{v:.0f}M") return f"{size/1024:.0f}K" def parse_index(sec): p = os.path.join(BASE, "sections", sec, "INDEX.md") h1 = re.search(r'^# Section \d{2} — (.+)$', open(p).read(), re.M).group(1) name = re.sub(r':\s*index.*$|:\s*INDEX.*$', '', h1, flags=re.I).strip() items = {} for line in open(p, encoding="utf-8"): m = re.match(r'^\|\s*(\d{2})\s*\|\s*([^|]*)\|\s*([^|]*)\s*\|\s*([^|]*)\s*\|\s*([^|]+?)\s*\|\s*([^|]+)\|', line) if m: num, grp, title, author, status, card = m.groups() is_xref = status.strip().startswith("xref") or card.strip().endswith("-xref.md") em = re.search(r'[✅🔶❌⬜]', status) items[num] = dict(title=title.strip(), author=author.strip(), status=(em.group(0) if em else status.strip()), is_xref=is_xref, card=card.strip()) return name, items def parse_manifest(sec): """items -> dict(num -> {'files': [(name,size,source)], 'xref': [(target_sec, target_item)]})""" p = os.path.join(BASE, "downloads", dldir(sec), "MANIFEST.md") out = {} cur = None for line in open(p, encoding="utf-8"): m = re.match(r'^### (\d{2}) — .*(?:— files in (sec\d\d(?:\+\d\d)*))?\s*$', line) if line.startswith("### "): num = line.split(" — ")[0].replace("### ", "").strip() cur = num out.setdefault(num, {"files": [], "xref": []}) if re.search(r'— files in sec', line) and not out[num]["files"]: tgts = re.findall(r'sec(\d\d)', line) out[num]["xref"].extend(("sec" + t, None, None) for t in tgts) continue if line.startswith("## "): cur = None continue if cur is None: continue # legacy prose: Файлы: `path1`, `path2`. mp = re.match(r'Файлы:\s*(.+)$', line) if mp: for path in re.findall(r'`([^`]+)`', mp.group(1)): nm = os.path.basename(path) out[cur]["files"].append(("@" + path, "", "")) continue if not line.startswith("|"): continue cells = [c.strip() for c in line.strip("|").split("|")] if not cells or cells[0] in ("file", "") or cells[0].startswith(":"): continue first = cells[0] # legacy 5-col rows: | EN/RU | file | size | source | verify | if first in ("EN", "RU", "DE", "ES", "FR") and len(cells) > 2 and re.search(r'\.\w{2,4}$', cells[1]): out[cur]["files"].append((cells[1], cells[2], cells[3] if len(cells) > 3 else "")) continue m = re.match(r'— \(', first) if m: tfile = None fm = re.search(r'(\d\d-[\w.\-]+\.\w{2,4}|\d\d-[\w.\-]+\*)', first) if fm: tfile = fm.group(1) secmm = re.search(r'sec\d\d', first) out[cur]["xref"].append((secmm.group(0) if secmm else "sec??", None, tfile)) continue for sec_m, item_m in re.findall(r'sec(\d\d)(?: #(\d{2}))?', first): out[cur]["xref"].append(("sec" + sec_m, item_m or None, None)) continue if re.search(r'\.\w{2,4}$', first): size = cells[1] if len(cells) > 1 else "" out[cur]["files"].append((first, size, cells[2] if len(cells) > 2 else "")) return out def size_from_str(s): m = re.match(r'([\d.]+)\s*(MB|GB|KB|pp)', s) if not m: return None v = float(m.group(1)) return {"KB": v*1024, "MB": v*1048576, "GB": v*1073741824}[m.group(2)] def card_path(sec, card): p = os.path.join(BASE, "sections", sec, card) if os.path.exists(p): return p # INDEX name drift: match by item-number prefix num = card[:2] d = os.path.join(BASE, "sections", sec) cands = [f for f in os.listdir(d) if f.startswith(num + "-") and f.endswith(".md")] if len(cands) == 1: return os.path.join(d, cands[0]) return None def card_ru_title(sec, card, _depth=0): p = card_path(sec, card) if p is None: return "" s = open(p, encoding="utf-8").read() m = re.search(r'## Editions — RU\s*\n(.*?)(?=\n## )', s, re.S) if m: for r in [x for x in m.group(1).splitlines() if x.strip().startswith("|")]: cells = [c.strip() for c in r.strip("|").split("|")] if not cells or all(set(c) <= set("-: ") for c in cells): continue # separator if cells[0].lower().endswith("title") or cells[0].startswith("Title"): continue # header if cells[0] in ("EN", "RU", "DE", "ES", "FR", "IT"): continue # lang-first table: no title column tt = cells[0].strip("«» ") if tt.startswith("—"): return "— нет" if tt: return tt[:90] # RU title via cross-ref in Status line (cross-section / xref cards) if _depth < 2: stat = re.search(r'\*\*Status:\*\*([^\n]*)', s) if stat: mm = (re.search(r'cross-ref (?:IWT: )?sec(\d\d) #(\d{2})', stat.group(1)) or re.search(r'see `../(\d\d)-[\w\-]+/`(?:\s*#|item )(\d{2})', stat.group(1))) if mm: full = SECNUM_TO_FULL.get(mm.group(1)) if full: _name, items = idxs[full] it = items.get(mm.group(2)) if it: sub = card_ru_title(full, it["card"], _depth + 1) if sub and sub != "— нет": return f"{sub} (sec{mm.group(1)})" return "" def file_links(sec, files, ddir): out = [] for name, size_s, _src in files: if name.startswith("@"): path = name[1:] if not os.path.exists(os.path.join(BASE, path)): import glob as _glob cands = _glob.glob(os.path.join(BASE, "downloads", "*", os.path.basename(path))) if len(cands) == 1: path = os.path.relpath(cands[0], BASE) else: continue nm = os.path.basename(path) secnum = os.path.basename(os.path.dirname(path))[:2] lang = "RU" if re.search(r'-ru[.\-]', nm) else ("EN" if re.search(r'-en[.\-]', nm) else "") sz = os.path.getsize(os.path.join(BASE, path)) out.append(f"[{lang} {nm.rsplit('.', 1)[-1]} {human(sz)} →sec{secnum}]({path})") continue lang = "RU" if re.search(r'-ru[.\-]', name) else ("EN" if re.search(r'-en[.\-]', name) else "") sz = size_from_str(size_s) or os.path.getsize(os.path.join(BASE, "downloads", ddir, name)) fmt = name.rsplit(".", 1)[-1] tag = f"{lang} {fmt} {human(sz)}" out.append(f"[{tag}](downloads/{ddir}/{name})") return out def mklink(name, ddir, sec, sz): lang = "RU " if re.search(r'-ru[.\-]', name) else ("EN " if re.search(r'-en[.\-]', name) else "") label = f"{lang}{name.rsplit('.', 1)[-1]} {human(sz) if sz else ''} →sec{sec}".replace(" ", " ").strip() return f"[{label}](downloads/{ddir}/{name})" def mkpath(name, sz_hint=""): """build (label, relpath) for a file name or @path entry""" if name.startswith("@"): path = name[1:] import glob as _glob if not os.path.exists(os.path.join(BASE, path)): cands = _glob.glob(os.path.join(BASE, "downloads", "*", os.path.basename(path))) if len(cands) == 1: path = os.path.relpath(cands[0], BASE) else: return None else: return None # caller handles plain names nm = os.path.basename(path) secnum = os.path.basename(os.path.dirname(path))[:2] lang = "RU " if re.search(r'-ru[.\-]', nm) else ("EN " if re.search(r'-en[.\-]', nm) else "") sz = os.path.getsize(os.path.join(BASE, path)) return f"{lang}{nm.rsplit('.', 1)[-1]} {human(sz)} →sec{secnum}", path def main(): # pre-parse all manifests (needed for cross-section resolution) manifs = {sec: parse_manifest(sec) for sec in SECS} global idxs idxs = {} for sec in SECS: idxs[sec] = parse_index(sec) out = [] out.append(f"# ISAP Zurich — Reading List: RU editions & downloaded files ({DATE})") out.append("") out.append("> Все 12 секций. После каждой книги — ссылки на скачанные файлы; у книг из") out.append("> предыдущих секций файлы лежат в канонической секции (тег `secNN` = где лежит файл).") out.append("> `✅` = RU-издание найдено, `🔶` = частично (глава/статья), `❌` = RU нет (verified).") out.append("> Карточка каждой книги — по клику на название. Детали изданий/каталогов — в карточке,") out.append("> провалы поиска — в `sections/NN/MISSING.md`.") out.append("") grand_i = grand_f = grand_l = 0 for sec in SECS: name, items = idxs[sec] mf = manifs[sec] ddir = dldir(sec) n = int(sec[:2]) dfiles = [f for f in os.listdir(os.path.join(BASE, "downloads", ddir)) if f != "MANIFEST.md"] dfiles_n = len(dfiles) f_en = sum(1 for f in dfiles if is_en(f)) f_ru = sum(1 for f in dfiles if is_ru(f)) f_other = dfiles_n - f_en - f_ru stat = {e: 0 for e in "✅🔶❌"} new_items = [it for it in items.values() if not it.get("is_xref")] for it in new_items: stat[it["status"]] = stat.get(it["status"], 0) + 1 en_dl = ru_dl = 0 for num, it in items.items(): if it.get("is_xref"): continue fns = [f[0] for f in mf.get(num, {"files": []}).get("files", [])] if any(is_en(f) for f in fns): en_dl += 1 if any(is_ru(f) for f in fns): ru_dl += 1 xref_n = len(items) - len(new_items) comp = " (все NEW)" if xref_n == 0 else f": {len(new_items)} NEW + {xref_n} cross-ref" ru_found = stat["✅"] + stat["🔶"] ru_dl_s = f"{ru_dl} из {ru_found}" if ru_found else "—" out.append(f"## {n}. {name}") out.append("") out.append(f"{len(items)} книг{comp}\\") out.append(f"RU-издание: ✅ {stat['✅']} найдено · 🔶 {stat['🔶']} частично · ❌ {stat['❌']} нет\\") out.append(f"EN-оригинал скачан: {en_dl} из {len(new_items)}\\") out.append(f"RU-перевод скачан: {ru_dl_s}\\") out.append(f"Файлов: {dfiles_n} (EN {f_en} / RU {f_ru} / other {f_other})") out.append("") out.append("| # | Book | Author | RU edition | Files |") out.append("|---|------|--------|------------|-------|") nfiles_xref = 0 nlinks = 0 for num in sorted(items): it = items[num] md = mf.get(num, {"files": [], "xref": []}) # files: local + resolved xref links = file_links(sec, md["files"], ddir) resolved_secs = set() for tsec, titem, tfile in md["xref"]: tsec2 = tsec[-2:] tmf = manifs.get(SECNUM_TO_FULL.get(tsec2, ""), {}) tinfo = tmf.get(titem, {}) if titem else {} tdir = SECNUM_TO_DIR.get(tsec2) if tfile and tdir: import glob as _glob pats = [os.path.join(BASE, "downloads", tdir, tfile)] if "*" not in tfile else _glob.glob(os.path.join(BASE, "downloads", tdir, tfile)) for pth in pats: if not os.path.exists(pth): continue nm = os.path.basename(pth) links.append(mklink(nm, tdir, tsec2, os.path.getsize(pth))) nfiles_xref += 1 resolved_secs.add(tsec2) continue elif tdir and tinfo.get("files"): for name, size_s, _s in tinfo["files"]: if name.startswith("@"): mp = mkpath(name) if mp: label, path = mp links.append(f"[{label}]({path})") nfiles_xref += 1 else: lang = "RU" if re.search(r'-ru[.\-]', name) else ("EN" if re.search(r'-en[.\-]', name) else "") sz = size_from_str(size_s) or 0 links.append(f"[{lang} {name.rsplit('.',1)[-1]} {human(sz)} →sec{tsec2[-2:]}](downloads/{tdir}/{name})") nfiles_xref += 1 resolved_secs.add(tsec2) elif tdir and tsec2 not in resolved_secs: # target item number unknown -> link to the section manifest links.append(f"[→sec{tsec2[-2:]}](downloads/{tdir}/MANIFEST.md)") # RU title from card ru = card_ru_title(sec, it["card"]) status = it["status"] ru_cell = f"{status} {ru}" if ru and status in "✅🔶" else (status if not ru else ru) if status == "❌": ru_cell = "—" # drop bare fallback links if a resolved file link for the same section exists def bare_sec(lk): m = re.match(r'^\[→sec(\d\d)\]\(', lk) return m.group(1) if m else None resolved_secs2 = set() for lk in links: m = re.search(r'→sec(\d\d)\](?!.*MANIFEST)', lk) if m and not lk.startswith(f"[→sec{m.group(1)}]"): resolved_secs2.add(m.group(1)) links = [lk for lk in links if bare_sec(lk) is None or bare_sec(lk) not in resolved_secs2] links_cell = " · ".join(dict.fromkeys(links)) if links else "—" nlinks += len(dict.fromkeys(links)) # Book link: use real card path if INDEX drifted cp = card_path(sec, it["card"]) card_link = f"sections/{sec}/{os.path.basename(cp)}" if cp else f"sections/{sec}/{it['card']}" out.append(f"| {num} | [{it['title']}]({card_link}) | {it['author']} | {ru_cell} | {links_cell} |") grand_i += len(items) grand_f += dfiles_n grand_l += nlinks out.append("") out.append("---") out.append(f"_Итого: {grand_i} позиций · {grand_f} файлов · {grand_l} ссылок (включая cross-section) · {DATE}._") out.append("") text = "\n".join(out) if "--check" not in sys.argv: open(os.path.join(BASE, "READING-LIST.md"), "w", encoding="utf-8").write(text) print(f"READING-LIST.md written: {len(text)} chars, {text.count(chr(10))} lines") else: print(text[:2000]) if __name__ == "__main__": main()