#!/usr/bin/env python3 """audit_content.py — content-triage of ALL downloaded files vs their item title. For each file NN-... in downloads/0N-*/: expected title = per-item header '### NN — ' of that section's MANIFEST. Extract a text sample (pdf pdftotext -l 3 / fb2 body / epub first xhtml / djvu djvutxt / doc strings) and score title-token overlap. Flags: SUSPECT (0 overlap), SMALL, NO-TEXT. RU files (name contains '-ru') get a light check (Cyrillic present, non-empty). Run: python3 tools/audit_content.py [sec] (sec = 01..06, default all) """ import os, re, sys, subprocess, json, html BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) DL = os.path.join(BASE, 'downloads') STOP = {'the','and','for','with','from','into','upon','about','its','his','her','vol','vols','chaps','ch','part','study','essays','essay','selected','works','collected','complete','book','pages','original','edition','second','first'} def tokens(title): t = re.sub(r'\(.*?\)', ' ', title) t = re.sub(r'[^a-zа-яё0-9 ]', '', t.lower()) return [w for w in t.split() if len(w) >= 4 and w not in STOP] def item_titles(sec_dir): mf = os.path.join(DL, sec_dir, 'MANIFEST.md') t = open(mf, encoding='utf8').read() part = t.split('## Not downloadable')[0] out = {} for m in re.finditer(r'^### (\d+) — (.+?) — (.+)$', part, re.M): out[m.group(1)] = m.group(2).strip() # items with '0 files' headers too (no local files, skip) return out def sample(path): ext = path.rsplit('.', 1)[-1].lower() try: if ext == 'pdf': r = subprocess.run(['pdftotext', '-l', '3', path, '-'], capture_output=True, timeout=120) txt = r.stdout.decode('utf8', errors='ignore') if not txt.strip(): r = subprocess.run(['pdftotext', path, '-'], capture_output=True, timeout=300) txt = r.stdout.decode('utf8', errors='ignore')[:3000] return txt[:3000] if ext == 'fb2': raw = open(path, encoding='utf8', errors='ignore').read() body = re.search(r'<body>(.*)', raw, re.S) txt = re.sub(r'<[^>]+>', ' ', body.group(1) if body else raw[:50000]) return re.sub(r'\s+', ' ', txt)[:3000] if ext == 'epub': import zipfile z = zipfile.ZipFile(path) names = [n for n in z.namelist() if re.search(r'\.(x?html?|htm)$', n) and 'nav' not in n.lower()] txt = '' for n in names[:8]: txt += z.read(n).decode('utf8', errors='ignore') if len(txt) > 4000: break return re.sub(r'<[^>]+>', ' ', txt)[:3000] if ext == 'djvu': r = subprocess.run(['djvutxt', path, '-'], capture_output=True, timeout=300) txt = r.stdout.decode('utf8', errors='ignore') return txt[:3000] if ext in ('doc', 'txt', 'html', 'mobi'): r = subprocess.run(['strings', '-n', '6', path], capture_output=True, timeout=60) return r.stdout.decode('utf8', errors='ignore')[:3000] except Exception as e: return f"__ERR__ {e}" return '' def main(): secs = sys.argv[1:] or sorted(d for d in os.listdir(DL) if re.match(r'^0\d-', d)) results = [] for sec in secs: titles = item_titles(sec) files = [f for f in os.listdir(os.path.join(DL, sec)) if f != 'MANIFEST.md' and re.match(r'^\d\d-', f)] for f in sorted(files): num = f[:2] title = titles.get(num, '') if not title: results.append((sec, f, 'NO-TITLE', 0, 0, '')) continue tok = tokens(title) is_ru = '-ru' in f or re.search(r'-(t\d-ru|ru-)', f) size = os.path.getsize(os.path.join(DL, sec, f)) s = sample(os.path.join(DL, sec, f)) low = s.lower() hits = sum(1 for w in tok if w in low) if tok else 0 score = hits / len(tok) if tok else 0 flags = [] if s.startswith('__ERR__'): flags.append('ERR') elif is_ru: if not re.search(r'[а-яё]', s): flags.append('NO-CYR') else: if not s.strip(): flags.append('NO-TEXT') elif score == 0 and tok: flags.append('SUSPECT') if size < 200000 and f.endswith('.pdf'): flags.append('SMALL') if size < 80000 and not f.endswith('.pdf'): flags.append('SMALL') if flags: results.append((sec, f, ','.join(flags), round(score, 2), size, s[:150].replace('\n', ' '))) for sec, f, fl, sc, sz, snip in results: print(f"{sec}/{f}\t{fl}\tscore={sc}\t{sz}\t{snip[:80]}") print(f"\nTOTAL flagged: {len(results)}", file=sys.stderr) if __name__ == '__main__': main()