#!/usr/bin/env python3
"""audit_content.py — content-triage of ALL downloaded files vs their item title.
For each file NN-... in downloads/0N-*/: expected title = per-item header
'### NN —
' of that section's MANIFEST. Extract a text sample
(pdf pdftotext -l 3 / fb2 body / epub first xhtml / djvu djvutxt / doc strings)
and score title-token overlap. Flags: SUSPECT (0 overlap), SMALL, NO-TEXT.
RU files (name contains '-ru') get a light check (Cyrillic present, non-empty).
Run: python3 tools/audit_content.py [sec] (sec = 01..06, default all)
"""
import os, re, sys, subprocess, json, html
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
DL = os.path.join(BASE, 'downloads')
STOP = {'the','and','for','with','from','into','upon','about','its','his','her','vol','vols','chaps','ch','part','study','essays','essay','selected','works','collected','complete','book','pages','original','edition','second','first'}
def tokens(title):
t = re.sub(r'\(.*?\)', ' ', title)
t = re.sub(r'[^a-zа-яё0-9 ]', '', t.lower())
return [w for w in t.split() if len(w) >= 4 and w not in STOP]
def item_titles(sec_dir):
mf = os.path.join(DL, sec_dir, 'MANIFEST.md')
t = open(mf, encoding='utf8').read()
part = t.split('## Not downloadable')[0]
out = {}
for m in re.finditer(r'^### (\d+) — (.+?) — (.+)$', part, re.M):
out[m.group(1)] = m.group(2).strip()
# items with '0 files' headers too (no local files, skip)
return out
def sample(path):
ext = path.rsplit('.', 1)[-1].lower()
try:
if ext == 'pdf':
r = subprocess.run(['pdftotext', '-l', '3', path, '-'], capture_output=True, timeout=120)
txt = r.stdout.decode('utf8', errors='ignore')
if not txt.strip():
r = subprocess.run(['pdftotext', path, '-'], capture_output=True, timeout=300)
txt = r.stdout.decode('utf8', errors='ignore')[:3000]
return txt[:3000]
if ext == 'fb2':
raw = open(path, encoding='utf8', errors='ignore').read()
body = re.search(r'(.*)', raw, re.S)
txt = re.sub(r'<[^>]+>', ' ', body.group(1) if body else raw[:50000])
return re.sub(r'\s+', ' ', txt)[:3000]
if ext == 'epub':
import zipfile
z = zipfile.ZipFile(path)
names = [n for n in z.namelist() if re.search(r'\.(x?html?|htm)$', n) and 'nav' not in n.lower()]
txt = ''
for n in names[:8]:
txt += z.read(n).decode('utf8', errors='ignore')
if len(txt) > 4000: break
return re.sub(r'<[^>]+>', ' ', txt)[:3000]
if ext == 'djvu':
r = subprocess.run(['djvutxt', path, '-'], capture_output=True, timeout=300)
txt = r.stdout.decode('utf8', errors='ignore')
return txt[:3000]
if ext in ('doc', 'txt', 'html', 'mobi'):
r = subprocess.run(['strings', '-n', '6', path], capture_output=True, timeout=60)
return r.stdout.decode('utf8', errors='ignore')[:3000]
except Exception as e:
return f"__ERR__ {e}"
return ''
def main():
secs = sys.argv[1:] or sorted(d for d in os.listdir(DL) if re.match(r'^0\d-', d))
results = []
for sec in secs:
titles = item_titles(sec)
files = [f for f in os.listdir(os.path.join(DL, sec)) if f != 'MANIFEST.md' and re.match(r'^\d\d-', f)]
for f in sorted(files):
num = f[:2]
title = titles.get(num, '')
if not title:
results.append((sec, f, 'NO-TITLE', 0, 0, ''))
continue
tok = tokens(title)
is_ru = '-ru' in f or re.search(r'-(t\d-ru|ru-)', f)
size = os.path.getsize(os.path.join(DL, sec, f))
s = sample(os.path.join(DL, sec, f))
low = s.lower()
hits = sum(1 for w in tok if w in low) if tok else 0
score = hits / len(tok) if tok else 0
flags = []
if s.startswith('__ERR__'):
flags.append('ERR')
elif is_ru:
if not re.search(r'[а-яё]', s): flags.append('NO-CYR')
else:
if not s.strip(): flags.append('NO-TEXT')
elif score == 0 and tok: flags.append('SUSPECT')
if size < 200000 and f.endswith('.pdf'): flags.append('SMALL')
if size < 80000 and not f.endswith('.pdf'): flags.append('SMALL')
if flags:
results.append((sec, f, ','.join(flags), round(score, 2), size, s[:150].replace('\n', ' ')))
for sec, f, fl, sc, sz, snip in results:
print(f"{sec}/{f}\t{fl}\tscore={sc}\t{sz}\t{snip[:80]}")
print(f"\nTOTAL flagged: {len(results)}", file=sys.stderr)
if __name__ == '__main__':
main()