- tools/audit_content.py: per-file vs item-title triage (SUSPECT/SMALL/NO-TEXT/NO-CYR) - sec01: .part removed; 10 files restored .pdf ext (LFS rename); #04/#19/#27 verify notes - sec02 #16: corrupted double-encoded txt -> clean flib fb2 b/844509; #17/#22 doc verified via catdoc - sec03 #11: -b epub = Spanish von Franz (Paidós 1983, forged EN OPF) -> bonus label - sec03 #42: КАРО 2012 'Irish Tales' = EN reader (not RU, not the listed book) -> bonus rename - sec03 #56: Onians b/c/d/e = Cambridge 24-87KB previews -> labeled - sec06 #18: filename 1981->1989 (Perera 'Descent to the Goddess' 1989; 1981 = other book) - sec06 #28: ia Zimmer OCR layer = foreign Devanagari text; images = RKP (p.100 verified) - sec04 #30: buksmart 2020 = RU/FR bilingual w/ VeryPDF watermarks -> note - linter: azw3/mobi ext + underscore in fname regex; manifest cross-check ignores off-disk rows; 8 legacy cards section-order fixed (CR before DL) - AGENTS.md: 'Content audit (2026-09-25)' lessons section - check-md: 0/219
103 lines
4.7 KiB
Python
103 lines
4.7 KiB
Python
#!/usr/bin/env python3
|
||
"""audit_content.py — content-triage of ALL downloaded files vs their item title.
|
||
|
||
For each file NN-... in downloads/0N-*/: expected title = per-item header
|
||
'### NN — <title>' of that section's MANIFEST. Extract a text sample
|
||
(pdf pdftotext -l 3 / fb2 body / epub first xhtml / djvu djvutxt / doc strings)
|
||
and score title-token overlap. Flags: SUSPECT (0 overlap), SMALL, NO-TEXT.
|
||
RU files (name contains '-ru') get a light check (Cyrillic present, non-empty).
|
||
Run: python3 tools/audit_content.py [sec] (sec = 01..06, default all)
|
||
"""
|
||
import os, re, sys, subprocess, json, html
|
||
|
||
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||
DL = os.path.join(BASE, 'downloads')
|
||
STOP = {'the','and','for','with','from','into','upon','about','its','his','her','vol','vols','chaps','ch','part','study','essays','essay','selected','works','collected','complete','book','pages','original','edition','second','first'}
|
||
|
||
def tokens(title):
|
||
t = re.sub(r'\(.*?\)', ' ', title)
|
||
t = re.sub(r'[^a-zа-яё0-9 ]', '', t.lower())
|
||
return [w for w in t.split() if len(w) >= 4 and w not in STOP]
|
||
|
||
def item_titles(sec_dir):
|
||
mf = os.path.join(DL, sec_dir, 'MANIFEST.md')
|
||
t = open(mf, encoding='utf8').read()
|
||
part = t.split('## Not downloadable')[0]
|
||
out = {}
|
||
for m in re.finditer(r'^### (\d+) — (.+?) — (.+)$', part, re.M):
|
||
out[m.group(1)] = m.group(2).strip()
|
||
# items with '0 files' headers too (no local files, skip)
|
||
return out
|
||
|
||
def sample(path):
|
||
ext = path.rsplit('.', 1)[-1].lower()
|
||
try:
|
||
if ext == 'pdf':
|
||
r = subprocess.run(['pdftotext', '-l', '3', path, '-'], capture_output=True, timeout=120)
|
||
txt = r.stdout.decode('utf8', errors='ignore')
|
||
if not txt.strip():
|
||
r = subprocess.run(['pdftotext', path, '-'], capture_output=True, timeout=300)
|
||
txt = r.stdout.decode('utf8', errors='ignore')[:3000]
|
||
return txt[:3000]
|
||
if ext == 'fb2':
|
||
raw = open(path, encoding='utf8', errors='ignore').read()
|
||
body = re.search(r'<body>(.*)', raw, re.S)
|
||
txt = re.sub(r'<[^>]+>', ' ', body.group(1) if body else raw[:50000])
|
||
return re.sub(r'\s+', ' ', txt)[:3000]
|
||
if ext == 'epub':
|
||
import zipfile
|
||
z = zipfile.ZipFile(path)
|
||
names = [n for n in z.namelist() if re.search(r'\.(x?html?|htm)$', n) and 'nav' not in n.lower()]
|
||
txt = ''
|
||
for n in names[:8]:
|
||
txt += z.read(n).decode('utf8', errors='ignore')
|
||
if len(txt) > 4000: break
|
||
return re.sub(r'<[^>]+>', ' ', txt)[:3000]
|
||
if ext == 'djvu':
|
||
r = subprocess.run(['djvutxt', path, '-'], capture_output=True, timeout=300)
|
||
txt = r.stdout.decode('utf8', errors='ignore')
|
||
return txt[:3000]
|
||
if ext in ('doc', 'txt', 'html', 'mobi'):
|
||
r = subprocess.run(['strings', '-n', '6', path], capture_output=True, timeout=60)
|
||
return r.stdout.decode('utf8', errors='ignore')[:3000]
|
||
except Exception as e:
|
||
return f"__ERR__ {e}"
|
||
return ''
|
||
|
||
def main():
|
||
secs = sys.argv[1:] or sorted(d for d in os.listdir(DL) if re.match(r'^0\d-', d))
|
||
results = []
|
||
for sec in secs:
|
||
titles = item_titles(sec)
|
||
files = [f for f in os.listdir(os.path.join(DL, sec)) if f != 'MANIFEST.md' and re.match(r'^\d\d-', f)]
|
||
for f in sorted(files):
|
||
num = f[:2]
|
||
title = titles.get(num, '')
|
||
if not title:
|
||
results.append((sec, f, 'NO-TITLE', 0, 0, ''))
|
||
continue
|
||
tok = tokens(title)
|
||
is_ru = '-ru' in f or re.search(r'-(t\d-ru|ru-)', f)
|
||
size = os.path.getsize(os.path.join(DL, sec, f))
|
||
s = sample(os.path.join(DL, sec, f))
|
||
low = s.lower()
|
||
hits = sum(1 for w in tok if w in low) if tok else 0
|
||
score = hits / len(tok) if tok else 0
|
||
flags = []
|
||
if s.startswith('__ERR__'):
|
||
flags.append('ERR')
|
||
elif is_ru:
|
||
if not re.search(r'[а-яё]', s): flags.append('NO-CYR')
|
||
else:
|
||
if not s.strip(): flags.append('NO-TEXT')
|
||
elif score == 0 and tok: flags.append('SUSPECT')
|
||
if size < 200000 and f.endswith('.pdf'): flags.append('SMALL')
|
||
if size < 80000 and not f.endswith('.pdf'): flags.append('SMALL')
|
||
if flags:
|
||
results.append((sec, f, ','.join(flags), round(score, 2), size, s[:150].replace('\n', ' ')))
|
||
for sec, f, fl, sc, sz, snip in results:
|
||
print(f"{sec}/{f}\t{fl}\tscore={sc}\t{sz}\t{snip[:80]}")
|
||
print(f"\nTOTAL flagged: {len(results)}", file=sys.stderr)
|
||
|
||
if __name__ == '__main__':
|
||
main()
|