content audit sec01-06: fixes (audit_content.py + 12 file/card corrections)
- tools/audit_content.py: per-file vs item-title triage (SUSPECT/SMALL/NO-TEXT/NO-CYR) - sec01: .part removed; 10 files restored .pdf ext (LFS rename); #04/#19/#27 verify notes - sec02 #16: corrupted double-encoded txt -> clean flib fb2 b/844509; #17/#22 doc verified via catdoc - sec03 #11: -b epub = Spanish von Franz (Paidós 1983, forged EN OPF) -> bonus label - sec03 #42: КАРО 2012 'Irish Tales' = EN reader (not RU, not the listed book) -> bonus rename - sec03 #56: Onians b/c/d/e = Cambridge 24-87KB previews -> labeled - sec06 #18: filename 1981->1989 (Perera 'Descent to the Goddess' 1989; 1981 = other book) - sec06 #28: ia Zimmer OCR layer = foreign Devanagari text; images = RKP (p.100 verified) - sec04 #30: buksmart 2020 = RU/FR bilingual w/ VeryPDF watermarks -> note - linter: azw3/mobi ext + underscore in fname regex; manifest cross-check ignores off-disk rows; 8 legacy cards section-order fixed (CR before DL) - AGENTS.md: 'Content audit (2026-09-25)' lessons section - check-md: 0/219
This commit is contained in:
parent
85c41acd3c
commit
f5cb34195e
44 changed files with 20976 additions and 83 deletions
103
tools/audit_content.py
Normal file
103
tools/audit_content.py
Normal file
|
|
@ -0,0 +1,103 @@
|
|||
#!/usr/bin/env python3
|
||||
"""audit_content.py — content-triage of ALL downloaded files vs their item title.
|
||||
|
||||
For each file NN-... in downloads/0N-*/: expected title = per-item header
|
||||
'### NN — <title>' of that section's MANIFEST. Extract a text sample
|
||||
(pdf pdftotext -l 3 / fb2 body / epub first xhtml / djvu djvutxt / doc strings)
|
||||
and score title-token overlap. Flags: SUSPECT (0 overlap), SMALL, NO-TEXT.
|
||||
RU files (name contains '-ru') get a light check (Cyrillic present, non-empty).
|
||||
Run: python3 tools/audit_content.py [sec] (sec = 01..06, default all)
|
||||
"""
|
||||
import os, re, sys, subprocess, json, html
|
||||
|
||||
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
DL = os.path.join(BASE, 'downloads')
|
||||
STOP = {'the','and','for','with','from','into','upon','about','its','his','her','vol','vols','chaps','ch','part','study','essays','essay','selected','works','collected','complete','book','pages','original','edition','second','first'}
|
||||
|
||||
def tokens(title):
|
||||
t = re.sub(r'\(.*?\)', ' ', title)
|
||||
t = re.sub(r'[^a-zа-яё0-9 ]', '', t.lower())
|
||||
return [w for w in t.split() if len(w) >= 4 and w not in STOP]
|
||||
|
||||
def item_titles(sec_dir):
|
||||
mf = os.path.join(DL, sec_dir, 'MANIFEST.md')
|
||||
t = open(mf, encoding='utf8').read()
|
||||
part = t.split('## Not downloadable')[0]
|
||||
out = {}
|
||||
for m in re.finditer(r'^### (\d+) — (.+?) — (.+)$', part, re.M):
|
||||
out[m.group(1)] = m.group(2).strip()
|
||||
# items with '0 files' headers too (no local files, skip)
|
||||
return out
|
||||
|
||||
def sample(path):
|
||||
ext = path.rsplit('.', 1)[-1].lower()
|
||||
try:
|
||||
if ext == 'pdf':
|
||||
r = subprocess.run(['pdftotext', '-l', '3', path, '-'], capture_output=True, timeout=120)
|
||||
txt = r.stdout.decode('utf8', errors='ignore')
|
||||
if not txt.strip():
|
||||
r = subprocess.run(['pdftotext', path, '-'], capture_output=True, timeout=300)
|
||||
txt = r.stdout.decode('utf8', errors='ignore')[:3000]
|
||||
return txt[:3000]
|
||||
if ext == 'fb2':
|
||||
raw = open(path, encoding='utf8', errors='ignore').read()
|
||||
body = re.search(r'<body>(.*)', raw, re.S)
|
||||
txt = re.sub(r'<[^>]+>', ' ', body.group(1) if body else raw[:50000])
|
||||
return re.sub(r'\s+', ' ', txt)[:3000]
|
||||
if ext == 'epub':
|
||||
import zipfile
|
||||
z = zipfile.ZipFile(path)
|
||||
names = [n for n in z.namelist() if re.search(r'\.(x?html?|htm)$', n) and 'nav' not in n.lower()]
|
||||
txt = ''
|
||||
for n in names[:8]:
|
||||
txt += z.read(n).decode('utf8', errors='ignore')
|
||||
if len(txt) > 4000: break
|
||||
return re.sub(r'<[^>]+>', ' ', txt)[:3000]
|
||||
if ext == 'djvu':
|
||||
r = subprocess.run(['djvutxt', path, '-'], capture_output=True, timeout=300)
|
||||
txt = r.stdout.decode('utf8', errors='ignore')
|
||||
return txt[:3000]
|
||||
if ext in ('doc', 'txt', 'html', 'mobi'):
|
||||
r = subprocess.run(['strings', '-n', '6', path], capture_output=True, timeout=60)
|
||||
return r.stdout.decode('utf8', errors='ignore')[:3000]
|
||||
except Exception as e:
|
||||
return f"__ERR__ {e}"
|
||||
return ''
|
||||
|
||||
def main():
|
||||
secs = sys.argv[1:] or sorted(d for d in os.listdir(DL) if re.match(r'^0\d-', d))
|
||||
results = []
|
||||
for sec in secs:
|
||||
titles = item_titles(sec)
|
||||
files = [f for f in os.listdir(os.path.join(DL, sec)) if f != 'MANIFEST.md' and re.match(r'^\d\d-', f)]
|
||||
for f in sorted(files):
|
||||
num = f[:2]
|
||||
title = titles.get(num, '')
|
||||
if not title:
|
||||
results.append((sec, f, 'NO-TITLE', 0, 0, ''))
|
||||
continue
|
||||
tok = tokens(title)
|
||||
is_ru = '-ru' in f or re.search(r'-(t\d-ru|ru-)', f)
|
||||
size = os.path.getsize(os.path.join(DL, sec, f))
|
||||
s = sample(os.path.join(DL, sec, f))
|
||||
low = s.lower()
|
||||
hits = sum(1 for w in tok if w in low) if tok else 0
|
||||
score = hits / len(tok) if tok else 0
|
||||
flags = []
|
||||
if s.startswith('__ERR__'):
|
||||
flags.append('ERR')
|
||||
elif is_ru:
|
||||
if not re.search(r'[а-яё]', s): flags.append('NO-CYR')
|
||||
else:
|
||||
if not s.strip(): flags.append('NO-TEXT')
|
||||
elif score == 0 and tok: flags.append('SUSPECT')
|
||||
if size < 200000 and f.endswith('.pdf'): flags.append('SMALL')
|
||||
if size < 80000 and not f.endswith('.pdf'): flags.append('SMALL')
|
||||
if flags:
|
||||
results.append((sec, f, ','.join(flags), round(score, 2), size, s[:150].replace('\n', ' ')))
|
||||
for sec, f, fl, sc, sz, snip in results:
|
||||
print(f"{sec}/{f}\t{fl}\tscore={sc}\t{sz}\t{snip[:80]}")
|
||||
print(f"\nTOTAL flagged: {len(results)}", file=sys.stderr)
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
|
|
@ -63,9 +63,9 @@ def lint_card(path, manifest_files):
|
|||
# Downloads table cross-check
|
||||
dm = re.search(r'^## Downloads\n(.*?)(?=^## |\Z)', c, re.M | re.S)
|
||||
if dm:
|
||||
rows = re.findall(r'^\|\s*(EN|RU\??|\?)\s*\|\s*([0-9]{2}-[a-zA-Z0-9\.\-]+)\s*\|', dm.group(1), re.M)
|
||||
rows = re.findall(r'^\|\s*(EN|RU\??|\?)\s*\|\s*([0-9]{2}-[a-zA-Z0-9_\.\-]+)\s*\|', dm.group(1), re.M)
|
||||
disk = set(os.path.basename(f) for f in glob.glob(f"downloads/{sec_of(path)}/*"))
|
||||
manifest = set(f for f, _, _ in manifest_files.get(num, []))
|
||||
manifest = set(f for f, _, _ in manifest_files.get(num, []) if f in disk) # skip legacy/renamed rows
|
||||
for lang, fname in rows:
|
||||
if fname not in disk:
|
||||
issues.append(f'Downloads file not on disk: {fname}')
|
||||
|
|
|
|||
|
|
@ -19,7 +19,7 @@ def parse_manifest_files(manifest_path):
|
|||
if not os.path.exists(manifest_path):
|
||||
return items
|
||||
c = open(manifest_path, encoding='utf-8').read()
|
||||
fname_re = r'([0-9]{2}-[a-zA-Z0-9\-]+(?:\.[a-zA-Z0-9]+)*\.(?:pdf|fb2|epub|doc|docx|chm|djvu|txt|zip))'
|
||||
fname_re = r'([0-9]{2}-[a-zA-Z0-9_\-]+(?:\.[a-zA-Z0-9]+)*\.(?:pdf|fb2|epub|doc|docx|chm|djvu|txt|zip|mobi|azw3))'
|
||||
# sec01/02: table rows | file | size | source | note |
|
||||
for m in re.finditer(r'^\|\s*' + fname_re + r'\s*\|\s*([^|]+)\|\s*([^|]+)\|', c, re.M):
|
||||
f, size, src = m.group(1), m.group(2).strip(), m.group(3).strip()
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue