- tools/rutitles.py: EN->RU token translation (data/ru-dict.tsv, 540+ pairs, Kogito/Castalia
conventions) + Jaccard diff against data/ru-titles.jsonl (692 titles: Kogito 518, Litres 24,
OPP 55, flibusta a/5272+57639+118921+122883+193690+193689). Matching is lemmatized
(pymorphy3) — case endings handled: 'великой матери' -> 'Великая мать' 1.25.
'The Great Mother' -> 'Великая мать' 1.25 top hit; 'The Symbolic Quest' -> correct negative.
- tools/sweep.py: batch driver for Phases 1-2 (rsl+lg+alib+flib+cogito+SearXNG-OZON-snippet
parse+rutitles diff per item; --nlr optional). Query log data/queries.log (JSONL).
Fixed: cogito nav-menu leak (parse bx_product_item only), libgen robot-block (lg.py curl fallback).
- tools/oa.py: OpenAlex + Crossref (Phase 0 identity/ISBN, no key; found Margaret Wilkinson,
Karen Evers-Fahey, Symbolic Quest Princeton ISBN).
- data/sweeps/01-fundamentals/input.tsv: 16 ❌ items loaded; background sweep running.
- AGENTS.md: source 10b (oa.py), flibusta .su = reduced mirror (dropped from pipeline),
pipeline v3 section (rutitles/oa/sweep/pymorphy3 note: pymorphy2 broken on py3.12).
- data/ru-titles.jsonl committed as the RU market universe asset.
205 lines
8.5 KiB
Python
Executable file
205 lines
8.5 KiB
Python
Executable file
#!/usr/bin/env python3
|
||
"""RU-title candidates: EN->RU token translation + match against known RU title DB.
|
||
|
||
Usage:
|
||
rutitles.py build # rebuild data/ru-titles.jsonl from data/catalogs/*
|
||
rutitles.py translate "The Symbolic Quest" # EN title -> RU token bag (+unknown EN words)
|
||
rutitles.py match "символический поиск" [N] # top-N fuzzy matches in the RU title DB
|
||
rutitles.py both "The Symbolic Quest" # translate + match in one go
|
||
|
||
DB sources (data/catalogs/*.md, title lines after the header):
|
||
cogito Jungian series (518), litres 140409, OPP/MISP list.
|
||
Extend: `rutitles.py addflib <authorId>` appends flibusta.is authorall book titles.
|
||
|
||
Matching: token Jaccard over normalized lowercase tokens (stopwords removed).
|
||
Order-agnostic — RU word order irrelevant. This is a DIFF against the known
|
||
RU market universe, NOT a translation check — verify identity on real hits.
|
||
"""
|
||
import json, os, re, sys, difflib
|
||
|
||
try:
|
||
import pymorphy3
|
||
_MORPH = pymorphy3.MorphAnalyzer()
|
||
except Exception:
|
||
_MORPH = None
|
||
|
||
def lemma(w):
|
||
if _MORPH and w.isalpha() and "а" <= w[0] <= "я":
|
||
try:
|
||
return _MORPH.parse(w)[0].normal_form
|
||
except Exception:
|
||
return w
|
||
return w
|
||
|
||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||
DICT = os.path.join(ROOT, "data", "ru-dict.tsv")
|
||
DB = os.path.join(ROOT, "data", "ru-titles.jsonl")
|
||
CATS = os.path.join(ROOT, "data", "catalogs")
|
||
|
||
STOP = set("""в вв во не на наа наа не при по от из до к к к к у за для с со и и и или но а же ли бы то что это тот эта этот эта эти те они он она мы вы они ее его их наш твой мой ваш их их мой твой его ее их наш ваш свой чей какой какой какие какое какое каких каких какие какие какой какой какой какой какая какое какие сколько сколько многие несколько весь все всё весь все всё весь все всё каждый каждый каждая каждое какие любой любая любое любые любой любая любое любые некий некая некое некие иной иная иное иные иной иная иное иные чужой чужая чужое чужие другой другая другое другие другой другая другое другие сам сама само сами сами сами сам сама само сами сам сам сама само сами""".split())
|
||
STOP = set(w for w in STOP if len(w) > 1)
|
||
STOP |= {"в", "на", "во", "о", "об", "с", "со", "из", "от", "по", "к", "у", "за", "и", "а", "но",
|
||
"не", "что", "который", "которая", "которое", "которые", "для", "про", "же", "ли"}
|
||
|
||
def load_dict():
|
||
phrases, words = [], {}
|
||
for line in open(DICT, encoding="utf-8"):
|
||
line = line.rstrip("\n")
|
||
if not line or line.startswith("#") or "\t" not in line:
|
||
continue
|
||
k, v = line.split("\t", 1)
|
||
k, v = k.strip(), v.strip()
|
||
if " " in k:
|
||
phrases.append((k.lower(), v))
|
||
else:
|
||
words[k.lower()] = v
|
||
phrases.sort(key=lambda kv: -len(kv[0]))
|
||
return phrases, words
|
||
|
||
def norm_tokens(s):
|
||
s = s.lower()
|
||
s = re.sub(r"[«»\"'’\-–—().,;:!?/\\|]", " ", s)
|
||
toks = [t for t in s.split() if t]
|
||
return [lemma(t) for t in toks if t not in STOP]
|
||
|
||
PHR, WORD = load_dict()
|
||
|
||
def translate(title):
|
||
"""EN title -> (ru_token_bag, [unknown_en_words])"""
|
||
text = title.lower().strip()
|
||
text = re.sub(r"[«»\"'’\-–—().,;:!?/\\|]", " ", text)
|
||
tokens, unknown = [], []
|
||
# phrases first (longest)
|
||
for ph, ru in PHR:
|
||
pat = re.compile(r"(?<![a-z])" + re.escape(ph) + r"(?![a-z])")
|
||
m = pat.search(text)
|
||
if m:
|
||
tokens.append(ru)
|
||
text = text[:m.start()] + " " * (m.end() - m.start()) + text[m.end():]
|
||
for w in text.split():
|
||
w = w.strip()
|
||
if not w or w in STOP or not w.isalpha():
|
||
continue
|
||
if w in WORD:
|
||
tokens.append(WORD[w])
|
||
elif re.fullmatch(r"[a-z]+", w):
|
||
unknown.append(w)
|
||
# expand variants for matching: keep all
|
||
return tokens, unknown
|
||
|
||
def load_db():
|
||
if not os.path.exists(DB):
|
||
return []
|
||
return [json.loads(l) for l in open(DB, encoding="utf-8") if l.strip()]
|
||
|
||
def build():
|
||
rows, seen = [], set()
|
||
# preserve non-catalog entries (flibusta author lists added via addflib)
|
||
if os.path.exists(DB):
|
||
for l in open(DB, encoding="utf-8"):
|
||
if l.strip():
|
||
r = json.loads(l)
|
||
if r["src"].startswith("flib:"):
|
||
rows.append(r)
|
||
seen.add(r["norm"])
|
||
for fn in sorted(os.listdir(CATS)):
|
||
if not fn.endswith((".md", ".txt")):
|
||
continue
|
||
src = os.path.basename(fn)
|
||
for line in open(os.path.join(CATS, fn), encoding="utf-8", errors="replace"):
|
||
line = line.strip()
|
||
if not line or line.startswith(("#", "-", "Источник", "Дата", "##")):
|
||
continue
|
||
if len(line) < 8:
|
||
continue
|
||
key = " ".join(norm_tokens(line))
|
||
if key in seen:
|
||
continue
|
||
seen.add(key)
|
||
rows.append({"title": line, "norm": key, "src": src})
|
||
with open(DB, "w", encoding="utf-8") as f:
|
||
for r in rows:
|
||
f.write(json.dumps(r, ensure_ascii=False) + "\n")
|
||
print("DB built: %d titles from %d catalogs" % (len(rows), len(os.listdir(CATS))))
|
||
|
||
def match(needle, n=10):
|
||
nt = set(norm_tokens(needle))
|
||
if not nt:
|
||
print("empty needle"); return
|
||
rows = load_db()
|
||
scored = []
|
||
for r in rows:
|
||
rt = set(r["norm"].split())
|
||
if not rt:
|
||
continue
|
||
inter = nt & rt
|
||
union = nt | rt
|
||
j = len(inter) / len(union) if union else 0
|
||
# boost: all needle tokens present
|
||
if nt <= rt:
|
||
j += 0.25
|
||
if j >= 0.25:
|
||
scored.append((j, r))
|
||
scored.sort(key=lambda x: -x[0])
|
||
for j, r in scored[:n]:
|
||
print("%.2f [%s] %s" % (j, r["src"][:12], r["title"][:100]))
|
||
|
||
def both(title, n=10):
|
||
tokens, unknown = translate(title)
|
||
print("RU tokens:", " | ".join(tokens))
|
||
if unknown:
|
||
print("UNTRANSLATED:", " ".join(unknown))
|
||
if tokens:
|
||
print("--- matches ---")
|
||
match(" ".join(tokens), n)
|
||
|
||
def addflib(author_id):
|
||
import subprocess
|
||
out = subprocess.run([sys.executable, os.path.join(ROOT, "tools", "flib.py"), "authorall", str(author_id)],
|
||
capture_output=True, text=True, timeout=600).stdout
|
||
rows = [json.loads(l) for l in open(DB, encoding="utf-8") if l.strip()] if os.path.exists(DB) else []
|
||
seen = set(r["norm"] for r in rows)
|
||
added = 0
|
||
for line in out.splitlines():
|
||
# flib.py authorall: "b/<id> <year> <fmt> <author> — <title>[ [пер. X]]"
|
||
m = re.match(r"^b/(\d+)\s+(\d+|\?)\s+\S+\s+([^—]+?)\s*—\s*(.+)$", line)
|
||
if not m:
|
||
continue
|
||
title = re.sub(r"\s*\[пер\. .+\]$", "", m.group(4)).strip()
|
||
key = " ".join(norm_tokens(title))
|
||
if not key or key in seen:
|
||
continue
|
||
seen.add(key)
|
||
rows.append({"title": title, "norm": key, "src": "flib:a%s" % author_id})
|
||
added += 1
|
||
with open(DB, "w", encoding="utf-8") as f:
|
||
for r in rows:
|
||
f.write(json.dumps(r, ensure_ascii=False) + "\n")
|
||
print("flib a/%s: +%d titles (DB now %d)" % (author_id, added, len(rows)))
|
||
|
||
def main():
|
||
if len(sys.argv) < 2:
|
||
print(__doc__); sys.exit(1)
|
||
cmd = sys.argv[1]
|
||
if cmd == "build":
|
||
build()
|
||
elif cmd == "translate":
|
||
tokens, unknown = translate(" ".join(sys.argv[2:]))
|
||
print(" | ".join(tokens))
|
||
if unknown:
|
||
print("UNTRANSLATED:", " ".join(unknown))
|
||
elif cmd == "match":
|
||
args = sys.argv[2:]
|
||
n = 10
|
||
if args and args[-1].isdigit():
|
||
n, args = int(args[-1]), args[:-1]
|
||
match(" ".join(args), n)
|
||
elif cmd == "both":
|
||
both(" ".join(sys.argv[2:]), int(sys.argv[3]) if len(sys.argv) > 3 else 10)
|
||
elif cmd == "addflib":
|
||
addflib(sys.argv[2])
|
||
else:
|
||
print(__doc__); sys.exit(1)
|
||
|
||
if __name__ == "__main__":
|
||
main()
|