pipeline v3: RU-title candidates (rutitles+ru-dict+pymorphy3), batch sweep driver, OpenAlex/Crossref
- tools/rutitles.py: EN->RU token translation (data/ru-dict.tsv, 540+ pairs, Kogito/Castalia
conventions) + Jaccard diff against data/ru-titles.jsonl (692 titles: Kogito 518, Litres 24,
OPP 55, flibusta a/5272+57639+118921+122883+193690+193689). Matching is lemmatized
(pymorphy3) — case endings handled: 'великой матери' -> 'Великая мать' 1.25.
'The Great Mother' -> 'Великая мать' 1.25 top hit; 'The Symbolic Quest' -> correct negative.
- tools/sweep.py: batch driver for Phases 1-2 (rsl+lg+alib+flib+cogito+SearXNG-OZON-snippet
parse+rutitles diff per item; --nlr optional). Query log data/queries.log (JSONL).
Fixed: cogito nav-menu leak (parse bx_product_item only), libgen robot-block (lg.py curl fallback).
- tools/oa.py: OpenAlex + Crossref (Phase 0 identity/ISBN, no key; found Margaret Wilkinson,
Karen Evers-Fahey, Symbolic Quest Princeton ISBN).
- data/sweeps/01-fundamentals/input.tsv: 16 ❌ items loaded; background sweep running.
- AGENTS.md: source 10b (oa.py), flibusta .su = reduced mirror (dropped from pipeline),
pipeline v3 section (rutitles/oa/sweep/pymorphy3 note: pymorphy2 broken on py3.12).
- data/ru-titles.jsonl committed as the RU market universe asset.
This commit is contained in:
parent
2157d058df
commit
5441dddbb8
11 changed files with 2101 additions and 3 deletions
16
tools/lg.py
16
tools/lg.py
|
|
@ -10,10 +10,24 @@ def fetch(url, tries=3):
|
|||
for i in range(tries):
|
||||
try:
|
||||
req = urllib.request.Request(url, headers={"User-Agent": UA, "Accept-Language": "ru-RU,ru;q=0.9"})
|
||||
return urllib.request.urlopen(req, timeout=90).read().decode("utf-8", "replace")
|
||||
html = urllib.request.urlopen(req, timeout=90).read().decode("utf-8", "replace")
|
||||
# robot-block detection: nginx default page has no req= result table
|
||||
if 'libgen' not in html.lower()[:3000] and 'Search Result' not in html and 'result' not in html.lower()[:3000]:
|
||||
time.sleep(15)
|
||||
continue
|
||||
return html
|
||||
except Exception as e:
|
||||
last = e
|
||||
time.sleep(5 + 5 * i)
|
||||
# curl fallback (works when urllib is fingerprinted/blocked)
|
||||
import subprocess
|
||||
try:
|
||||
r = subprocess.run(["curl", "-s", "-A", UA, "--max-time", "90", url],
|
||||
capture_output=True, text=True, timeout=120)
|
||||
if r.stdout:
|
||||
return r.stdout
|
||||
except Exception:
|
||||
pass
|
||||
raise SystemExit(f"fetch failed after {tries} tries: {last}")
|
||||
|
||||
def main():
|
||||
|
|
|
|||
98
tools/oa.py
Executable file
98
tools/oa.py
Executable file
|
|
@ -0,0 +1,98 @@
|
|||
#!/usr/bin/env python3
|
||||
"""OpenAlex + Crossref — Phase 0 identity/ISBN resolution (free, no key).
|
||||
|
||||
Usage:
|
||||
oa.py "Exact EN Title" # OpenAlex work search (names, year, DOI, ISBN via locations)
|
||||
oa.py --crossref "Title words" # Crossref bibliographic search (author, ISBN, publisher)
|
||||
oa.py --author "Last, First" # OpenAlex author entity (ID, works count, cited)
|
||||
oa.py both "Title" # both APIs in one go
|
||||
|
||||
OpenAlex: https://api.openalex.org/works?search=... (no key, 10 req/s pool)
|
||||
Crossref: https://api.crossref.org/works?query.bibliographic=... (polite pool ok)
|
||||
|
||||
Notes:
|
||||
- OpenAlex 'display_name' + authorships[].author.display_name = full EN names
|
||||
(better than OL for modern academic books; OL still primary for pre-1990).
|
||||
- Crossref ISBN list = direct EN ISBN candidates (13-digit, Routledge/Springer/Karnac).
|
||||
- DNS here is flaky for python sockets in some contexts; use urllib (works).
|
||||
"""
|
||||
import json, sys, time, urllib.parse, urllib.request
|
||||
|
||||
UA = "jung-ru-editions-research/1.0 (mailto:dmitry@kokorin.org)"
|
||||
|
||||
def get(url):
|
||||
req = urllib.request.Request(url, headers={"User-Agent": UA})
|
||||
with urllib.request.urlopen(req, timeout=45) as r:
|
||||
return json.loads(r.read().decode("utf-8", "replace"))
|
||||
|
||||
def openalex_works(q, n=5):
|
||||
url = "https://api.openalex.org/works?search=%s&per-page=%d" % (urllib.parse.quote(q), n)
|
||||
d = get(url)
|
||||
out = []
|
||||
for w in d.get("results", []):
|
||||
auths = [a["author"]["display_name"] for a in w.get("authorships", [])]
|
||||
isbns = set()
|
||||
for l in w.get("locations", []):
|
||||
if l.get("pdf") or l.get("landing_page_url"):
|
||||
pass
|
||||
# ISBNs often in biblio
|
||||
bib = w.get("biblio") or {}
|
||||
ids = w.get("ids", {})
|
||||
doi = ids.get("doi")
|
||||
out.append({
|
||||
"title": w.get("display_name"),
|
||||
"authors": auths,
|
||||
"year": w.get("publication_year"),
|
||||
"doi": doi,
|
||||
"publisher": (w.get("primary_location") or {}).get("source", {}).get("display_name") if w.get("primary_location") else None,
|
||||
"openalex_id": w.get("id"),
|
||||
})
|
||||
return out
|
||||
|
||||
def crossref(q, n=3):
|
||||
url = "https://api.crossref.org/works?query.bibliographic=%s&rows=%d" % (urllib.parse.quote(q), n)
|
||||
d = get(url)
|
||||
out = []
|
||||
for m in d.get("message", {}).get("items", []):
|
||||
auths = [("%s %s" % (a.get("given", ""), a.get("family", ""))).strip() for a in m.get("author", [])]
|
||||
out.append({
|
||||
"title": (m.get("title") or ["?"])[0],
|
||||
"authors": auths,
|
||||
"isbn": m.get("ISBN") or [],
|
||||
"issn": m.get("ISSN") or [],
|
||||
"publisher": m.get("publisher"),
|
||||
"year": (m.get("issued", {}).get("date-parts") or [[None]])[0][0],
|
||||
"doi": m.get("DOI"),
|
||||
})
|
||||
return out
|
||||
|
||||
def openalex_author(q):
|
||||
url = "https://api.openalex.org/authors?search=%s&per-page=5" % urllib.parse.quote(q)
|
||||
d = get(url)
|
||||
out = []
|
||||
for a in d.get("results", []):
|
||||
out.append({
|
||||
"name": a.get("display_name"),
|
||||
"id": a.get("id"),
|
||||
"works": a.get("works_count"),
|
||||
"cited": a.get("cited_by_count"),
|
||||
})
|
||||
return out
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 3:
|
||||
print(__doc__); sys.exit(1)
|
||||
mode, q = sys.argv[1], " ".join(sys.argv[2:])
|
||||
if mode == "both":
|
||||
for r in openalex_works(q): print("OA ", r)
|
||||
time.sleep(1)
|
||||
for r in crossref(q): print("CR ", r)
|
||||
elif mode == "--crossref":
|
||||
for r in crossref(q): print(r)
|
||||
elif mode == "--author":
|
||||
for r in openalex_author(q): print(r)
|
||||
else:
|
||||
for r in openalex_works(q): print(r)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
205
tools/rutitles.py
Executable file
205
tools/rutitles.py
Executable file
|
|
@ -0,0 +1,205 @@
|
|||
#!/usr/bin/env python3
|
||||
"""RU-title candidates: EN->RU token translation + match against known RU title DB.
|
||||
|
||||
Usage:
|
||||
rutitles.py build # rebuild data/ru-titles.jsonl from data/catalogs/*
|
||||
rutitles.py translate "The Symbolic Quest" # EN title -> RU token bag (+unknown EN words)
|
||||
rutitles.py match "символический поиск" [N] # top-N fuzzy matches in the RU title DB
|
||||
rutitles.py both "The Symbolic Quest" # translate + match in one go
|
||||
|
||||
DB sources (data/catalogs/*.md, title lines after the header):
|
||||
cogito Jungian series (518), litres 140409, OPP/MISP list.
|
||||
Extend: `rutitles.py addflib <authorId>` appends flibusta.is authorall book titles.
|
||||
|
||||
Matching: token Jaccard over normalized lowercase tokens (stopwords removed).
|
||||
Order-agnostic — RU word order irrelevant. This is a DIFF against the known
|
||||
RU market universe, NOT a translation check — verify identity on real hits.
|
||||
"""
|
||||
import json, os, re, sys, difflib
|
||||
|
||||
try:
|
||||
import pymorphy3
|
||||
_MORPH = pymorphy3.MorphAnalyzer()
|
||||
except Exception:
|
||||
_MORPH = None
|
||||
|
||||
def lemma(w):
|
||||
if _MORPH and w.isalpha() and "а" <= w[0] <= "я":
|
||||
try:
|
||||
return _MORPH.parse(w)[0].normal_form
|
||||
except Exception:
|
||||
return w
|
||||
return w
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
DICT = os.path.join(ROOT, "data", "ru-dict.tsv")
|
||||
DB = os.path.join(ROOT, "data", "ru-titles.jsonl")
|
||||
CATS = os.path.join(ROOT, "data", "catalogs")
|
||||
|
||||
STOP = set("""в вв во не на наа наа не при по от из до к к к к у за для с со и и и или но а же ли бы то что это тот эта этот эта эти те они он она мы вы они ее его их наш твой мой ваш их их мой твой его ее их наш ваш свой чей какой какой какие какое какое каких каких какие какие какой какой какой какой какая какое какие сколько сколько многие несколько весь все всё весь все всё весь все всё каждый каждый каждая каждое какие любой любая любое любые любой любая любое любые некий некая некое некие иной иная иное иные иной иная иное иные чужой чужая чужое чужие другой другая другое другие другой другая другое другие сам сама само сами сами сами сам сама само сами сам сам сама само сами""".split())
|
||||
STOP = set(w for w in STOP if len(w) > 1)
|
||||
STOP |= {"в", "на", "во", "о", "об", "с", "со", "из", "от", "по", "к", "у", "за", "и", "а", "но",
|
||||
"не", "что", "который", "которая", "которое", "которые", "для", "про", "же", "ли"}
|
||||
|
||||
def load_dict():
|
||||
phrases, words = [], {}
|
||||
for line in open(DICT, encoding="utf-8"):
|
||||
line = line.rstrip("\n")
|
||||
if not line or line.startswith("#") or "\t" not in line:
|
||||
continue
|
||||
k, v = line.split("\t", 1)
|
||||
k, v = k.strip(), v.strip()
|
||||
if " " in k:
|
||||
phrases.append((k.lower(), v))
|
||||
else:
|
||||
words[k.lower()] = v
|
||||
phrases.sort(key=lambda kv: -len(kv[0]))
|
||||
return phrases, words
|
||||
|
||||
def norm_tokens(s):
|
||||
s = s.lower()
|
||||
s = re.sub(r"[«»\"'’\-–—().,;:!?/\\|]", " ", s)
|
||||
toks = [t for t in s.split() if t]
|
||||
return [lemma(t) for t in toks if t not in STOP]
|
||||
|
||||
PHR, WORD = load_dict()
|
||||
|
||||
def translate(title):
|
||||
"""EN title -> (ru_token_bag, [unknown_en_words])"""
|
||||
text = title.lower().strip()
|
||||
text = re.sub(r"[«»\"'’\-–—().,;:!?/\\|]", " ", text)
|
||||
tokens, unknown = [], []
|
||||
# phrases first (longest)
|
||||
for ph, ru in PHR:
|
||||
pat = re.compile(r"(?<![a-z])" + re.escape(ph) + r"(?![a-z])")
|
||||
m = pat.search(text)
|
||||
if m:
|
||||
tokens.append(ru)
|
||||
text = text[:m.start()] + " " * (m.end() - m.start()) + text[m.end():]
|
||||
for w in text.split():
|
||||
w = w.strip()
|
||||
if not w or w in STOP or not w.isalpha():
|
||||
continue
|
||||
if w in WORD:
|
||||
tokens.append(WORD[w])
|
||||
elif re.fullmatch(r"[a-z]+", w):
|
||||
unknown.append(w)
|
||||
# expand variants for matching: keep all
|
||||
return tokens, unknown
|
||||
|
||||
def load_db():
|
||||
if not os.path.exists(DB):
|
||||
return []
|
||||
return [json.loads(l) for l in open(DB, encoding="utf-8") if l.strip()]
|
||||
|
||||
def build():
|
||||
rows, seen = [], set()
|
||||
# preserve non-catalog entries (flibusta author lists added via addflib)
|
||||
if os.path.exists(DB):
|
||||
for l in open(DB, encoding="utf-8"):
|
||||
if l.strip():
|
||||
r = json.loads(l)
|
||||
if r["src"].startswith("flib:"):
|
||||
rows.append(r)
|
||||
seen.add(r["norm"])
|
||||
for fn in sorted(os.listdir(CATS)):
|
||||
if not fn.endswith((".md", ".txt")):
|
||||
continue
|
||||
src = os.path.basename(fn)
|
||||
for line in open(os.path.join(CATS, fn), encoding="utf-8", errors="replace"):
|
||||
line = line.strip()
|
||||
if not line or line.startswith(("#", "-", "Источник", "Дата", "##")):
|
||||
continue
|
||||
if len(line) < 8:
|
||||
continue
|
||||
key = " ".join(norm_tokens(line))
|
||||
if key in seen:
|
||||
continue
|
||||
seen.add(key)
|
||||
rows.append({"title": line, "norm": key, "src": src})
|
||||
with open(DB, "w", encoding="utf-8") as f:
|
||||
for r in rows:
|
||||
f.write(json.dumps(r, ensure_ascii=False) + "\n")
|
||||
print("DB built: %d titles from %d catalogs" % (len(rows), len(os.listdir(CATS))))
|
||||
|
||||
def match(needle, n=10):
|
||||
nt = set(norm_tokens(needle))
|
||||
if not nt:
|
||||
print("empty needle"); return
|
||||
rows = load_db()
|
||||
scored = []
|
||||
for r in rows:
|
||||
rt = set(r["norm"].split())
|
||||
if not rt:
|
||||
continue
|
||||
inter = nt & rt
|
||||
union = nt | rt
|
||||
j = len(inter) / len(union) if union else 0
|
||||
# boost: all needle tokens present
|
||||
if nt <= rt:
|
||||
j += 0.25
|
||||
if j >= 0.25:
|
||||
scored.append((j, r))
|
||||
scored.sort(key=lambda x: -x[0])
|
||||
for j, r in scored[:n]:
|
||||
print("%.2f [%s] %s" % (j, r["src"][:12], r["title"][:100]))
|
||||
|
||||
def both(title, n=10):
|
||||
tokens, unknown = translate(title)
|
||||
print("RU tokens:", " | ".join(tokens))
|
||||
if unknown:
|
||||
print("UNTRANSLATED:", " ".join(unknown))
|
||||
if tokens:
|
||||
print("--- matches ---")
|
||||
match(" ".join(tokens), n)
|
||||
|
||||
def addflib(author_id):
|
||||
import subprocess
|
||||
out = subprocess.run([sys.executable, os.path.join(ROOT, "tools", "flib.py"), "authorall", str(author_id)],
|
||||
capture_output=True, text=True, timeout=600).stdout
|
||||
rows = [json.loads(l) for l in open(DB, encoding="utf-8") if l.strip()] if os.path.exists(DB) else []
|
||||
seen = set(r["norm"] for r in rows)
|
||||
added = 0
|
||||
for line in out.splitlines():
|
||||
# flib.py authorall: "b/<id> <year> <fmt> <author> — <title>[ [пер. X]]"
|
||||
m = re.match(r"^b/(\d+)\s+(\d+|\?)\s+\S+\s+([^—]+?)\s*—\s*(.+)$", line)
|
||||
if not m:
|
||||
continue
|
||||
title = re.sub(r"\s*\[пер\. .+\]$", "", m.group(4)).strip()
|
||||
key = " ".join(norm_tokens(title))
|
||||
if not key or key in seen:
|
||||
continue
|
||||
seen.add(key)
|
||||
rows.append({"title": title, "norm": key, "src": "flib:a%s" % author_id})
|
||||
added += 1
|
||||
with open(DB, "w", encoding="utf-8") as f:
|
||||
for r in rows:
|
||||
f.write(json.dumps(r, ensure_ascii=False) + "\n")
|
||||
print("flib a/%s: +%d titles (DB now %d)" % (author_id, added, len(rows)))
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 2:
|
||||
print(__doc__); sys.exit(1)
|
||||
cmd = sys.argv[1]
|
||||
if cmd == "build":
|
||||
build()
|
||||
elif cmd == "translate":
|
||||
tokens, unknown = translate(" ".join(sys.argv[2:]))
|
||||
print(" | ".join(tokens))
|
||||
if unknown:
|
||||
print("UNTRANSLATED:", " ".join(unknown))
|
||||
elif cmd == "match":
|
||||
args = sys.argv[2:]
|
||||
n = 10
|
||||
if args and args[-1].isdigit():
|
||||
n, args = int(args[-1]), args[:-1]
|
||||
match(" ".join(args), n)
|
||||
elif cmd == "both":
|
||||
both(" ".join(sys.argv[2:]), int(sys.argv[3]) if len(sys.argv) > 3 else 10)
|
||||
elif cmd == "addflib":
|
||||
addflib(sys.argv[2])
|
||||
else:
|
||||
print(__doc__); sys.exit(1)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
205
tools/sweep.py
Normal file
205
tools/sweep.py
Normal file
|
|
@ -0,0 +1,205 @@
|
|||
#!/usr/bin/env python3
|
||||
"""sweep.py — batch driver for pipeline Phases 1-2 (mechanical part).
|
||||
|
||||
Input: data/sweeps/<sec>/input.tsv with columns (tab-separated):
|
||||
nn slug en_title ru_surnames('|'-separated variants) ru_title_candidates('|'-separated)
|
||||
(e.g. 30 the-symbolic-quest The Symbolic Quest Уайтмонт|Уитмонт символический поиск|символическая погоня)
|
||||
|
||||
Usage:
|
||||
sweep.py 01 [--nlr] [--only 30 37 46] [--skip-existing]
|
||||
Reads sections/01-fundamentals/INDEX.md statuses; by default sweeps items NOT marked ✅.
|
||||
--only restricts to given nn's; --skip-existing skips items with an output file.
|
||||
|
||||
Per item (polite delays): rsl, lg, alib(surname + title), flib books, cogito search,
|
||||
SearXNG (with OZON snippet parse: year/pages/ISBN), rutitles diff.
|
||||
--nlr adds НРБ (slow, CRW-rendered, ~15s/query).
|
||||
Query log: data/queries.log (JSONL). Output: data/sweeps/<sec>/<nn>-<slug>.md.
|
||||
"""
|
||||
import json, os, re, sys, time, subprocess, urllib.parse, urllib.request
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
UA = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"}
|
||||
SX = "http://localhost:8888/search"
|
||||
LOG = os.path.join(ROOT, "data", "queries.log")
|
||||
|
||||
def log(source, item, query, hits, summary=""):
|
||||
rec = {"ts": time.strftime("%Y-%m-%dT%H:%M:%S"), "item": item, "src": source,
|
||||
"q": query, "hits": hits, "sum": summary[:200]}
|
||||
with open(LOG, "a", encoding="utf-8") as f:
|
||||
f.write(json.dumps(rec, ensure_ascii=False) + "\n")
|
||||
|
||||
def run(cmd, timeout=180):
|
||||
try:
|
||||
r = subprocess.run(cmd, capture_output=True, text=True, timeout=timeout)
|
||||
return (r.stdout or "") + (r.stderr or "")
|
||||
except Exception as e:
|
||||
return "(error: %s)" % e
|
||||
|
||||
def sh_run(*cmd, timeout=180):
|
||||
return run(list(cmd), timeout)
|
||||
|
||||
def rsl(q):
|
||||
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "rsl.py"), q)
|
||||
hits = len(re.findall(r"^\[\d+\]", out, re.M))
|
||||
return out, hits
|
||||
|
||||
def lg(q):
|
||||
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "lg.py"), q)
|
||||
hits = out.count("libgen.vg/edition.php?id=")
|
||||
return out, hits
|
||||
|
||||
def alib(q):
|
||||
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "alib.py"), q)
|
||||
hits = out.count("Купить") + len(re.findall(r"\d+\.(?:\d+)?\s+руб", out))
|
||||
return out, hits
|
||||
|
||||
def flib(q):
|
||||
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "flib.py"), "books", q)
|
||||
hits = len(re.findall(r"^b/\d+", out, re.M))
|
||||
return out, hits
|
||||
|
||||
def cogito(q):
|
||||
url = "https://cogito-shop.com/search/?q=" + urllib.parse.quote(q)
|
||||
out = run(["curl", "-sL", "-A", UA["User-Agent"], "--max-time", "60", url])
|
||||
# product cards only (bx_product_item divs); the first bx_product_list block is the nav menu
|
||||
items = re.split(r'<div class="bx_product_item[^"]*"', out)
|
||||
prods = []
|
||||
for blk in items[1:]:
|
||||
m = re.search(r'<a[^>]*href="(/catalog/[^"]+)"[^>]*>(?:\s|<[^>]+>)*([^<]{15,120})', blk[:6000])
|
||||
if m and m.group(2).strip():
|
||||
prods.append(m.group(2).strip())
|
||||
seen, lines = set(), []
|
||||
for x in prods:
|
||||
if x not in seen:
|
||||
seen.add(x); lines.append(x)
|
||||
return "\n".join(lines), len(lines)
|
||||
|
||||
def sx_ozon(q):
|
||||
url = SX + "?q=" + urllib.parse.quote(q) + "&format=json&language=ru"
|
||||
try:
|
||||
req = urllib.request.Request(url, headers=UA)
|
||||
d = json.loads(urllib.request.urlopen(req, timeout=45).read().decode("utf-8", "replace"))
|
||||
except Exception as e:
|
||||
return "(error: %s)" % e, 0
|
||||
lines, oz = [], 0
|
||||
for r in d.get("results", [])[:12]:
|
||||
urlr = r.get("url", "")
|
||||
title = r.get("title", "")
|
||||
snip = re.sub(r"<[^>]+>", " ", r.get("content", ""))
|
||||
lines.append("- %s\n %s" % (title[:110], snip[:220]))
|
||||
if "ozon.ru" in urlr:
|
||||
oz += 1
|
||||
for m in re.finditer(r"(19|20)\d{2}\s*г?\.\s*—?\s*(\d{2,4})\s+стр", snip):
|
||||
lines.append(" >> OZON meta: %s, %s стр" % (m.group(0)[:30], m.group(2)))
|
||||
for m in re.finditer(r"ISBN[:\s]*(978-[\d\-]{11,14})", snip):
|
||||
lines.append(" >> OZON ISBN: " + m.group(1))
|
||||
return "\n".join(lines), oz
|
||||
|
||||
def nlr(q):
|
||||
out = sh_run(sys.executable, os.path.join(ROOT, "tools", "nlr.py"), q, timeout=300)
|
||||
hits = len(re.findall(r"^\d+\.", out, re.M))
|
||||
return out, hits
|
||||
|
||||
def rutitles(en_title):
|
||||
return sh_run(sys.executable, os.path.join(ROOT, "tools", "rutitles.py"), "both", en_title, "8")
|
||||
|
||||
def index_statuses(secdir):
|
||||
st = {}
|
||||
p = os.path.join(ROOT, "sections", secdir, "INDEX.md")
|
||||
for m in re.finditer(r"^\|\s*(\d{2})\s*\|[^|]*\|[^|]*\|[^|]*\|\s*(✅|🔶|❌|⬜)", open(p, encoding="utf-8").read(), re.M):
|
||||
st[m.group(1)] = m.group(2)
|
||||
return st
|
||||
|
||||
def main():
|
||||
args = sys.argv[1:]
|
||||
sec = args[0] if args else None
|
||||
if not sec or not re.fullmatch(r"\d{2}-[a-z0-9\-]+", sec):
|
||||
print(__doc__); sys.exit(1)
|
||||
do_nlr = "--nlr" in args
|
||||
only = []
|
||||
if "--only" in args:
|
||||
only = args[args.index("--only") + 1: args.index("--only") + 2 + 999]
|
||||
only = [x for x in only if x.isdigit()]
|
||||
skip_existing = "--skip-existing" in args
|
||||
|
||||
secdir = os.path.join(ROOT, "sections", sec)
|
||||
inp = os.path.join(ROOT, "data", "sweeps", sec, "input.tsv")
|
||||
outdir = os.path.join(ROOT, "data", "sweeps", sec)
|
||||
os.makedirs(outdir, exist_ok=True)
|
||||
if not os.path.exists(inp):
|
||||
sys.exit("no %s (create input.tsv: nn slug en_title ru_surnames ru_title_candidates)" % inp)
|
||||
|
||||
st = index_statuses(sec)
|
||||
for line in open(inp, encoding="utf-8"):
|
||||
line = line.strip()
|
||||
if not line or line.startswith("#"):
|
||||
continue
|
||||
parts = line.split("\t")
|
||||
while len(parts) < 5:
|
||||
parts.append("")
|
||||
nn, slug, en_title, surnames, rtitles = parts[:5]
|
||||
if only and nn not in only:
|
||||
continue
|
||||
if not only and st.get(nn) == "✅":
|
||||
continue
|
||||
op = os.path.join(outdir, "%s-%s.md" % (nn, slug))
|
||||
if skip_existing and os.path.exists(op):
|
||||
print("skip %s (exists)" % nn)
|
||||
continue
|
||||
print("== %s %s (%s)" % (nn, en_title, st.get(nn, "?")))
|
||||
rep = ["# Sweep %s — %s" % (nn, en_title), "",
|
||||
"Run: %s" % time.strftime("%Y-%m-%d %H:%M"), "", "status: %s" % st.get(nn, "?"), ""]
|
||||
surnames = [s for s in surnames.split("|") if s]
|
||||
rtitles = [t for t in rtitles.split("|") if t]
|
||||
|
||||
for s in surnames:
|
||||
print(" rsl:", s)
|
||||
out, h = rsl(s); time.sleep(3)
|
||||
rep.append("## RSL «%s» — %d" % (s, h)); rep.append(out.strip()[:3000]); rep.append("")
|
||||
log("rsl", nn, s, h)
|
||||
print(" lg:", s)
|
||||
out, h = lg(s); time.sleep(4)
|
||||
rep.append("## libgen «%s» — %d" % (s, h)); rep.append(out.strip()[:3000]); rep.append("")
|
||||
log("libgen", nn, s, h)
|
||||
print(" alib:", s)
|
||||
out, h = alib(s); time.sleep(2)
|
||||
rep.append("## alib «%s» — %d" % (s, h)); rep.append(out.strip()[:2000]); rep.append("")
|
||||
log("alib", nn, s, h)
|
||||
print(" flib:", s)
|
||||
out, h = flib(s); time.sleep(2)
|
||||
rep.append("## flibusta «%s» — %d" % (s, h)); rep.append(out.strip()[:2000]); rep.append("")
|
||||
log("flibusta", nn, s, h)
|
||||
print(" cogito:", s)
|
||||
out, h = cogito(s); time.sleep(2)
|
||||
rep.append("## cogito «%s» — %d" % (s, h)); rep.append(out.strip()[:2000]); rep.append("")
|
||||
log("cogito", nn, s, h)
|
||||
|
||||
for t in rtitles:
|
||||
print(" sx:", t)
|
||||
out, h = sx_ozon(t); time.sleep(2)
|
||||
rep.append("## SearXNG/OZON «%s» — %d ozon" % (t, h)); rep.append(out.strip()[:3000]); rep.append("")
|
||||
log("searxng", nn, t, h)
|
||||
print(" alib(title):", t)
|
||||
out, h = alib(t); time.sleep(2)
|
||||
rep.append("## alib title «%s» — %d" % (t, h)); rep.append(out.strip()[:2000]); rep.append("")
|
||||
log("alib-title", nn, t, h)
|
||||
|
||||
print(" rutitles:")
|
||||
out = rutitles(en_title)
|
||||
rep.append("## rutitles diff (EN→RU + DB match)"); rep.append(out.strip()[:2500]); rep.append("")
|
||||
log("rutitles", nn, en_title, 0)
|
||||
|
||||
if do_nlr and surnames:
|
||||
print(" nlr:", surnames[0])
|
||||
out, h = nlr(surnames[0])
|
||||
rep.append("## NLR «%s» — %d" % (surnames[0], h)); rep.append(out.strip()[:2500]); rep.append("")
|
||||
log("nlr", nn, surnames[0], h)
|
||||
time.sleep(5)
|
||||
|
||||
open(op, "w", encoding="utf-8").write("\n".join(rep))
|
||||
print(" -> %s" % op)
|
||||
|
||||
print("done")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Loading…
Add table
Add a link
Reference in a new issue