74 lines
2.7 KiB
Python
74 lines
2.7 KiB
Python
# retoor <retoor@molodetz.nl>
|
|||
|
|
import math
|
||
|
|
import re
|
||
|
|
from collections import Counter
|
||
|
|
from functools import lru_cache
|
||
|
|
|
||
|
|
from bs4 import BeautifulSoup
|
||
|
|
|
||
|
|
from molodetz.docs_api import ordered_groups
|
||
|
|
from molodetz.docs_prose import DOCS_PAGES, render_page
|
||
|
|
|
||
|
|
TOKEN = re.compile(r"[a-z0-9_]+")
|
||
|
|
K1 = 1.5
|
||
|
|
B = 0.75
|
||
|
|
|
||
|
|
|
||
|
|
def tokenize(text):
|
||
|
|
return TOKEN.findall((text or "").lower())
|
||
|
|
|
||
|
|
|
||
|
|
@lru_cache(maxsize=1)
|
||
|
|
def build_index():
|
||
|
|
documents = []
|
||
|
|
for page in DOCS_PAGES:
|
||
|
|
text = BeautifulSoup(render_page(page["slug"]), "html.parser").get_text(" ")
|
||
|
|
documents.append({"title": page["title"], "url": f"/docs/{page['slug']}", "text": text, "admin": bool(page.get("admin"))})
|
||
|
|
for group in ordered_groups():
|
||
|
|
lines = [group["title"], group["description"]]
|
||
|
|
for item in group["endpoints"]:
|
||
|
|
lines.append(f"{item['method']} {item['path']} {item['title']} {item['description']}")
|
||
|
|
documents.append(
|
||
|
|
{"title": f"API: {group['title']}", "url": f"/docs/api/{group['slug']}", "text": " ".join(lines), "admin": group["slug"] == "beheer"}
|
||
|
|
)
|
||
|
|
tokens = [Counter(tokenize(doc["title"] + " " + doc["text"])) for doc in documents]
|
||
|
|
lengths = [sum(counter.values()) for counter in tokens]
|
||
|
|
average = sum(lengths) / max(1, len(lengths))
|
||
|
|
frequency = Counter()
|
||
|
|
for counter in tokens:
|
||
|
|
frequency.update(counter.keys())
|
||
|
|
return documents, tokens, lengths, average, frequency
|
||
|
|
|
||
|
|
|
||
|
|
def _snippet(text, terms, width=160):
|
||
|
|
lowered = text.lower()
|
||
|
|
position = min((lowered.find(term) for term in terms if lowered.find(term) >= 0), default=0)
|
||
|
|
start = max(0, position - width // 3)
|
||
|
|
snippet = text[start : start + width].strip()
|
||
|
|
for term in terms:
|
||
|
|
snippet = re.sub(f"({re.escape(term)})", r"[[\1]]", snippet, flags=re.IGNORECASE)
|
||
|
|
return snippet
|
||
|
|
|
||
|
|
|
||
|
|
def search(query, viewer_is_admin=False, limit=20):
|
||
|
|
terms = tokenize(query)
|
||
|
|
if not terms:
|
||
|
|
return []
|
||
|
|
documents, tokens, lengths, average, frequency = build_index()
|
||
|
|
total = len(documents)
|
||
|
|
results = []
|
||
|
|
for index, doc in enumerate(documents):
|
||
|
|
if doc["admin"] and not viewer_is_admin:
|
||
|
|
continue
|
||
|
|
score = 0.0
|
||
|
|
for term in terms:
|
||
|
|
tf = tokens[index].get(term, 0)
|
||
|
|
if not tf:
|
||
|
|
continue
|
||
|
|
idf = math.log(1 + (total - frequency[term] + 0.5) / (frequency[term] + 0.5))
|
||
|
|
score += idf * tf * (K1 + 1) / (tf + K1 * (1 - B + B * lengths[index] / average))
|
||
|
|
if score > 0:
|
||
|
|
results.append({"title": doc["title"], "url": doc["url"], "score": round(score, 3), "snippet": _snippet(doc["text"], terms)})
|
||
|
|
results.sort(key=lambda item: item["score"], reverse=True)
|
||
|
|
return results[:limit]
|