Molodetz blogrol volgens DPP-template
This commit is contained in:
@@ -0,0 +1,73 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
import math
|
||||
import re
|
||||
from collections import Counter
|
||||
from functools import lru_cache
|
||||
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
from molodetz.docs_api import ordered_groups
|
||||
from molodetz.docs_prose import DOCS_PAGES, render_page
|
||||
|
||||
TOKEN = re.compile(r"[a-z0-9_]+")
|
||||
K1 = 1.5
|
||||
B = 0.75
|
||||
|
||||
|
||||
def tokenize(text):
|
||||
return TOKEN.findall((text or "").lower())
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def build_index():
|
||||
documents = []
|
||||
for page in DOCS_PAGES:
|
||||
text = BeautifulSoup(render_page(page["slug"]), "html.parser").get_text(" ")
|
||||
documents.append({"title": page["title"], "url": f"/docs/{page['slug']}", "text": text, "admin": bool(page.get("admin"))})
|
||||
for group in ordered_groups():
|
||||
lines = [group["title"], group["description"]]
|
||||
for item in group["endpoints"]:
|
||||
lines.append(f"{item['method']} {item['path']} {item['title']} {item['description']}")
|
||||
documents.append(
|
||||
{"title": f"API: {group['title']}", "url": f"/docs/api/{group['slug']}", "text": " ".join(lines), "admin": group["slug"] == "beheer"}
|
||||
)
|
||||
tokens = [Counter(tokenize(doc["title"] + " " + doc["text"])) for doc in documents]
|
||||
lengths = [sum(counter.values()) for counter in tokens]
|
||||
average = sum(lengths) / max(1, len(lengths))
|
||||
frequency = Counter()
|
||||
for counter in tokens:
|
||||
frequency.update(counter.keys())
|
||||
return documents, tokens, lengths, average, frequency
|
||||
|
||||
|
||||
def _snippet(text, terms, width=160):
|
||||
lowered = text.lower()
|
||||
position = min((lowered.find(term) for term in terms if lowered.find(term) >= 0), default=0)
|
||||
start = max(0, position - width // 3)
|
||||
snippet = text[start : start + width].strip()
|
||||
for term in terms:
|
||||
snippet = re.sub(f"({re.escape(term)})", r"[[\1]]", snippet, flags=re.IGNORECASE)
|
||||
return snippet
|
||||
|
||||
|
||||
def search(query, viewer_is_admin=False, limit=20):
|
||||
terms = tokenize(query)
|
||||
if not terms:
|
||||
return []
|
||||
documents, tokens, lengths, average, frequency = build_index()
|
||||
total = len(documents)
|
||||
results = []
|
||||
for index, doc in enumerate(documents):
|
||||
if doc["admin"] and not viewer_is_admin:
|
||||
continue
|
||||
score = 0.0
|
||||
for term in terms:
|
||||
tf = tokens[index].get(term, 0)
|
||||
if not tf:
|
||||
continue
|
||||
idf = math.log(1 + (total - frequency[term] + 0.5) / (frequency[term] + 0.5))
|
||||
score += idf * tf * (K1 + 1) / (tf + K1 * (1 - B + B * lengths[index] / average))
|
||||
if score > 0:
|
||||
results.append({"title": doc["title"], "url": doc["url"], "score": round(score, 3), "snippet": _snippet(doc["text"], terms)})
|
||||
results.sort(key=lambda item: item["score"], reverse=True)
|
||||
return results[:limit]
|
||||
Reference in New Issue
Block a user