import json import re import sys import urllib.request from html.parser import HTMLParser from pathlib import Path import numpy as np SITE = Path(sys.argv[1] if len(sys.argv) > 1 else ".") OLLAMA = "http://127.0.0.1:11434" EMBED_MODEL = "nomic-embed-text" WINDOW, OVERLAP, MAX_SECTION = 300, 50, 360 SKIP_TAGS = {"script", "style", "svg", "nav", "figure", "button", "canvas", "noscript"} SKIP_CLASSES = {"fig", "dither", "author", "keep", "byline"} VOID = {"br", "hr", "img", "input", "meta", "link", "wbr", "source", "col", "area"} BREAKS = {"p", "li", "pre", "tr", "h3", "h4", "blockquote", "table", "ul", "ol", "div"} class ArticleParser(HTMLParser): def __init__(self): super().__init__() self.stack, self.in_article = [], False self.title, self.heading, self.buf = "", None, [] self.sections = [["Introduction", []]] self.table, self.cell = None, None def skipping(self): return bool(self.stack) and self.stack[-1][1] def flush(self): self.sections[-1][1].append("".join(self.buf)) self.buf = [] def handle_starttag(self, tag, attrs): if tag == "article": self.in_article = True if not self.in_article or tag in VOID: return classes = set((dict(attrs).get("class") or "").split()) skip = self.skipping() or tag in SKIP_TAGS or bool(classes & SKIP_CLASSES) self.stack.append((tag, skip)) if skip: return if self.table_start(tag): return if tag == "h2": self.flush() self.heading = [] elif tag in BREAKS: self.buf.append("\n") def handle_endtag(self, tag): if tag == "article" and self.in_article: self.flush() self.in_article = False if not self.in_article or tag in VOID or tag not in {t for t, _ in self.stack}: return while self.stack: open_tag, skip = self.stack.pop() if open_tag == tag: break if skip or self.table_end(tag): return if tag == "h2" and self.heading is not None: self.sections.append([clean("".join(self.heading)), []]) self.heading = None elif tag == "h1": self.title = clean("".join(self.buf)) self.buf = [] elif tag in {"h3", "h4"}: self.buf.append(". ") if tag in BREAKS | {"td", "th"}: self.buf.append(" ") self.flush() def table_start(self, tag): if tag == "table": self.table = {"headers": [], "head": False, "row": [], "tags": []} elif self.table is None: return False elif tag == "thead": self.table["head"] = True elif tag == "tr": self.table["row"], self.table["tags"] = [], [] elif tag in ("td", "th"): self.cell = [] self.table["tags"].append(tag) return True def table_end(self, tag): t = self.table if t is None: return False if tag == "thead": t["head"] = False elif tag in ("td", "th") and self.cell is not None: t["row"].append(clean("".join(self.cell))) self.cell = None elif tag == "tr": self.emit_row(t) elif tag == "table": self.table = None self.flush() return True def emit_row(self, t): cells = t["row"] if t["head"] or (not t["headers"] and all(tag == "th" for tag in t["tags"])): t["headers"] = cells return if len(t["headers"]) == len(cells): line = "; ".join(f"{h}: {v}" if h else v for h, v in zip(t["headers"], cells) if v) else: line = " | ".join(cells) self.buf.append(f" {line}. ") self.flush() def handle_data(self, data): if not self.in_article or self.skipping(): return if self.cell is not None: self.cell.append(data) return (self.heading if self.heading is not None else self.buf).append(data) def clean(text): return re.sub(r"\s+", " ", text).strip() def windows(words): if len(words) <= MAX_SECTION: return [words] step = WINDOW - OVERLAP return [words[i:i + WINDOW] for i in range(0, len(words) - OVERLAP, step)] def parse(path, kind): parser = ArticleParser() parser.feed(path.read_text(encoding="utf-8")) url = f"https://rohitghumare.com/{kind}/{path.parent.name}/" for section, parts in parser.sections: words = clean(" ".join(parts)).split() for part in windows(words) if words else []: yield {"url": url, "title": parser.title, "section": section, "text": " ".join(part)} def embed(texts, prefix): body = json.dumps({"model": EMBED_MODEL, "input": [prefix + t for t in texts]}).encode() req = urllib.request.Request(f"{OLLAMA}/api/embed", body, {"Content-Type": "application/json"}) with urllib.request.urlopen(req, timeout=600) as resp: vectors = np.array(json.load(resp)["embeddings"], dtype=np.float32) return vectors / np.linalg.norm(vectors, axis=1, keepdims=True) def main(): chunks = [] for kind in ("blog", "guides"): for path in sorted(SITE.glob(f"{kind}/*/index.html")): chunks.extend(parse(path, kind)) for i, chunk in enumerate(chunks): chunk["id"] = f"c{i}" docs = [f"{c['title']} | {c['section']}\n{c['text']}" for c in chunks] vectors = np.concatenate([embed(docs[i:i + 32], "search_document: ") for i in range(0, len(docs), 32)]) with open("chunks.jsonl", "w") as f: f.writelines(json.dumps(c) + "\n" for c in chunks) np.save("embeddings.npy", vectors) pages = len({c["url"] for c in chunks}) print(f"{pages} pages, {len(chunks)} chunks, embeddings {vectors.shape}", file=sys.stderr) if __name__ == "__main__": main()