Files
Joel Hooks 016dccc295 redesign: cool finds index — full titles, readable descriptions, compact cards
- Titles wrap fully instead of truncating
- Relevance text shows complete (no line-clamp)
- Date shown as relative time (2d ago) above title, not stealing width
- Single type badge (repo/article/video) replaces full tag list
- Source URL removed from index (shown on detail page)
- Tighter card spacing for better scan density
- Re-removed duplicate self-hosting discovery (also from Vault source)
2026-02-22 08:07:35 -08:00

59 lines
1.5 KiB
Python
Executable File

#!/usr/bin/env python3
"""Embed text via all-mpnet-base-v2 (768-dim). Reads JSON lines from stdin, writes JSON array to stdout.
Input (one JSON per line): {"id": "...", "text": "..."}
Output (JSON array): [{"id": "...", "vector": [0.1, ...]}]
Or single text via --text flag:
embed.py --text "some text"
Output: [0.1, 0.2, ...]
"""
import sys
import json
import argparse
from sentence_transformers import SentenceTransformer
# Suppress the "loading from different task" note
import logging
logging.getLogger("sentence_transformers").setLevel(logging.ERROR)
model = SentenceTransformer("all-mpnet-base-v2")
def main():
parser = argparse.ArgumentParser()
parser.add_argument("--text", type=str, help="Single text to embed")
args = parser.parse_args()
if args.text:
vec = model.encode(args.text).tolist()
json.dump(vec, sys.stdout)
sys.stdout.write("\n")
return
# Batch mode: JSON lines from stdin
items = []
for line in sys.stdin:
line = line.strip()
if not line:
continue
items.append(json.loads(line))
if not items:
json.dump([], sys.stdout)
sys.stdout.write("\n")
return
texts = [item["text"] for item in items]
vectors = model.encode(texts).tolist()
results = []
for item, vec in zip(items, vectors):
results.append({"id": item["id"], "vector": vec})
json.dump(results, sys.stdout)
sys.stdout.write("\n")
if __name__ == "__main__":
main()