"""LLMLingua-2 token-pruning sidecar for correx. A tiny HTTP service correx calls (via HttpTokenPruner) to prune low-perplexity tokens from freeform prose before it's sent to the local LLM. Kept in Python because LLMLingua-2 is a torch/BERT classifier with no JVM equivalent (pipeline ยง4). Contract: POST /prune {"text": str, "protected": [str], "rate": float} -> {"compressed": str} rate = fraction of tokens to KEEP (0.55 keeps ~55%, i.e. ~45% compression). `protected` substrings are force-kept verbatim (IDs, numbers, paths, code). GET /health -> {"status": "ok"} Run: pip install -r requirements.txt uvicorn server:app --host 127.0.0.1 --port 8199 """ from __future__ import annotations import os from fastapi import FastAPI from pydantic import BaseModel app = FastAPI(title="correx-llmlingua") _MODEL = os.environ.get("LLMLINGUA_MODEL", "microsoft/llmlingua-2-xlm-roberta-large-meetingbank") _compressor = None def _get_compressor(): # Lazy-load so /health works (and the process starts fast) before torch spins up. global _compressor if _compressor is None: from llmlingua import PromptCompressor _compressor = PromptCompressor(model_name=_MODEL, use_llmlingua2=True) return _compressor class PruneRequest(BaseModel): text: str protected: list[str] = [] rate: float = 0.55 class PruneResponse(BaseModel): compressed: str @app.get("/health") def health(): return {"status": "ok"} @app.post("/prune", response_model=PruneResponse) def prune(req: PruneRequest): text = req.text.strip() if not text: return PruneResponse(compressed=req.text) compressor = _get_compressor() # LLMLingua-2 asserts len(force_tokens) <= max_force_token (default 100). Large static docs # (CLAUDE.md/AGENTS.md) yield hundreds of protected spans, which used to blow the assert and # 500 -> the doc came back uncompressed. Dedup and keep the longest spans (most load-bearing: # full paths, hashes, code fences beat bare numbers) up to the cap. cap = getattr(compressor, "max_force_token", 100) forced = sorted(set(req.protected), key=len, reverse=True)[:cap] or None # rate is fraction to keep; LLMLingua-2 force_tokens keeps the protected spans verbatim. result = compressor.compress_prompt( text, rate=max(0.1, min(1.0, req.rate)), force_tokens=forced, drop_consecutive=True, ) return PruneResponse(compressed=result["compressed_prompt"])