diff --git a/.github/RELEASE_NOTES.md b/.github/RELEASE_NOTES.md new file mode 100644 index 0000000..285b708 --- /dev/null +++ b/.github/RELEASE_NOTES.md @@ -0,0 +1,19 @@ +All 26 headers in one download: **llm-cpp-headers.zip**. + +Inside: + +- `include/`: the current `llm_*.hpp` from each of the 26 library repos +- `examples/offline/`: six programs that need no API key, with the output they print +- `README.txt`: the three-step setup and which headers need libcurl +- `LICENSE` (MIT) + +Before this zip was built, CI compiled every implementation with g++ (C++17), linked all 26 into one binary with libcurl, and re-ran the offline examples against these exact headers, diffing their output. + +Quick try, no key needed: + +```bash +unzip llm-cpp-headers.zip && cd llm-cpp-headers +g++ -std=c++17 -Iinclude examples/offline/cache.cpp -o cache && ./cache +``` + +Browse what each header does: https://mattbusel.github.io/llm-cpp/ diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 0339cf8..e95fcf1 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -38,3 +38,24 @@ jobs: run: | python tools/build_site.py git diff --exit-code docs/index.html + link-all: + name: All 26 implementations compile and link together (g++, libcurl) + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - name: Install libcurl + run: sudo apt-get update -qq && sudo apt-get install -y -qq libcurl4-openssl-dev + - name: Fetch headers, compile each implementation, link one binary + run: | + mkdir -p third_party build/impl + python3 -c "import json;[print(l['name'], l['macro']) for l in json.load(open('tools/libraries.json'))['libraries']]" > libs.txt + while read name macro; do + file=$(echo "${name#llm-}" | tr - _) + curl -fsSL -o "third_party/llm_$file.hpp" \ + "https://raw.githubusercontent.com/Mattbusel/$name/main/include/llm_$file.hpp" + printf '#define %s\n#include "llm_%s.hpp"\n' "$macro" "$file" > "build/impl/$file.cpp" + g++ -std=c++17 -O1 -Ithird_party -c "build/impl/$file.cpp" -o "build/impl/$file.o" + done < libs.txt + echo 'int main() { return 0; }' > build/main.cpp + g++ -std=c++17 build/main.cpp build/impl/*.o -lcurl -lpthread -o build/all26 + ./build/all26 && echo "linked $(ls build/impl/*.o | wc -l) implementations" diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml new file mode 100644 index 0000000..781b262 --- /dev/null +++ b/.github/workflows/release.yml @@ -0,0 +1,91 @@ +name: Release + +# Push a tag like v0.2.0. This fetches the current header from each of the +# 26 library repos, proves every implementation compiles and that all 26 link +# into one binary, rebuilds the offline examples and diffs their output, then +# attaches llm-cpp-headers.zip to a GitHub release. +on: + push: + tags: [ 'v*' ] + workflow_dispatch: + inputs: + tag: + description: 'Existing tag to (re)build the zip for' + required: true + +permissions: + contents: write + +jobs: + zip: + runs-on: ubuntu-latest + env: + TAG: ${{ github.event.inputs.tag || github.ref_name }} + GH_TOKEN: ${{ github.token }} + steps: + - uses: actions/checkout@v4 + - name: Install libcurl + run: sudo apt-get update -qq && sudo apt-get install -y -qq libcurl4-openssl-dev zip + - name: Fetch the 26 headers + run: | + mkdir -p pkg/llm-cpp-headers/include + python3 -c "import json;[print(l['name'], l['macro']) for l in json.load(open('tools/libraries.json'))['libraries']]" > libs.txt + while read name macro; do + file=$(echo "${name#llm-}" | tr - _) + curl -fsSL -o "pkg/llm-cpp-headers/include/llm_$file.hpp" \ + "https://raw.githubusercontent.com/Mattbusel/$name/main/include/llm_$file.hpp" + done < libs.txt + ls pkg/llm-cpp-headers/include | wc -l + - name: Every implementation compiles, all 26 link into one binary + run: | + mkdir -p build/impl + while read name macro; do + file=$(echo "${name#llm-}" | tr - _) + printf '#define %s\n#include "llm_%s.hpp"\n' "$macro" "$file" > "build/impl/$file.cpp" + g++ -std=c++17 -O1 -Ipkg/llm-cpp-headers/include -c "build/impl/$file.cpp" -o "build/impl/$file.o" + done < libs.txt + echo 'int main() { return 0; }' > build/main.cpp + g++ -std=c++17 build/main.cpp build/impl/*.o -lcurl -lpthread -o build/all26 + ./build/all26 && echo "linked $(ls build/impl/*.o | wc -l) implementations" + - name: Offline examples still print what the README shows + run: | + for lib in cache cost guard json format compress; do + g++ -std=c++17 -O2 -Ipkg/llm-cpp-headers/include examples/offline/$lib.cpp -o build/$lib + ./build/$lib > build/$lib.txt + diff -u examples/offline/output/$lib.txt build/$lib.txt + done + - name: Build the zip + run: | + d=pkg/llm-cpp-headers + mkdir -p $d/examples + cp -r examples/offline $d/examples/ + cp LICENSE $d/ + { + echo "llm-cpp $TAG: 26 single-header C++17 libraries for LLM features" + echo "https://github.com/Mattbusel/llm-cpp" + echo + echo "1. Copy the headers you want from include/ into your project." + echo "2. In exactly ONE .cpp file, define the implementation macro before including:" + echo " #define LLM_CACHE_IMPLEMENTATION" + echo " #include \"llm_cache.hpp\"" + echo " Every other file just includes the header." + echo "3. Compile as C++17. Headers marked libcurl also need -lcurl." + echo + echo "Try one with no API key:" + echo " g++ -std=c++17 -Iinclude examples/offline/cache.cpp -o cache && ./cache" + echo " cl /std:c++17 /EHsc /Iinclude examples\\offline\\cache.cpp" + echo + echo "Header needs implementation macro" + python3 -c "import json;[print(f\"llm_{l['name'][4:].replace('-','_')}.hpp\".ljust(24)+('libcurl' if l['deps']=='curl' else 'none').ljust(10)+l['macro']) for l in json.load(open('tools/libraries.json'))['libraries']]" + echo + echo "Headers fetched from each library's main branch on $(date -u +%Y-%m-%d)." + } > $d/README.txt + (cd pkg && zip -qr ../llm-cpp-headers.zip llm-cpp-headers) + unzip -l llm-cpp-headers.zip | tail -1 + - name: Publish + run: | + if gh release view "$TAG" >/dev/null 2>&1; then + gh release upload "$TAG" llm-cpp-headers.zip --clobber + else + gh release create "$TAG" llm-cpp-headers.zip --title "llm-cpp $TAG" --notes-file .github/RELEASE_NOTES.md + fi diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index e228ac9..8f1596a 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -1,29 +1,22 @@ # Contributing to llm-cpp -llm-cpp is a collection of 26 single-header C++20 libraries. Each header is intentionally self-contained — zero dependencies, drop-in integration. +llm-cpp is the index for 26 single-header C++17 libraries. Each library has its own repository, `github.com/Mattbusel/llm-`, with the header in `include/llm_.hpp`. -## What we want +## Where to send what -- **New headers** — if you need a primitive that fits the single-header philosophy, propose it -- **Bug fixes** — correctness issues, edge cases, compiler compatibility -- **Performance improvements** — anything that reduces overhead on the hot path -- **New providers** — extending existing headers to support additional LLM APIs -- **Tests** — the `tests/` directory uses a lightweight harness; more coverage is always welcome +- **A bug or improvement in one library**: open the issue or pull request in that library's repo. +- **This repo** holds the README, the catalogue site (`docs/`, built by `python tools/build_site.py` from `tools/libraries.json` and `tools/site_template.html`), the offline examples in `examples/offline/`, and the release workflow that zips all 26 headers. -## What we don't want +## Rules for the headers -- Headers that require external dependencies (defeats the purpose) -- C++17 or earlier features (this is a C++20 library) -- Anything that adds build system requirements beyond a C++20 compiler +- C++17, single file, stb-style: declarations always, implementation only under `#ifdef LLM__IMPLEMENTATION`. +- Offline libraries use only the standard library. Network libraries may use libcurl and nothing else. -## How to contribute +## Checking a change here -1. Fork and clone -2. Add or modify a header in the appropriate location -3. Add tests to `tests/` -4. Verify with `g++ -std=c++20 -Wall -Wextra tests/your_test.cpp -o test && ./test` (or MSVC equivalent) -5. Open a PR +```bash +python tools/build_site.py # regenerates docs/index.html; CI fails if it drifts +g++ -std=c++17 -Ithird_party examples/offline/cache.cpp -o cache && ./cache +``` -## Questions - -Open a [Discussion](https://github.com/Mattbusel/llm-cpp/discussions). +If an example's output changes, update the matching file in `examples/offline/output/`; CI diffs them. diff --git a/README.md b/README.md index 749c96b..664a42a 100644 --- a/README.md +++ b/README.md @@ -3,231 +3,118 @@ llm-cpp: the llm_cache.hpp header next to a terminal that downloads it, compiles an example with MSVC and prints real cache hits and evictions -[![CI](https://github.com/Mattbusel/llm-cpp/actions/workflows/ci.yml/badge.svg)](https://github.com/Mattbusel/llm-cpp/actions/workflows/ci.yml) - -**[Browse the catalogue](https://mattbusel.github.io/llm-cpp/)**: filter all 26 libraries by what they need, see real output, and get an install command for the headers you pick. +# llm-cpp: single-header C++ libraries for LLM features -**26 single-header C++17 libraries for building LLM features into native code.** Streaming, retries, caching, cost estimation, RAG, reranking, tracing, structured output, agents and more. Each library is one `.hpp` file you copy into your project. +**Add ChatGPT or Claude features to a C++ program by copying one file.** Streaming, retries, caching, cost estimates, RAG, structured JSON output, tool-calling agents and 19 more, as 26 single-header C++17 libraries for the OpenAI and Anthropic APIs. No SDK, no package manager. -Most LLM tooling assumes Python or Node. If you are shipping a game, a desktop app, a trading system, an embedded tool or a C++ service, you usually end up hand-rolling HTTP calls, retry loops and JSON parsing. llm-cpp is that plumbing, split into small pieces so you take only what you need: no SDK, no package manager, no framework. The offline libraries have no dependencies at all; the ones that talk to OpenAI or Anthropic need only libcurl. +**Who it's for:** C++ developers shipping a game, desktop app, trading system, embedded tool or service who want LLM calls without a Python sidecar. -## Start here - -| I want to... | Use | -|---|---| -| Call a model and stream tokens | [llm-stream](https://github.com/Mattbusel/llm-stream) | -| Build a chatbot with memory | [llm-chat](https://github.com/Mattbusel/llm-chat) + [llm-retry](https://github.com/Mattbusel/llm-retry) | -| Answer questions over my documents | [llm-parse](https://github.com/Mattbusel/llm-parse) + [llm-embed](https://github.com/Mattbusel/llm-embed) or [llm-rag](https://github.com/Mattbusel/llm-rag) + [llm-rank](https://github.com/Mattbusel/llm-rank) | -| Get valid JSON back every time | [llm-format](https://github.com/Mattbusel/llm-format) + [llm-json](https://github.com/Mattbusel/llm-json) | -| Let the model call my C++ functions | [llm-agent](https://github.com/Mattbusel/llm-agent) | -| Know what my calls cost and where time goes | [llm-cost](https://github.com/Mattbusel/llm-cost) + [llm-log](https://github.com/Mattbusel/llm-log) + [llm-trace](https://github.com/Mattbusel/llm-trace) | -| Unit-test LLM code without the network | [llm-mock](https://github.com/Mattbusel/llm-mock) | - -## The libraries - -"libcurl" means the implementation makes HTTPS calls (OpenAI and/or Anthropic APIs). "none" means it is fully offline and uses only the standard library. - -### Core - -| Library | What it does | Needs | -|---|---|---| -| [llm-stream](https://github.com/Mattbusel/llm-stream) | Stream OpenAI and Anthropic chat responses token by token over SSE | libcurl | -| [llm-retry](https://github.com/Mattbusel/llm-retry) | Exponential backoff with jitter, provider failover and a circuit breaker | none | -| [llm-cost](https://github.com/Mattbusel/llm-cost) | Approximate token counts and cost estimates for built-in OpenAI and Anthropic models, budget checks | none | -| [llm-cache](https://github.com/Mattbusel/llm-cache) | LRU response cache with TTL and hit/miss stats, so identical prompts skip the API | none | -| [llm-format](https://github.com/Mattbusel/llm-format) | Define a schema, validate model JSON against it, and re-prompt until the output conforms | none | -| [llm-json](https://github.com/Mattbusel/llm-json) | Small JSON parser and builder for request bodies and model output | none | - -### Data and retrieval - -| Library | What it does | Needs | -|---|---|---| -| [llm-parse](https://github.com/Mattbusel/llm-parse) | Strip HTML and markdown, extract titles, links, headings and code blocks, chunk text | none | -| [llm-embed](https://github.com/Mattbusel/llm-embed) | OpenAI embeddings, cosine/dot/euclidean similarity and a small on-disk vector store | libcurl | -| [llm-rag](https://github.com/Mattbusel/llm-rag) | End-to-end RAG: chunk, embed, persist an index, retrieve top-k and answer | libcurl | -| [llm-rank](https://github.com/Mattbusel/llm-rank) | Rerank passages with offline BM25, LLM relevance scoring, or a hybrid of both | libcurl (linked; BM25 itself is offline) | -| [llm-compress](https://github.com/Mattbusel/llm-compress) | Shrink conversation history: head/tail/smart truncation, sliding window, LLM summary | none (libcurl only with `LLM_COMPRESS_SUMMARIZE`) | -| [llm-batch](https://github.com/Mattbusel/llm-batch) | Run a JSONL file of prompts through a thread pool with rate limiting and resumable checkpoints | libcurl | - -### Operations and testing - -| Library | What it does | Needs | -|---|---|---| -| [llm-log](https://github.com/Mattbusel/llm-log) | Structured JSONL log of every call with latency, tokens and cost, plus query and summary | none | -| [llm-trace](https://github.com/Mattbusel/llm-trace) | RAII spans with parent/child nesting, token and cost attributes, OTLP-style JSON export | none | -| [llm-pool](https://github.com/Mattbusel/llm-pool) | Worker pool with priority queue and requests-per-minute and tokens-per-minute limits | none | -| [llm-mock](https://github.com/Mattbusel/llm-mock) | Fake LLM with scripted, pattern, random or echo responses, simulated latency and streaming | none | -| [llm-eval](https://github.com/Mattbusel/llm-eval) | Run a prompt N times, measure consistency, compare models or prompts, score responses | libcurl | -| [llm-ab](https://github.com/Mattbusel/llm-ab) | A/B test prompts or models with Welch's t-test, Cohen's d and custom scorers | libcurl | - -### Application features - -| Library | What it does | Needs | -|---|---|---| -| [llm-chat](https://github.com/Mattbusel/llm-chat) | Multi-turn conversation with token-budget trimming, pinned system prompt, save and restore | libcurl | -| [llm-agent](https://github.com/Mattbusel/llm-agent) | Tool-calling agent loop: register C++ lambdas as tools and let the model call them | libcurl | -| [llm-vision](https://github.com/Mattbusel/llm-vision) | Send images (file or URL) plus a prompt to OpenAI or Anthropic vision models | libcurl | -| [llm-template](https://github.com/Mattbusel/llm-template) | Mustache-style prompt templates with loops, conditionals and token-budget truncation | none | -| [llm-router](https://github.com/Mattbusel/llm-router) | Pick a model per prompt from a complexity score and a cost, latency, quality or budget strategy | none | -| [llm-guard](https://github.com/Mattbusel/llm-guard) | Detect and scrub PII (email, phone, SSN, card numbers, API keys) and score prompt-injection risk | none | -| [llm-audio](https://github.com/Mattbusel/llm-audio) | Whisper transcription and translation, and text-to-speech, via the OpenAI API | libcurl | -| [llm-finetune](https://github.com/Mattbusel/llm-finetune) | OpenAI fine-tuning lifecycle: write JSONL, upload, create, poll, cancel, list models | libcurl | - -## Why single-header - - - - The llm-cache repo on the left with only include/llm_cache.hpp highlighted, copied with curl into third_party/ of your project on the right - +[![CI](https://github.com/Mattbusel/llm-cpp/actions/workflows/ci.yml/badge.svg)](https://github.com/Mattbusel/llm-cpp/actions/workflows/ci.yml) +[![Latest release](https://img.shields.io/github/v/release/Mattbusel/llm-cpp)](https://github.com/Mattbusel/llm-cpp/releases/latest) +[![License: MIT](https://img.shields.io/badge/license-MIT-blue)](LICENSE) -- **Nothing to install.** `curl -O` one file, `#include` it. It works the same with CMake, Make, Bazel, MSBuild or a one-line `g++` command. -- **You can read all of it.** Each library is 210 to 572 lines; all 26 together are 8,923. When something misbehaves you open one file, not a dependency tree. -- **You pay for what you use.** Need retries and a cache? Take two headers. Nothing else is pulled in, and the offline ones add no link dependencies at all. -- **Easy to vendor.** Copy the headers into `third_party/`, pin them in your own repo, patch them if you need to. No version resolver involved. +## Download -## Install +### [Download all 26 headers (.zip)](https://github.com/Mattbusel/llm-cpp/releases/latest/download/llm-cpp-headers.zip) -Grab the headers you want (each lives at `include/.hpp` in its repo): +The zip holds `include/` with every `llm_*.hpp`, six offline example programs and a short README. Or take just the one you need: ```bash -mkdir -p third_party && cd third_party -for lib in stream retry log; do - curl -fsSLO https://raw.githubusercontent.com/Mattbusel/llm-$lib/main/include/llm_$lib.hpp -done +curl -fsSLO https://raw.githubusercontent.com/Mattbusel/llm-cache/main/include/llm_cache.hpp ``` -Every header follows the stb-style pattern: include it anywhere for the declarations, and in exactly one `.cpp` file define `LLM__IMPLEMENTATION` before including it to compile the implementation. - -## Real output, no API key +**[Browse the catalogue site](https://mattbusel.github.io/llm-cpp/)**: filter the 26 libraries, see real output, and get an install command written for the headers you pick. -Six of the offline libraries have complete example programs in [`examples/offline`](examples/offline), with the output they printed committed next to them. CI downloads each library's current header, builds every example with g++ and diffs the output, so these stay honest. +## Which header do I need? - - guard.cpp scans a prompt for an email, a card number and an API key, scores it 0.75 for injection and prints the scrubbed text + + A chart of sixteen tasks mapped to headers. Stream a reply: llm_stream. Retry and fail over: llm_retry. Skip repeat prompts: llm_cache. Price a prompt: llm_cost. Valid JSON: llm_format plus llm_json. Chatbot with memory: llm_chat plus llm_retry. Documents Q and A: llm_rag plus llm_rank. Long chats: llm_compress. Tool calling: llm_agent. Scrub PII and keys: llm_guard. Cheaper model routing: llm_router. Logging: llm_log plus llm_trace. Batch prompts: llm_batch. Tests without network: llm_mock. Images: llm_vision. Audio: llm_audio. Green headers are offline; orange ones need libcurl. -| Example | Shows | -|---|---| -| [cache.cpp](examples/offline/cache.cpp) | LRU cache: case-insensitive hits, evictions, stats | -| [cost.cpp](examples/offline/cost.cpp) | Price one prompt across the built-in models, block a call over budget | -| [guard.cpp](examples/offline/guard.cpp) | Find and scrub PII and API keys, score prompt injection | -| [format.cpp](examples/offline/format.cpp) | Validate JSON against a schema and re-prompt until it conforms | -| [json.cpp](examples/offline/json.cpp) | Build a request body, read a response, reject bad input | -| [compress.cpp](examples/offline/compress.cpp) | Keep a long chat inside a token budget with a sliding window | +Green headers use only the C++ standard library. Orange ones call the OpenAI or Anthropic API and need libcurl. All 26, with one line each: [docs/REFERENCE.md](docs/REFERENCE.md). -```bash -curl -fsSLO https://raw.githubusercontent.com/Mattbusel/llm-guard/main/include/llm_guard.hpp -g++ -std=c++17 -I. examples/offline/guard.cpp -o guard && ./guard -``` +## How it works + + + + Four steps. 1: curl llm_cache.hpp into your project. 2: the header's lines 1 to 85 are declarations, lines 86 to 210 are the implementation behind #ifdef LLM_CACHE_IMPLEMENTATION. 3: cache.cpp defines LLM_CACHE_IMPLEMENTATION before including it and gets the code; any other file just includes it. 4: cl compiles cache.cpp and the program prints one cache hit, four misses and two evictions. + -## Using several together +Every header follows this stb-style pattern, so once you have used one you have used all 26. Using several at once? Give each implementation its own `.cpp`: [docs/USING-SEVERAL.md](docs/USING-SEVERAL.md). -**Give each implementation its own `.cpp` file.** Several headers use the same internal helper names (for example `llm::detail::json_escape`), so defining two `*_IMPLEMENTATION` macros in one translation unit can fail to compile (llm-log with llm-stream is one such pair). In separate translation units they link together fine. As a check, all 26 implementations, each in its own `.cpp`, were compiled and linked into a single binary with MSVC 19.44 and libcurl on 2026-09-25. +## Examples (real output, no API key) -```cpp -// llm_impl_log.cpp -#define LLM_LOG_IMPLEMENTATION -#include "llm_log.hpp" +These are real programs in [`examples/offline`](examples/offline). CI rebuilds them with g++ against each library's current header on every push and fails if the output below changes. -// llm_impl_retry.cpp -#define LLM_RETRY_IMPLEMENTATION -#include "llm_retry.hpp" +**Scrub personal data and API keys before a prompt leaves your app** ([guard.cpp](examples/offline/guard.cpp), llm-guard): -// llm_impl_stream.cpp -#define LLM_STREAM_IMPLEMENTATION -#include "llm_stream.hpp" -``` + + + guard.cpp scans a prompt for an email, a card number and an API key, scores it 0.75 for injection and prints the scrubbed text + -Then use them together anywhere. This streams a completion, retries it on failure and writes a JSONL log line: - -```cpp -// main.cpp -#include "llm_log.hpp" -#include "llm_retry.hpp" -#include "llm_stream.hpp" -#include -#include - -int main() { - const char* key = std::getenv("OPENAI_API_KEY"); - if (!key) { std::cerr << "set OPENAI_API_KEY\n"; return 1; } - - llm::Config cfg; - cfg.api_key = key; - cfg.model = "gpt-4o-mini"; - const std::string prompt = "Explain backpressure in one paragraph."; - - llm::Logger logger(llm::LogConfig{"calls.jsonl"}); - llm::Logger::ScopedCall call(logger, cfg.model, prompt); // written on scope exit - - auto result = llm::with_retry([&]() -> std::string { - std::string text, error; - llm::stream(prompt, cfg, - [&](std::string_view tok) { std::cout << tok << std::flush; text += tok; }, - nullptr, - [&](std::string_view err) { error = err; }); - if (!error.empty()) throw llm::LLMError{0, error, true}; // retry - return text; - }); - - call.set_response(result.value); - std::cout << "\n(" << result.attempts_used << " attempt(s))\n"; -} +**Price one prompt across models, and refuse a call that would cost more than a cent** ([cost.cpp](examples/offline/cost.cpp), llm-cost): + +```text +C:\demo> cl /nologo /std:c++17 /EHsc cost.cpp && cost.exe +cost.cpp +gpt-6-luna 4080 tokens 0.0408¢ +gpt-4o-mini 4080 tokens 0.0612¢ +claude-haiku-4-5 4080 tokens 0.4080¢ +gpt-6-sol 4080 tokens 0.8160¢ +claude-sonnet-5 4080 tokens 0.8160¢ +gpt-4o 4080 tokens $0.0102 +claude-sonnet-4-5 4080 tokens $0.0122 +claude-opus-5-5 4080 tokens $0.0163 +claude-opus-4-5 4080 tokens $0.0204 +gpt-6-astra 4080 tokens $0.0408 +gpt-4-turbo 4080 tokens $0.0408 +claude-fable-5-1 4080 tokens $0.0408 + +blocked: Budget exceeded: estimated $0.0204 > limit $0.0100 (4080 tokens on claude-opus-4-5) ``` -```bash -g++ -std=c++17 -O2 -Ithird_party main.cpp llm_impl_log.cpp llm_impl_retry.cpp llm_impl_stream.cpp -lcurl -o app -``` +**Get schema-valid JSON, re-prompting until the model complies** ([format.cpp](examples/offline/format.cpp), llm-format; a stand-in lambda plays the model): -An offline pipeline needs no key and no network: clean a document with llm-parse, rank passages with llm-rank's BM25, and render the final prompt with llm-template. - -```cpp -#include "llm_parse.hpp" -#include "llm_rank.hpp" -#include "llm_template.hpp" -#include - -int main() { - std::string doc = llm::strip_html( - "

Deploying

Run make release to build the binary.

" - "

Copy config.yaml next to the binary.

Our office is in Berlin.

"); - llm::ChunkConfig cc; - cc.chunk_size = 60; - cc.overlap = 0; - auto passages = llm::chunk(doc, cc); - - std::string question = "how do I build the binary"; - auto ranked = llm::rerank_local(question, passages); - - llm::Template prompt("Answer using only this context:\n" - "{{#ctx}}- {{text}}\n{{/ctx}}\nQuestion: {{q}}\n"); - llm::TemplateContext ctx; - ctx.vars["q"] = question; - for (size_t i = 0; i < ranked.size() && i < 2; ++i) - ctx.lists["ctx"].push_back({{"text", ranked[i].passage}}); - - std::cout << prompt.render(ctx); +```text +valid: yes after 2 attempt(s) +{ + "priority": 1, + "tags": [ + "auth" + ], + "title": "Login fails" } +error: Field "title" has wrong type: expected string +error: Missing required field: "priority" +error: Missing required field: "tags" ``` -## Requirements +Also in the folder: [cache.cpp](examples/offline/cache.cpp) (the output in the diagram above), [json.cpp](examples/offline/json.cpp) and [compress.cpp](examples/offline/compress.cpp). Compiled with MSVC 19.44 on 2026-09-28; token counts in llm-cost are approximations. -| | | -|---|---| -| Language | C++17 or later | -| Compilers | GCC, Clang, MSVC. Each library repo builds its examples in CI with CMake. | -| Network libraries | libcurl: preinstalled on macOS, `apt install libcurl4-openssl-dev` on Debian/Ubuntu, `vcpkg install curl` on Windows | -| Providers | OpenAI-compatible chat, embeddings, audio and fine-tuning endpoints; Anthropic Messages API in llm-stream and llm-vision | +## Use it in 3 steps -## Status +1. **Copy** the header into your project (from the zip, or `curl -fsSLO` as above). +2. **Turn on the code** in exactly one `.cpp`: + ```cpp + #define LLM_CACHE_IMPLEMENTATION + #include "llm_cache.hpp" + ``` + Every other file just writes `#include "llm_cache.hpp"`. +3. **Compile as C++17**: `g++ -std=c++17 main.cpp` or `cl /std:c++17 /EHsc main.cpp`. Headers marked libcurl also need `-lcurl` (preinstalled on macOS, `apt install libcurl4-openssl-dev`, `vcpkg install curl`). -These are small, focused libraries, not a full SDK. The HTTP code targets the public OpenAI and Anthropic endpoints and uses hand-written JSON handling, and token counts in llm-cost are approximations. Issues and pull requests are welcome in the individual repositories. +## Documentation -## Related - -- [LLMTokenStreamQuantEngine](https://github.com/Mattbusel/LLMTokenStreamQuantEngine): C++20 engine that turns streaming LLM tokens into trade signals. +| | | +|---|---| +| [Catalogue site](https://mattbusel.github.io/llm-cpp/) | Filter all 26, real output, generated install commands | +| [docs/REFERENCE.md](docs/REFERENCE.md) | Every library with what it does and what it needs, requirements, status | +| [docs/USING-SEVERAL.md](docs/USING-SEVERAL.md) | Combining headers: a stream + retry + log program, an offline RAG pipeline | +| [examples/offline](examples/offline) | Six programs that need no API key, with their committed output | +| [Releases](https://github.com/Mattbusel/llm-cpp/releases) | The headers zip, built and link-checked by CI | +Each library lives in its own repo (`github.com/Mattbusel/llm-`); issues and pull requests are welcome there. ## Hire the author diff --git a/docs/REFERENCE.md b/docs/REFERENCE.md new file mode 100644 index 0000000..09db234 --- /dev/null +++ b/docs/REFERENCE.md @@ -0,0 +1,94 @@ +# llm-cpp reference + +Everything about the 26 libraries in one place. Back to the [README](../README.md) or the [catalogue site](https://mattbusel.github.io/llm-cpp/). + +## By task + +| I want to... | Use | +|---|---| +| Call a model and stream tokens | [llm-stream](https://github.com/Mattbusel/llm-stream) | +| Build a chatbot with memory | [llm-chat](https://github.com/Mattbusel/llm-chat) + [llm-retry](https://github.com/Mattbusel/llm-retry) | +| Answer questions over my documents | [llm-parse](https://github.com/Mattbusel/llm-parse) + [llm-embed](https://github.com/Mattbusel/llm-embed) or [llm-rag](https://github.com/Mattbusel/llm-rag) + [llm-rank](https://github.com/Mattbusel/llm-rank) | +| Get valid JSON back every time | [llm-format](https://github.com/Mattbusel/llm-format) + [llm-json](https://github.com/Mattbusel/llm-json) | +| Let the model call my C++ functions | [llm-agent](https://github.com/Mattbusel/llm-agent) | +| Know what my calls cost and where time goes | [llm-cost](https://github.com/Mattbusel/llm-cost) + [llm-log](https://github.com/Mattbusel/llm-log) + [llm-trace](https://github.com/Mattbusel/llm-trace) | +| Unit-test LLM code without the network | [llm-mock](https://github.com/Mattbusel/llm-mock) | + +## The libraries + +"libcurl" means the implementation makes HTTPS calls (OpenAI and/or Anthropic APIs). "none" means it is fully offline and uses only the standard library. + +### Core + +| Library | What it does | Needs | +|---|---|---| +| [llm-stream](https://github.com/Mattbusel/llm-stream) | Stream OpenAI and Anthropic chat responses token by token over SSE | libcurl | +| [llm-retry](https://github.com/Mattbusel/llm-retry) | Exponential backoff with jitter, provider failover and a circuit breaker | none | +| [llm-cost](https://github.com/Mattbusel/llm-cost) | Approximate token counts and cost estimates for built-in OpenAI and Anthropic models, budget checks | none | +| [llm-cache](https://github.com/Mattbusel/llm-cache) | LRU response cache with TTL and hit/miss stats, so identical prompts skip the API | none | +| [llm-format](https://github.com/Mattbusel/llm-format) | Define a schema, validate model JSON against it, and re-prompt until the output conforms | none | +| [llm-json](https://github.com/Mattbusel/llm-json) | Small JSON parser and builder for request bodies and model output | none | + +### Data and retrieval + +| Library | What it does | Needs | +|---|---|---| +| [llm-parse](https://github.com/Mattbusel/llm-parse) | Strip HTML and markdown, extract titles, links, headings and code blocks, chunk text | none | +| [llm-embed](https://github.com/Mattbusel/llm-embed) | OpenAI embeddings, cosine/dot/euclidean similarity and a small on-disk vector store | libcurl | +| [llm-rag](https://github.com/Mattbusel/llm-rag) | End-to-end RAG: chunk, embed, persist an index, retrieve top-k and answer | libcurl | +| [llm-rank](https://github.com/Mattbusel/llm-rank) | Rerank passages with offline BM25, LLM relevance scoring, or a hybrid of both | libcurl (linked; BM25 itself is offline) | +| [llm-compress](https://github.com/Mattbusel/llm-compress) | Shrink conversation history: head/tail/smart truncation, sliding window, LLM summary | none (libcurl only with `LLM_COMPRESS_SUMMARIZE`) | +| [llm-batch](https://github.com/Mattbusel/llm-batch) | Run a JSONL file of prompts through a thread pool with rate limiting and resumable checkpoints | libcurl | + +### Operations and testing + +| Library | What it does | Needs | +|---|---|---| +| [llm-log](https://github.com/Mattbusel/llm-log) | Structured JSONL log of every call with latency, tokens and cost, plus query and summary | none | +| [llm-trace](https://github.com/Mattbusel/llm-trace) | RAII spans with parent/child nesting, token and cost attributes, OTLP-style JSON export | none | +| [llm-pool](https://github.com/Mattbusel/llm-pool) | Worker pool with priority queue and requests-per-minute and tokens-per-minute limits | none | +| [llm-mock](https://github.com/Mattbusel/llm-mock) | Fake LLM with scripted, pattern, random or echo responses, simulated latency and streaming | none | +| [llm-eval](https://github.com/Mattbusel/llm-eval) | Run a prompt N times, measure consistency, compare models or prompts, score responses | libcurl | +| [llm-ab](https://github.com/Mattbusel/llm-ab) | A/B test prompts or models with Welch's t-test, Cohen's d and custom scorers | libcurl | + +### Application features + +| Library | What it does | Needs | +|---|---|---| +| [llm-chat](https://github.com/Mattbusel/llm-chat) | Multi-turn conversation with token-budget trimming, pinned system prompt, save and restore | libcurl | +| [llm-agent](https://github.com/Mattbusel/llm-agent) | Tool-calling agent loop: register C++ lambdas as tools and let the model call them | libcurl | +| [llm-vision](https://github.com/Mattbusel/llm-vision) | Send images (file or URL) plus a prompt to OpenAI or Anthropic vision models | libcurl | +| [llm-template](https://github.com/Mattbusel/llm-template) | Mustache-style prompt templates with loops, conditionals and token-budget truncation | none | +| [llm-router](https://github.com/Mattbusel/llm-router) | Pick a model per prompt from a complexity score and a cost, latency, quality or budget strategy | none | +| [llm-guard](https://github.com/Mattbusel/llm-guard) | Detect and scrub PII (email, phone, SSN, card numbers, API keys) and score prompt-injection risk | none | +| [llm-audio](https://github.com/Mattbusel/llm-audio) | Whisper transcription and translation, and text-to-speech, via the OpenAI API | libcurl | +| [llm-finetune](https://github.com/Mattbusel/llm-finetune) | OpenAI fine-tuning lifecycle: write JSONL, upload, create, poll, cancel, list models | libcurl | + +## Why single-header + + + + The llm-cache repo on the left with only include/llm_cache.hpp highlighted, copied with curl into third_party/ of your project on the right + + +- **Nothing to install.** `curl -O` one file, `#include` it. It works the same with CMake, Make, Bazel, MSBuild or a one-line `g++` command. +- **You can read all of it.** Each library is 210 to 572 lines; all 26 together are 8,948 (counted 2026-09-28). When something misbehaves you open one file, not a dependency tree. +- **You pay for what you use.** Need retries and a cache? Take two headers. Nothing else is pulled in, and the offline ones add no link dependencies at all. +- **Easy to vendor.** Copy the headers into `third_party/`, pin them in your own repo, patch them if you need to. No version resolver involved. + +## Requirements + +| | | +|---|---| +| Language | C++17 or later | +| Compilers | GCC, Clang, MSVC. Each library repo builds its examples in CI with CMake. | +| Network libraries | libcurl: preinstalled on macOS, `apt install libcurl4-openssl-dev` on Debian/Ubuntu, `vcpkg install curl` on Windows | +| Providers | OpenAI-compatible chat, embeddings, audio and fine-tuning endpoints; Anthropic Messages API in llm-stream and llm-vision | + +## Status + +These are small, focused libraries, not a full SDK. The HTTP code targets the public OpenAI and Anthropic endpoints and uses hand-written JSON handling, and token counts in llm-cost are approximations. Issues and pull requests are welcome in the individual repositories. + +## Related + +- [LLMTokenStreamQuantEngine](https://github.com/Mattbusel/LLMTokenStreamQuantEngine): C++20 engine that turns streaming LLM tokens into trade signals. diff --git a/docs/USING-SEVERAL.md b/docs/USING-SEVERAL.md new file mode 100644 index 0000000..1358a28 --- /dev/null +++ b/docs/USING-SEVERAL.md @@ -0,0 +1,92 @@ +# Using several llm-cpp headers together + +Back to the [README](../README.md) or the [reference](REFERENCE.md). + + +**Give each implementation its own `.cpp` file.** Several headers use the same internal helper names (for example `llm::detail::json_escape`), so defining two `*_IMPLEMENTATION` macros in one translation unit can fail to compile (llm-log with llm-stream is one such pair). In separate translation units they link together fine. As a check, all 26 implementations, each in its own `.cpp`, were compiled and linked into a single binary with MSVC 19.44 and libcurl on 2026-09-25. CI repeats the same check with g++ and libcurl on every push (the `link-all` job in [ci.yml](../.github/workflows/ci.yml)). + +```cpp +// llm_impl_log.cpp +#define LLM_LOG_IMPLEMENTATION +#include "llm_log.hpp" + +// llm_impl_retry.cpp +#define LLM_RETRY_IMPLEMENTATION +#include "llm_retry.hpp" + +// llm_impl_stream.cpp +#define LLM_STREAM_IMPLEMENTATION +#include "llm_stream.hpp" +``` + +Then use them together anywhere. This streams a completion, retries it on failure and writes a JSONL log line: + +```cpp +// main.cpp +#include "llm_log.hpp" +#include "llm_retry.hpp" +#include "llm_stream.hpp" +#include +#include + +int main() { + const char* key = std::getenv("OPENAI_API_KEY"); + if (!key) { std::cerr << "set OPENAI_API_KEY\n"; return 1; } + + llm::Config cfg; + cfg.api_key = key; + cfg.model = "gpt-4o-mini"; + const std::string prompt = "Explain backpressure in one paragraph."; + + llm::Logger logger(llm::LogConfig{"calls.jsonl"}); + llm::Logger::ScopedCall call(logger, cfg.model, prompt); // written on scope exit + + auto result = llm::with_retry([&]() -> std::string { + std::string text, error; + llm::stream(prompt, cfg, + [&](std::string_view tok) { std::cout << tok << std::flush; text += tok; }, + nullptr, + [&](std::string_view err) { error = err; }); + if (!error.empty()) throw llm::LLMError{0, error, true}; // retry + return text; + }); + + call.set_response(result.value); + std::cout << "\n(" << result.attempts_used << " attempt(s))\n"; +} +``` + +```bash +g++ -std=c++17 -O2 -Ithird_party main.cpp llm_impl_log.cpp llm_impl_retry.cpp llm_impl_stream.cpp -lcurl -o app +``` + +An offline pipeline needs no key and no network: clean a document with llm-parse, rank passages with llm-rank's BM25, and render the final prompt with llm-template. + +```cpp +#include "llm_parse.hpp" +#include "llm_rank.hpp" +#include "llm_template.hpp" +#include + +int main() { + std::string doc = llm::strip_html( + "

Deploying

Run make release to build the binary.

" + "

Copy config.yaml next to the binary.

Our office is in Berlin.

"); + llm::ChunkConfig cc; + cc.chunk_size = 60; + cc.overlap = 0; + auto passages = llm::chunk(doc, cc); + + std::string question = "how do I build the binary"; + auto ranked = llm::rerank_local(question, passages); + + llm::Template prompt("Answer using only this context:\n" + "{{#ctx}}- {{text}}\n{{/ctx}}\nQuestion: {{q}}\n"); + llm::TemplateContext ctx; + ctx.vars["q"] = question; + for (size_t i = 0; i < ranked.size() && i < 2; ++i) + ctx.lists["ctx"].push_back({{"text", ranked[i].passage}}); + + std::cout << prompt.render(ctx); +} +``` diff --git a/docs/img/how-it-works-dark.svg b/docs/img/how-it-works-dark.svg new file mode 100644 index 0000000..6740bb7 --- /dev/null +++ b/docs/img/how-it-works-dark.svg @@ -0,0 +1,67 @@ +How a single-header llm-cpp library goes into your projectFour steps. 1: download llm_cache.hpp into your project. 2: the header holds declarations in lines 1 to 85 and the implementation in lines 86 to 210 behind #ifdef LLM_CACHE_IMPLEMENTATION. 3: exactly one .cpp file defines LLM_CACHE_IMPLEMENTATION before including it and gets the code; every other file just includes it. 4: compile with any C++17 compiler and run; the real output shows one cache hit, four misses and two evictions. + + + +1 +Copy one file into your project +No package manager. Any build system. + + + + +$ curl -fsSLO \ + https://raw.githubusercontent.com/Mattbusel/llm-cache/main/include/llm_cache.hpp + + + +2 +The header has two halves + +llm_cache.hpp +210 lines + +lines 1-85 +struct CacheConfig; class ResponseCache; + +lines 86-210 +#ifdef LLM_CACHE_IMPLEMENTATION ... #endif +Top half: declarations. +Every file that includes +the header sees them. + +Bottom half: the code. +It is only compiled where +you ask for it. + + + +3 +Turn the code on in exactly one .cpp + +cache.cpp +#define LLM_CACHE_IMPLEMENTATION +#include "llm_cache.hpp" +gets both halves: compiled once + +any_other_file.cpp +#include "llm_cache.hpp" +declarations only: no duplicate code + + + +4 +Compile and run + + + + +> cl /nologo /std:c++17 /EHsc cache.cpp && cache.exe +cache.cpp +What is RAII? -> answer #1 +what is raii? -> answer #1 +Explain move semantics -> answer #2 +What is SFINAE? -> answer #3 +What is RAII? -> answer #4 + +api calls 4 | hits 1 | misses 4 | evictions 2 + \ No newline at end of file diff --git a/docs/img/how-it-works-light.svg b/docs/img/how-it-works-light.svg new file mode 100644 index 0000000..3e5b284 --- /dev/null +++ b/docs/img/how-it-works-light.svg @@ -0,0 +1,67 @@ +How a single-header llm-cpp library goes into your projectFour steps. 1: download llm_cache.hpp into your project. 2: the header holds declarations in lines 1 to 85 and the implementation in lines 86 to 210 behind #ifdef LLM_CACHE_IMPLEMENTATION. 3: exactly one .cpp file defines LLM_CACHE_IMPLEMENTATION before including it and gets the code; every other file just includes it. 4: compile with any C++17 compiler and run; the real output shows one cache hit, four misses and two evictions. + + + +1 +Copy one file into your project +No package manager. Any build system. + + + + +$ curl -fsSLO \ + https://raw.githubusercontent.com/Mattbusel/llm-cache/main/include/llm_cache.hpp + + + +2 +The header has two halves + +llm_cache.hpp +210 lines + +lines 1-85 +struct CacheConfig; class ResponseCache; + +lines 86-210 +#ifdef LLM_CACHE_IMPLEMENTATION ... #endif +Top half: declarations. +Every file that includes +the header sees them. + +Bottom half: the code. +It is only compiled where +you ask for it. + + + +3 +Turn the code on in exactly one .cpp + +cache.cpp +#define LLM_CACHE_IMPLEMENTATION +#include "llm_cache.hpp" +gets both halves: compiled once + +any_other_file.cpp +#include "llm_cache.hpp" +declarations only: no duplicate code + + + +4 +Compile and run + + + + +> cl /nologo /std:c++17 /EHsc cache.cpp && cache.exe +cache.cpp +What is RAII? -> answer #1 +what is raii? -> answer #1 +Explain move semantics -> answer #2 +What is SFINAE? -> answer #3 +What is RAII? -> answer #4 + +api calls 4 | hits 1 | misses 4 | evictions 2 + \ No newline at end of file diff --git a/docs/img/picker-dark.svg b/docs/img/picker-dark.svg new file mode 100644 index 0000000..6b00669 --- /dev/null +++ b/docs/img/picker-dark.svg @@ -0,0 +1,115 @@ +Which llm-cpp header solves which problemSixteen common tasks, each mapped to the one or two single-header files that do it. Green headers need only the C++ standard library; orange headers also need libcurl. + +I want to... +Copy this file + +offline, std lib only + +needs libcurl + + +Stream a model's reply token by token + + + +llm_stream.hpp +Retry failed calls, fail over to another provider + + + +llm_retry.hpp + +Stop paying twice for the same prompt + + + +llm_cache.hpp +Know what a prompt costs before you send it + + + +llm_cost.hpp + +Get valid JSON back every time + + + +llm_format.hpp ++ + + +llm_json.hpp +Build a chatbot that remembers the conversation + + + +llm_chat.hpp ++ + + +llm_retry.hpp + +Answer questions over my own documents + + + +llm_rag.hpp ++ + + +llm_rank.hpp +Keep a long chat under the token limit + + + +llm_compress.hpp + +Let the model call my C++ functions + + + +llm_agent.hpp +Scrub emails, card numbers and API keys + + + +llm_guard.hpp + +Send easy prompts to a cheaper model + + + +llm_router.hpp +Log every call with latency, tokens and cost + + + +llm_log.hpp ++ + + +llm_trace.hpp + +Run a file of prompts in parallel, resumably + + + +llm_batch.hpp +Unit-test LLM code without the network + + + +llm_mock.hpp + +Send images to a vision model + + + +llm_vision.hpp +Transcribe audio, or turn text into speech + + + +llm_audio.hpp +All 26 headers, including embed, eval, ab, pool, template and finetune: mattbusel.github.io/llm-cpp + \ No newline at end of file diff --git a/docs/img/picker-light.svg b/docs/img/picker-light.svg new file mode 100644 index 0000000..fe4c077 --- /dev/null +++ b/docs/img/picker-light.svg @@ -0,0 +1,115 @@ +Which llm-cpp header solves which problemSixteen common tasks, each mapped to the one or two single-header files that do it. Green headers need only the C++ standard library; orange headers also need libcurl. + +I want to... +Copy this file + +offline, std lib only + +needs libcurl + + +Stream a model's reply token by token + + + +llm_stream.hpp +Retry failed calls, fail over to another provider + + + +llm_retry.hpp + +Stop paying twice for the same prompt + + + +llm_cache.hpp +Know what a prompt costs before you send it + + + +llm_cost.hpp + +Get valid JSON back every time + + + +llm_format.hpp ++ + + +llm_json.hpp +Build a chatbot that remembers the conversation + + + +llm_chat.hpp ++ + + +llm_retry.hpp + +Answer questions over my own documents + + + +llm_rag.hpp ++ + + +llm_rank.hpp +Keep a long chat under the token limit + + + +llm_compress.hpp + +Let the model call my C++ functions + + + +llm_agent.hpp +Scrub emails, card numbers and API keys + + + +llm_guard.hpp + +Send easy prompts to a cheaper model + + + +llm_router.hpp +Log every call with latency, tokens and cost + + + +llm_log.hpp ++ + + +llm_trace.hpp + +Run a file of prompts in parallel, resumably + + + +llm_batch.hpp +Unit-test LLM code without the network + + + +llm_mock.hpp + +Send images to a vision model + + + +llm_vision.hpp +Transcribe audio, or turn text into speech + + + +llm_audio.hpp +All 26 headers, including embed, eval, ab, pool, template and finetune: mattbusel.github.io/llm-cpp + \ No newline at end of file diff --git a/docs/index.html b/docs/index.html index 545a29b..08a9306 100644 --- a/docs/index.html +++ b/docs/index.html @@ -4,12 +4,24 @@ llm-cpp: single-header C++ libraries for LLM features - + + + + - + + + + + + + +