diff --git a/README.md b/README.md index 42b771b..1dae574 100644 --- a/README.md +++ b/README.md @@ -47,6 +47,7 @@ Refer to the [Installation section](https://getdozer.io/docs/installation) for i | | [IMDB Analytics](./usecases/imdb-analytics) | Use Dozer to get interesting analytics using an IMDb dataset | | | Use Dozer to Instrument (Coming soon) | Combine Log data to get real time insights | | | Real Time Model Scoring (Coming soon) | Deploy trained models to get real time insights as APIs | +| | [LLM Vector Banking Chatbot](./usecases/llm-vector-banking-chatbot) | Dozer + LangChain + vector retrieval for card personalization (#1690) | | | | | | Client Libraries | [Dozer React Starter](./usecases/react/) | Instantly start building real time views using Dozer and React | | | [Ingest Polars/Pandas Dataframes](./client-samples/ingest-python-sample) | Instantly ingest Polars/Pandas dataframes using Arrow format and deploy APIs | diff --git a/usecases/llm-vector-banking-chatbot/.gitignore b/usecases/llm-vector-banking-chatbot/.gitignore new file mode 100644 index 0000000..a5f2b9f --- /dev/null +++ b/usecases/llm-vector-banking-chatbot/.gitignore @@ -0,0 +1,9 @@ +.dozer/ +__pycache__/ +*.pyc +.venv/ +venv/ +.env +.chroma/ +*.egg-info/ +.pytest_cache/ diff --git a/usecases/llm-vector-banking-chatbot/README.md b/usecases/llm-vector-banking-chatbot/README.md new file mode 100644 index 0000000..2dc5716 --- /dev/null +++ b/usecases/llm-vector-banking-chatbot/README.md @@ -0,0 +1,105 @@ +# Dozer + LLM + Vector DB + LangChain (Banking Chatbot) + +Sample for [getdozer/dozer#1690](https://github.com/getdozer/dozer/issues/1690) ($250 Algora). + +Inspired by the Dozer hyper-personalization banking chatbot write-up: unify customer +profiles, accounts, and transactions in Dozer, then pass that context into a LangChain +retrieval flow over credit-card products (vector store + optional LLM). + +## What this sample includes + +| Piece | Path | Role | +| --- | --- | --- | +| Dozer config | `dozer-config.yaml` | LocalStorage CSV sources + spend-profile SQL + REST endpoints | +| Seed data | `data/*/*.csv` | customers, accounts, transactions, card products | +| RAG stub | `app/rag_chatbot.py` | fixture mode (offline) + `--use-dozer` + optional `--llm` | +| Unit tests | `tests/test_rag_chatbot.py` | fixture load, segment gate, keyword/rerank | +| Deps | `app/requirements.txt` | LangChain / Chroma / OpenAI (optional) | +| Demo log | `demo-log.txt` | text capture of live Dozer REST + chatbot (no video yet) | + +## Prerequisites + +- **Dozer 0.3.x** on `PATH` for the live path (API-era binary with `endpoints`). + Dozer **≥0.4.0** removed built-in `endpoints` / `dozer-api` — that binary will reject + this config. See [`SAMPLE_NOTES.md`](./SAMPLE_NOTES.md). +- Python 3.10+ +- `protoc` (`protobuf-compiler`) for `dozer build` on 0.3.x +- Optional: `OPENAI_API_KEY` for the ChatOpenAI answer path + +## Quick start (offline fixture — no Dozer binary) + +```bash +cd usecases/llm-vector-banking-chatbot +python3 app/rag_chatbot.py \ + --customer C001 \ + --question "Which card is best for travel and dining rewards?" +``` + +Expected shape: a recommendation for **Dozer Travel Signature** grounded in C001's +travel/dining spend. + +Student example: + +```bash +python3 app/rag_chatbot.py \ + --customer C003 \ + --question "I need a no-fee student card for education" +``` + +## Live Dozer path (verified on 0.3.0) + +```bash +cd usecases/llm-vector-banking-chatbot +rm -rf .dozer +dozer --ignore-pipe build -c dozer-config.yaml +dozer --ignore-pipe run -c dozer-config.yaml +# elsewhere: +python3 -m venv .venv && .venv/bin/pip install -r app/requirements.txt +.venv/bin/python app/rag_chatbot.py --use-dozer --customer C001 \ + --question "Which card is best for travel and dining rewards?" +``` + +Dozer REST listens at `http://localhost:8080` (override with `--dozer-base` or +`DOZER_API_BASE`). Responses are bare JSON lists; the client filters by `customer_id`. + +Optional LLM polish: + +```bash +export OPENAI_API_KEY=sk-... +.venv/bin/python app/rag_chatbot.py --use-dozer --llm --customer C001 \ + --question "Which card is best for travel and dining rewards?" +``` + +## Tests + +```bash +cd usecases/llm-vector-banking-chatbot +python3 -m venv .venv && .venv/bin/pip install pytest httpx +.venv/bin/pytest -q tests/test_rag_chatbot.py +``` + +## APIs exposed (once Dozer 0.3.x is up) + +- `GET /customer_profiles` +- `GET /accounts` +- `GET /transactions` +- `GET /card_products` +- `GET /customer_spending_profile` — SQL aggregate used as chatbot context + +## Architecture (sample) + +``` +CSV seeds ──► Dozer LocalStorage ──► SQL spend profile ──► REST/gRPC APIs + │ + LangChain retriever (Chroma / FakeEmbeddings) │ + + customer-aware rerank ◄─────────────────────┘ + │ + deterministic answer or ChatOpenAI (--llm) +``` + +## Status / honesty + +See [`SAMPLE_NOTES.md`](./SAMPLE_NOTES.md). Live `dozer run` (0.3.0) + `--use-dozer` +chatbot verified; text demo log attached. **No payout claimed.** Demo *video* and +maintainer confirmation of bounty eligibility remain open. Crowded claim field +(`dozer-samples` #139–#145, etc.). diff --git a/usecases/llm-vector-banking-chatbot/SAMPLE_NOTES.md b/usecases/llm-vector-banking-chatbot/SAMPLE_NOTES.md new file mode 100644 index 0000000..815207b --- /dev/null +++ b/usecases/llm-vector-banking-chatbot/SAMPLE_NOTES.md @@ -0,0 +1,62 @@ +# SAMPLE_NOTES — getdozer/dozer#1690 + +**Branch:** `bounty/1690-llm-vector-banking-scaffold` +**PR:** https://github.com/getdozer/dozer-samples/pull/144 +**Issue:** https://github.com/getdozer/dozer/issues/1690 +**Bounty:** $250 Algora — Dozer + LLM + Vector DB + LangChain sample + +## Do NOT invent payout success + +No maintainer acceptance, no Algora payout confirmation. Multiple competing open PRs +(`dozer-samples` #139–#145 and others). Algora may already show claims/completed — +treat reward eligibility as **uncertain**. + +## What works now (verified locally) + +| Check | Result | +| --- | --- | +| Offline fixture chatbot | OK (C001→Travel Signature, C003→Campus Starter) | +| `dozer build/run` **0.3.0** against `dozer-config.yaml` | OK (`--ignore-pipe`; needs `protoc` + OpenSSL 1.1) | +| REST `GET /customer_profiles`, `/customer_spending_profile`, `/card_products` | 200, bare JSON lists | +| `rag_chatbot.py --use-dozer` | OK after client-side customer filter fix | +| Unit tests `tests/test_rag_chatbot.py` | 7 passed | +| Text demo log | `demo-log.txt` | +| Demo **video** | **Still missing** | + +### Commands that succeeded + +```text +dozer --ignore-pipe build -c dozer-config.yaml # dozer 0.3.0 +dozer --ignore-pipe run -c dozer-config.yaml +python3 app/rag_chatbot.py --use-dozer --customer C001 --question "Which card is best for travel and dining rewards?" +python3 app/rag_chatbot.py --use-dozer --customer C003 --question "I need a no-fee student card for education" +pytest -q tests/test_rag_chatbot.py # 7 passed +``` + +## Config / client fixes in this follow-up + +1. Explicit `api.rest` / `api.grpc` host+port (8080 / 50051). +2. Composite PK on `customer_spending_profile`: `customer_id` + `top_category` + (matches `GROUP BY customer_id, category`). +3. `--use-dozer` no longer trusts `?customer_id=` (0.3 ignores it and would always + pick the first profile); always filter client-side on bare lists. +4. Document that **Dozer ≥0.4.0 rejects `endpoints`** (`unknown field endpoints`; + sinks-only era). This sample targets the API-era 0.3.x binary that matches the + historical LocalStorage samples in this repo. + +## Remaining gaps + +1. **No demo video** (Algora often wants ≤30s screen capture). +2. **Architecture drift:** issue comments note current main may prefer + Dozer→sink→app API. This PR intentionally stays on 0.3-style REST endpoints to + match existing `dozer-samples` patterns; maintainers may ask for a sink rewrite. +3. **Blog URL** `https://getdozer.io/blog/llm-chatbot` still 404 when last checked. +4. **Crowded claims** — many open PRs; confirm with maintainers before banking on $250. +5. Optional Chroma/LangChain path needs `pip install -r app/requirements.txt`; tests + and live demo use keyword retrieve + fixture/Dozer context without Chroma. + +## Confidence + +**~0.70** that this is a credible, runnable sample PR after live verify + tests. +**~0.20** that it wins $250 given competition + possible architecture preference for +sinks — **do not bank on payout**. diff --git a/usecases/llm-vector-banking-chatbot/app/__init__.py b/usecases/llm-vector-banking-chatbot/app/__init__.py new file mode 100644 index 0000000..c0bf531 --- /dev/null +++ b/usecases/llm-vector-banking-chatbot/app/__init__.py @@ -0,0 +1 @@ +"""LangChain RAG stub for the Dozer banking chatbot sample.""" diff --git a/usecases/llm-vector-banking-chatbot/app/rag_chatbot.py b/usecases/llm-vector-banking-chatbot/app/rag_chatbot.py new file mode 100644 index 0000000..3a237b6 --- /dev/null +++ b/usecases/llm-vector-banking-chatbot/app/rag_chatbot.py @@ -0,0 +1,276 @@ +#!/usr/bin/env python3 +""" +Dozer + LangChain RAG stub for credit-card personalization. + +Modes +----- +* fixture (default): reads local CSVs — no Dozer binary, no API key required +* --use-dozer: fetch customer context from Dozer REST (http://localhost:8080) +* --llm: optional ChatOpenAI final answer when OPENAI_API_KEY is set + +Fixture mode needs no Dozer binary. Live mode (--use-dozer) was verified +against Dozer 0.3.0 REST (bare JSON lists; client-side customer filter). +""" + +from __future__ import annotations + +import argparse +import csv +import os +import sys +from collections import defaultdict +from dataclasses import dataclass +from pathlib import Path + +SAMPLE_ROOT = Path(__file__).resolve().parents[1] +DATA_DIR = SAMPLE_ROOT / "data" +DEFAULT_DOZER = os.environ.get("DOZER_API_BASE", "http://localhost:8080") + + +@dataclass +class CustomerContext: + customer_id: str + full_name: str + segment: str + city: str + spend_by_category: dict[str, float] + total_spend: float + + def summary(self) -> str: + top = sorted(self.spend_by_category.items(), key=lambda x: -x[1])[:3] + top_s = ", ".join(f"{c}=${v:,.0f}" for c, v in top) or "n/a" + return ( + f"{self.full_name} ({self.customer_id}), segment={self.segment}, " + f"city={self.city}, total_spend=${self.total_spend:,.0f}, top=[{top_s}]" + ) + + +def _csv_rows(table: str) -> list[dict[str, str]]: + path = DATA_DIR / table / f"{table}.csv" + with path.open(newline="", encoding="utf-8") as fh: + return list(csv.DictReader(fh)) + + +def load_fixture_context(customer_id: str) -> CustomerContext: + profiles = {r["customer_id"]: r for r in _csv_rows("customer_profiles")} + if customer_id not in profiles: + raise SystemExit(f"Unknown customer_id={customer_id}") + p = profiles[customer_id] + spend: dict[str, float] = defaultdict(float) + for txn in _csv_rows("transactions"): + if txn["customer_id"] == customer_id: + spend[txn["category"]] += float(txn["amount"]) + return CustomerContext( + customer_id=customer_id, + full_name=p["full_name"], + segment=p["segment"], + city=p["city"], + spend_by_category=dict(spend), + total_spend=sum(spend.values()), + ) + + +def load_dozer_context(customer_id: str, base: str) -> CustomerContext: + """Fetch unified profile from Dozer REST once the binary is running. + + Dozer 0.3 REST returns a bare JSON list and does not honor simple + ``?customer_id=`` query filters, so we always filter client-side. + """ + try: + import httpx + except ImportError as exc: + raise SystemExit("httpx required for --use-dozer; pip install httpx") from exc + + def get(path: str) -> list[dict]: + r = httpx.get(f"{base.rstrip('/')}{path}", timeout=10.0) + r.raise_for_status() + payload = r.json() + # Dozer 0.3 returns a bare list; older docs mentioned {"data": [...]} + if isinstance(payload, dict): + return payload.get("data") or payload.get("records") or [] + if isinstance(payload, list): + return payload + return [] + + profiles = [p for p in get("/customer_profiles") if p.get("customer_id") == customer_id] + if not profiles: + raise SystemExit(f"Dozer returned no profile for {customer_id}") + p = profiles[0] + spend_rows = [ + row + for row in get("/customer_spending_profile") + if row.get("customer_id") == customer_id + ] + spend = {row["top_category"]: float(row["total_spend"]) for row in spend_rows} + return CustomerContext( + customer_id=customer_id, + full_name=p.get("full_name", customer_id), + segment=p.get("segment", "unknown"), + city=p.get("city", ""), + spend_by_category=spend, + total_spend=sum(spend.values()), + ) + + +def load_card_docs() -> list[dict[str, str]]: + return _csv_rows("card_products") + + +def segment_ok(customer_segment: str, min_segment: str) -> bool: + order = ["student", "mass_market", "mass_affluent", "premium"] + try: + return order.index(customer_segment) >= order.index(min_segment) + except ValueError: + return True + + +def keyword_retrieve(question: str, cards: list[dict[str, str]], k: int = 3) -> list[dict[str, str]]: + q = question.lower() + scored: list[tuple[int, dict[str, str]]] = [] + for card in cards: + blob = " ".join( + [ + card["name"], + card["reward_categories"], + card["perk_summary"], + card["tags"], + ] + ).lower() + score = sum(1 for token in q.replace("?", " ").split() if token and token in blob) + scored.append((score, card)) + scored.sort(key=lambda x: -x[0]) + return [c for s, c in scored[:k] if s > 0] or [c for _, c in scored[:k]] + + +def vector_retrieve(question: str, cards: list[dict[str, str]], k: int = 3) -> list[dict[str, str]]: + """LangChain + Chroma path when deps are installed; else keyword fallback.""" + try: + from langchain_community.embeddings import FakeEmbeddings + from langchain_community.vectorstores import Chroma + from langchain_core.documents import Document + except Exception: + return keyword_retrieve(question, cards, k=k) + + docs = [ + Document( + page_content=( + f"{c['name']}. Rewards: {c['reward_categories']}. " + f"{c['perk_summary']}. Tags: {c['tags']}" + ), + metadata={"product_id": c["product_id"], "name": c["name"]}, + ) + for c in cards + ] + store = Chroma.from_documents(docs, embedding=FakeEmbeddings(size=32)) + hits = store.similarity_search(question, k=k) + by_id = {c["product_id"]: c for c in cards} + out = [] + for h in hits: + pid = h.metadata.get("product_id") + if pid in by_id: + out.append(by_id[pid]) + return out or keyword_retrieve(question, cards, k=k) + + +def rerank_for_customer( + ctx: CustomerContext, + candidates: list[dict[str, str]], + question: str = "", +) -> list[dict[str, str]]: + top_cats = {c for c, _ in sorted(ctx.spend_by_category.items(), key=lambda x: -x[1])[:3]} + q_tokens = {t for t in question.lower().replace("?", " ").split() if len(t) > 2} + + def score(card: dict[str, str]) -> float: + if not segment_ok(ctx.segment, card["min_segment"]): + return -1.0 + rewards = set(card["reward_categories"].split("|")) + overlap = float(len(top_cats & rewards)) + blob = f"{card['reward_categories']}|{card['tags']}|{card['name']}".lower() + q_hit = sum(1.0 for t in q_tokens if t in blob) + fee = float(card["annual_fee"]) + fee_penalty = 1.0 if fee >= 100000 else (0.2 if fee >= 40000 else 0.0) + return overlap + 0.75 * q_hit - fee_penalty + + ranked = sorted(candidates, key=score, reverse=True) + return [c for c in ranked if score(c) >= 0] or ranked + + +def deterministic_answer(ctx: CustomerContext, question: str, picks: list[dict[str, str]]) -> str: + if not picks: + return ( + f"No eligible card for {ctx.full_name} given segment={ctx.segment}. " + "Ask about everyday cashback or student products." + ) + best = picks[0] + return ( + f"For {ctx.summary()}, given question «{question}», " + f"recommend **{best['name']}** ({best['product_id']}): {best['perk_summary']} " + f"(annual fee {best['annual_fee']} ARS)." + ) + + +def maybe_llm_answer(ctx: CustomerContext, question: str, picks: list[dict[str, str]]) -> str | None: + if not os.environ.get("OPENAI_API_KEY"): + return None + try: + from langchain_openai import ChatOpenAI + from langchain_core.messages import HumanMessage, SystemMessage + except Exception: + return None + + catalog = "\n".join( + f"- {c['name']}: {c['perk_summary']} (fee={c['annual_fee']})" for c in picks + ) + llm = ChatOpenAI(model=os.environ.get("OPENAI_MODEL", "gpt-4o-mini"), temperature=0) + msg = llm.invoke( + [ + SystemMessage( + content=( + "You are a bank product assistant. Use only the provided customer " + "context and card candidates. Be concise." + ) + ), + HumanMessage( + content=( + f"Customer context:\n{ctx.summary()}\n\n" + f"Candidates:\n{catalog}\n\nQuestion: {question}" + ) + ), + ] + ) + return str(msg.content) + + +def run(customer_id: str, question: str, use_dozer: bool, use_llm: bool, dozer_base: str) -> str: + ctx = ( + load_dozer_context(customer_id, dozer_base) + if use_dozer + else load_fixture_context(customer_id) + ) + cards = load_card_docs() + candidates = vector_retrieve(question, cards, k=4) + picks = rerank_for_customer(ctx, candidates, question=question) + if use_llm: + llm_text = maybe_llm_answer(ctx, question, picks) + if llm_text: + return llm_text + return deterministic_answer(ctx, question, picks) + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--customer", default="C001") + parser.add_argument( + "--question", + default="Which card is best for travel and dining rewards?", + ) + parser.add_argument("--use-dozer", action="store_true") + parser.add_argument("--llm", action="store_true", help="Try ChatOpenAI if key present") + parser.add_argument("--dozer-base", default=DEFAULT_DOZER) + args = parser.parse_args(argv) + print(run(args.customer, args.question, args.use_dozer, args.llm, args.dozer_base)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/usecases/llm-vector-banking-chatbot/app/requirements.txt b/usecases/llm-vector-banking-chatbot/app/requirements.txt new file mode 100644 index 0000000..392fffc --- /dev/null +++ b/usecases/llm-vector-banking-chatbot/app/requirements.txt @@ -0,0 +1,7 @@ +# Optional live path (install when running against Dozer + OpenAI) +langchain>=0.2.0 +langchain-community>=0.2.0 +langchain-openai>=0.1.0 +chromadb>=0.5.0 +httpx>=0.27.0 +openai>=1.30.0 diff --git a/usecases/llm-vector-banking-chatbot/data/accounts/accounts.csv b/usecases/llm-vector-banking-chatbot/data/accounts/accounts.csv new file mode 100644 index 0000000..5f93ecc --- /dev/null +++ b/usecases/llm-vector-banking-chatbot/data/accounts/accounts.csv @@ -0,0 +1,9 @@ +account_id,customer_id,product_type,currency,balance,status,opened_at +A1001,C001,checking,ARS,845000.50,active,2019-03-12 +A1002,C001,credit_card,ARS,120000.00,active,2020-01-15 +A1003,C002,checking,ARS,210400.00,active,2021-07-01 +A1004,C002,savings,ARS,550000.00,active,2021-08-10 +A1005,C003,checking,ARS,45200.75,active,2023-01-18 +A1006,C004,checking,ARS,1500000.00,active,2018-11-05 +A1007,C004,credit_card,ARS,0.00,active,2019-02-01 +A1008,C005,checking,ARS,12800.00,active,2022-09-22 diff --git a/usecases/llm-vector-banking-chatbot/data/card_products/card_products.csv b/usecases/llm-vector-banking-chatbot/data/card_products/card_products.csv new file mode 100644 index 0000000..43751ad --- /dev/null +++ b/usecases/llm-vector-banking-chatbot/data/card_products/card_products.csv @@ -0,0 +1,6 @@ +product_id,name,annual_fee,reward_categories,perk_summary,min_segment,tags +P_TRAVEL,Dozer Travel Signature,45000,"travel|dining","3x points on travel and dining, lounge access, travel insurance",premium,"travel,dining,lounge" +P_CASHBACK,Dozer Everyday Cashback,0,"groceries|fuel|shopping","2% cash back on groceries and fuel, 1% everywhere else",mass_market,"cashback,everyday" +P_STUDENT,Dozer Campus Starter,0,"education|dining|transit","1.5x points on education and transit, no annual fee",student,"student,education,no_fee" +P_LUXURY,Dozer Black Reserve,120000,"shopping|travel","5x luxury shopping, concierge, airport transfer credits",premium,"luxury,concierge,high_fee" +P_BALANCE,Dozer Balance Transfer,15000,"shopping","0% APR intro on transfers, modest rewards elsewhere",mass_affluent,"balance_transfer,apr" diff --git a/usecases/llm-vector-banking-chatbot/data/customer_profiles/customer_profiles.csv b/usecases/llm-vector-banking-chatbot/data/customer_profiles/customer_profiles.csv new file mode 100644 index 0000000..b3a3fa9 --- /dev/null +++ b/usecases/llm-vector-banking-chatbot/data/customer_profiles/customer_profiles.csv @@ -0,0 +1,6 @@ +customer_id,full_name,segment,city,preferred_channel,risk_tier,joined_at +C001,Ava Chen,premium,Buenos Aires,mobile,low,2019-03-12 +C002,Mateo Ruiz,mass_affluent,Córdoba,web,medium,2021-07-01 +C003,Sofia Almeida,student,Rosario,mobile,medium,2023-01-18 +C004,Diego Morales,premium,Mendoza,branch,low,2018-11-05 +C005,Luna Ferreira,mass_market,La Plata,mobile,high,2022-09-22 diff --git a/usecases/llm-vector-banking-chatbot/data/transactions/transactions.csv b/usecases/llm-vector-banking-chatbot/data/transactions/transactions.csv new file mode 100644 index 0000000..d623d21 --- /dev/null +++ b/usecases/llm-vector-banking-chatbot/data/transactions/transactions.csv @@ -0,0 +1,19 @@ +txn_id,account_id,customer_id,posted_at,merchant,category,amount,currency +T9001,A1002,C001,2026-08-01,Aerolineas Argentinas,travel,185000.00,ARS +T9002,A1002,C001,2026-08-03,Osaka Sushi,dining,42000.00,ARS +T9003,A1002,C001,2026-08-05,Hilton Buenos Aires,travel,98000.00,ARS +T9004,A1002,C001,2026-08-08,Mercado Libre,shopping,31000.00,ARS +T9005,A1002,C001,2026-08-12,Cafe Tortoni,dining,8500.00,ARS +T9006,A1003,C002,2026-08-02,Shell,fuel,28000.00,ARS +T9007,A1003,C002,2026-08-04,Carrefour,groceries,45000.00,ARS +T9008,A1003,C002,2026-08-09,Farmacity,health,12000.00,ARS +T9009,A1003,C002,2026-08-14,Netflix,subscriptions,6500.00,ARS +T9010,A1005,C003,2026-08-03,Udemy,education,22000.00,ARS +T9011,A1005,C003,2026-08-07,Starbucks,dining,4800.00,ARS +T9012,A1005,C003,2026-08-11,Subte BA,transit,2500.00,ARS +T9013,A1007,C004,2026-08-01,Louis Vuitton,shopping,320000.00,ARS +T9014,A1007,C004,2026-08-06,Four Seasons,travel,210000.00,ARS +T9015,A1007,C004,2026-08-10,Tegui,dining,95000.00,ARS +T9016,A1008,C005,2026-08-02,Easy,home,18000.00,ARS +T9017,A1008,C005,2026-08-08,YPF,fuel,15000.00,ARS +T9018,A1008,C005,2026-08-13,Coto,groceries,22000.00,ARS diff --git a/usecases/llm-vector-banking-chatbot/demo-fixture.mp4 b/usecases/llm-vector-banking-chatbot/demo-fixture.mp4 new file mode 100644 index 0000000..ec1c95f Binary files /dev/null and b/usecases/llm-vector-banking-chatbot/demo-fixture.mp4 differ diff --git a/usecases/llm-vector-banking-chatbot/demo-log.txt b/usecases/llm-vector-banking-chatbot/demo-log.txt new file mode 100644 index 0000000..1695718 --- /dev/null +++ b/usecases/llm-vector-banking-chatbot/demo-log.txt @@ -0,0 +1,61 @@ +=== DEMO LOG — llm-vector-banking-chatbot (Dozer 0.3.0 + RAG stub) === +Captured: 2026-09-08 05:46:36 ART + +--- dozer -V --- +dozer 0.3.0 + +--- REST endpoints (Dozer already running) --- +["/customer_profiles","/accounts","/transactions","/card_products","/customer_spending_profile"] + +--- GET /customer_profiles (first record) --- +{ + "customer_id": "C001", + "full_name": "Ava Chen", + "segment": "premium", + "city": "Buenos Aires", + "preferred_channel": "mobile", + "risk_tier": "low", + "joined_at": "2019-03-12", + "__dozer_record_id": 0, + "__dozer_record_version": 1 +} + +--- GET /customer_spending_profile (C001 rows) --- +[ + { + "customer_id": "C001", + "txn_count": 2, + "total_spend": 283000.0, + "top_category": "travel", + "__dozer_record_id": 0, + "__dozer_record_version": 2 + }, + { + "customer_id": "C001", + "txn_count": 1, + "total_spend": 31000.0, + "top_category": "shopping", + "__dozer_record_id": 2, + "__dozer_record_version": 1 + }, + { + "customer_id": "C001", + "txn_count": 2, + "total_spend": 50500.0, + "top_category": "dining", + "__dozer_record_id": 1, + "__dozer_record_version": 2 + } +] + +--- chatbot --use-dozer C001 --- +For Ava Chen (C001), segment=premium, city=Buenos Aires, total_spend=$364,500, top=[travel=$283,000, dining=$50,500, shopping=$31,000], given question «Which card is best for travel and dining rewards?», recommend **Dozer Travel Signature** (P_TRAVEL): 3x points on travel and dining, lounge access, travel insurance (annual fee 45000 ARS). + +--- chatbot --use-dozer C003 --- +For Sofia Almeida (C003), segment=student, city=Rosario, total_spend=$29,300, top=[education=$22,000, dining=$4,800, transit=$2,500], given question «I need a no-fee student card for education», recommend **Dozer Campus Starter** (P_STUDENT): 1.5x points on education and transit, no annual fee (annual fee 0 ARS). + +--- pytest --- +....... [100%] +7 passed in 0.01s + +=== END DEMO LOG === diff --git a/usecases/llm-vector-banking-chatbot/dozer-config.yaml b/usecases/llm-vector-banking-chatbot/dozer-config.yaml new file mode 100644 index 0000000..c5e4bd9 --- /dev/null +++ b/usecases/llm-vector-banking-chatbot/dozer-config.yaml @@ -0,0 +1,105 @@ +# Dozer + LLM banking chatbot sample (issue getdozer/dozer#1690) +# Sources multiple LocalStorage CSV datasets, materializes a unified +# customer spending profile, and exposes REST/gRPC APIs for LangChain. +# +# Verified against Dozer 0.3.0 (API-era binary). Dozer >=0.4.0 removed +# built-in `endpoints` / dozer-api; see SAMPLE_NOTES.md. +app_name: llm-vector-banking-chatbot +version: 1 + +api: + rest: + port: 8080 + host: 0.0.0.0 + grpc: + port: 50051 + host: 0.0.0.0 + +connections: + - name: bank_local + config: !LocalStorage + details: + path: data + tables: + - !Table + name: customer_profiles + config: !CSV + path: customer_profiles + extension: .csv + - !Table + name: accounts + config: !CSV + path: accounts + extension: .csv + - !Table + name: transactions + config: !CSV + path: transactions + extension: .csv + - !Table + name: card_products + config: !CSV + path: card_products + extension: .csv + +sources: + - name: customer_profiles + table_name: customer_profiles + connection: bank_local + - name: accounts + table_name: accounts + connection: bank_local + - name: transactions + table_name: transactions + connection: bank_local + - name: card_products + table_name: card_products + connection: bank_local + +sql: | + -- Unified spend profile: total spend + category per customer + SELECT + t.customer_id AS customer_id, + COUNT(t.txn_id) AS txn_count, + SUM(t.amount) AS total_spend, + t.category AS top_category + INTO customer_spending_profile + FROM transactions t + GROUP BY t.customer_id, t.category; + +endpoints: + - name: customer_profiles + path: /customer_profiles + table_name: customer_profiles + index: + primary_key: + - customer_id + + - name: accounts + path: /accounts + table_name: accounts + index: + primary_key: + - account_id + + - name: transactions + path: /transactions + table_name: transactions + index: + primary_key: + - txn_id + + - name: card_products + path: /card_products + table_name: card_products + index: + primary_key: + - product_id + + - name: customer_spending_profile + path: /customer_spending_profile + table_name: customer_spending_profile + index: + primary_key: + - customer_id + - top_category diff --git a/usecases/llm-vector-banking-chatbot/tests/test_rag_chatbot.py b/usecases/llm-vector-banking-chatbot/tests/test_rag_chatbot.py new file mode 100644 index 0000000..1fadea4 --- /dev/null +++ b/usecases/llm-vector-banking-chatbot/tests/test_rag_chatbot.py @@ -0,0 +1,80 @@ +"""Unit tests for the offline fixture / ranking path (no Dozer binary required).""" + +from __future__ import annotations + +import sys +from pathlib import Path + +import pytest + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT / "app")) + +from rag_chatbot import ( # noqa: E402 + CustomerContext, + deterministic_answer, + keyword_retrieve, + load_card_docs, + load_fixture_context, + rerank_for_customer, + segment_ok, +) + + +def test_segment_ok_ordering() -> None: + assert segment_ok("premium", "student") + assert segment_ok("student", "student") + assert not segment_ok("student", "premium") + assert segment_ok("mass_affluent", "mass_market") + + +def test_load_fixture_context_c001() -> None: + ctx = load_fixture_context("C001") + assert ctx.full_name == "Ava Chen" + assert ctx.segment == "premium" + assert ctx.spend_by_category["travel"] == pytest.approx(283000.0) + assert ctx.total_spend == pytest.approx(364500.0) + + +def test_load_fixture_unknown_customer() -> None: + with pytest.raises(SystemExit): + load_fixture_context("NOPE") + + +def test_keyword_retrieve_travel() -> None: + cards = load_card_docs() + hits = keyword_retrieve("travel dining lounge rewards", cards, k=3) + assert hits + assert hits[0]["product_id"] == "P_TRAVEL" + + +def test_rerank_prefers_travel_for_premium_spender() -> None: + ctx = load_fixture_context("C001") + cards = load_card_docs() + ranked = rerank_for_customer(ctx, cards, question="travel and dining rewards") + assert ranked[0]["product_id"] == "P_TRAVEL" + + +def test_rerank_gates_student_away_from_luxury() -> None: + ctx = load_fixture_context("C003") + cards = load_card_docs() + ranked = rerank_for_customer(ctx, cards, question="no-fee student education") + ids = [c["product_id"] for c in ranked] + assert "P_LUXURY" not in ids + assert "P_TRAVEL" not in ids + assert ranked[0]["product_id"] == "P_STUDENT" + + +def test_deterministic_answer_mentions_card() -> None: + ctx = CustomerContext( + customer_id="C001", + full_name="Ava Chen", + segment="premium", + city="Buenos Aires", + spend_by_category={"travel": 100.0}, + total_spend=100.0, + ) + cards = load_card_docs() + text = deterministic_answer(ctx, "travel?", [cards[0]]) + assert "Dozer Travel Signature" in text + assert "C001" in text