From 4b83ff6bae85a1fedd0ec8800109fac2fe8a214e Mon Sep 17 00:00:00 2001 From: saurav-codes <52165188+saurav-codes@users.noreply.github.com> Date: Tue, 8 Sep 2026 10:26:07 +0530 Subject: [PATCH] feat: add Dozer + LangChain + vector database chatbot sample --- README.md | 1 + usecases/llm-langchain/README.md | 36 +++++++ usecases/llm-langchain/chatbot.py | 97 +++++++++++++++++++ usecases/llm-langchain/data/card_products.csv | 5 + usecases/llm-langchain/data/customers.csv | 7 ++ usecases/llm-langchain/data/transactions.csv | 16 +++ usecases/llm-langchain/dozer-config.yaml | 71 ++++++++++++++ usecases/llm-langchain/requirements.txt | 5 + 8 files changed, 238 insertions(+) create mode 100644 usecases/llm-langchain/README.md create mode 100644 usecases/llm-langchain/chatbot.py create mode 100644 usecases/llm-langchain/data/card_products.csv create mode 100644 usecases/llm-langchain/data/customers.csv create mode 100644 usecases/llm-langchain/data/transactions.csv create mode 100644 usecases/llm-langchain/dozer-config.yaml create mode 100644 usecases/llm-langchain/requirements.txt diff --git a/README.md b/README.md index 42b771b0..ef48c596 100644 --- a/README.md +++ b/README.md @@ -45,6 +45,7 @@ Refer to the [Installation section](https://getdozer.io/docs/installation) for i | Use Cases | [Flight Microservices](./usecases/pg-flights) | Build APIs over multiple microservices. | | | [Scaling Ecommerce](./usecases/scaling-ecommerce) | Profile and benchmark Dozer using an ecommerce data set | | | [IMDB Analytics](./usecases/imdb-analytics) | Use Dozer to get interesting analytics using an IMDb dataset | +| | [LLM Chatbot](./usecases/llm-langchain) | Recommend credit cards using Dozer, LangChain, and a vector database | | | Use Dozer to Instrument (Coming soon) | Combine Log data to get real time insights | | | Real Time Model Scoring (Coming soon) | Deploy trained models to get real time insights as APIs | | | | | diff --git a/usecases/llm-langchain/README.md b/usecases/llm-langchain/README.md new file mode 100644 index 00000000..08c2d307 --- /dev/null +++ b/usecases/llm-langchain/README.md @@ -0,0 +1,36 @@ +# Dozer + LLM + Vector Database + LangChain + +This sample implements the flow from the [LLM chatbot blog post](https://getdozer.io/blog/llm-chatbot): a bank chatbot that hyper-personalizes credit card recommendations. + +Dozer sources multiple datasets (customer profiles, transactions, card products) from local CSV files, builds a real-time spending aggregate in SQL, and serves it all as low-latency REST APIs. LangChain embeds the card products into a Chroma vector database and passes the unified customer profile as context to an LLM. + +## Prerequisites + +- [Dozer](https://getdozer.io/docs/installation) +- Python 3.9 or higher +- An OpenAI API key + +## Run + +### 1. Install the Python dependencies + +```bash +pip install -r requirements.txt +export OPENAI_API_KEY=sk-... +``` + +### 2. Start Dozer + +```bash +dozer run -c dozer-config.yaml +``` + +Dozer exposes REST APIs on port 8080: `/customers`, `/transactions`, `/card_products`, and `/customer_spending` (per-customer spending aggregated by category). + +### 3. Chat + +```bash +python chatbot.py --customer C001 --question "Which card best fits this customer, and why?" +``` + +The chatbot fetches the customer profile and spending breakdown from Dozer, retrieves the most relevant card products from the Chroma vector store, and asks the LLM for a personalized recommendation. diff --git a/usecases/llm-langchain/chatbot.py b/usecases/llm-langchain/chatbot.py new file mode 100644 index 00000000..e013cffb --- /dev/null +++ b/usecases/llm-langchain/chatbot.py @@ -0,0 +1,97 @@ +"""Bank chatbot powered by Dozer REST APIs, LangChain, and a Chroma vector store. + +Dozer serves customers, transactions, and a spending aggregate as REST APIs. +This script embeds card products into Chroma and lets an LLM recommend a card +against the unified customer profile fetched from Dozer. +""" + +import argparse +import os + +import requests +from langchain.chains import RetrievalQA +from langchain_community.vectorstores import Chroma +from langchain_openai import ChatOpenAI, OpenAIEmbeddings + +DOZER_API = os.environ.get("DOZER_API", "http://localhost:8080") + + +def fetch_records(endpoint, limit=1000): + response = requests.get( + f"{DOZER_API}{endpoint}", params={"$limit": limit}, timeout=30 + ) + response.raise_for_status() + return response.json().get("records", []) + + +def build_vector_store(products): + descriptions = [ + ( + f"{p['product_name']} (annual fee ${p['annual_fee']}, " + f"min income {p['min_income']}): {p['description']}" + ) + for p in products + ] + return Chroma.from_texts( + descriptions, + embedding=OpenAIEmbeddings(), + collection_name="card_products", + ) + + +def customer_context(customers, spending, customer_id): + customer = next( + (c for c in customers if c["customer_id"] == customer_id), None + ) + if customer is None: + raise SystemExit( + f"Customer {customer_id} not found at the /customers endpoint." + ) + breakdown = [ + f"{s['category']}: ${s['total_amount']}" + for s in spending + if s["customer_id"] == customer_id + ] + return ( + f"{customer['name']}, age {customer['age']}, " + f"income ${customer['income']}, " + f"spending breakdown: {', '.join(breakdown) or 'no transactions'}" + ) + + +def main(): + if not os.environ.get("OPENAI_API_KEY"): + raise SystemExit("Set OPENAI_API_KEY before running the chatbot.") + + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--customer", + default="C001", + help="customer id from the Dozer /customers endpoint (default C001)", + ) + parser.add_argument( + "--question", + default="Which card best fits this customer, and why?", + help="question to ask the LLM about the customer", + ) + args = parser.parse_args() + + customers = fetch_records("/customers") + spending = fetch_records("/customer_spending") + products = fetch_records("/card_products") + + vector_store = build_vector_store(products) + chain = RetrievalQA.from_chain_type( + llm=ChatOpenAI(model="gpt-4o-mini", temperature=0), + retriever=vector_store.as_retriever(search_kwargs={"k": 2}), + ) + prompt = ( + f"Customer context: " + f"{customer_context(customers, spending, args.customer)}\n" + f"Question: {args.question}" + ) + print(chain.invoke(prompt)["result"]) + + +if __name__ == "__main__": + main() diff --git a/usecases/llm-langchain/data/card_products.csv b/usecases/llm-langchain/data/card_products.csv new file mode 100644 index 00000000..fd5a0b53 --- /dev/null +++ b/usecases/llm-langchain/data/card_products.csv @@ -0,0 +1,5 @@ +card_id,product_name,description,annual_fee,min_income +P001,Explorer Travel Card,2x points on travel and dining plus lounge access,95.0,40000 +P002,Cashback Everyday Card,3% cashback on grocery and fuel with no annual fee,0.0,0 +P003,Premium Rewards Card,5x points on retail and travel with concierge service,450.0,120000 +P004,Starter Student Card,Low credit limit card for students with no fees,0.0,20000 diff --git a/usecases/llm-langchain/data/customers.csv b/usecases/llm-langchain/data/customers.csv new file mode 100644 index 00000000..36235a31 --- /dev/null +++ b/usecases/llm-langchain/data/customers.csv @@ -0,0 +1,7 @@ +customer_id,name,age,income,city +C001,Ava Patel,29,95000.0,New York +C002,Marco Ruiz,41,62000.0,Austin +C003,Priya Nair,35,145000.0,San Francisco +C004,Dan Okafor,23,38000.0,Chicago +C005,Lena Schmidt,52,98000.0,Seattle +C006,Ryo Tanaka,47,73000.0,Boston diff --git a/usecases/llm-langchain/data/transactions.csv b/usecases/llm-langchain/data/transactions.csv new file mode 100644 index 00000000..508ce6f1 --- /dev/null +++ b/usecases/llm-langchain/data/transactions.csv @@ -0,0 +1,16 @@ +transaction_id,customer_id,category,amount +T001,C001,TRAVEL,820.5 +T002,C001,DINING,64.0 +T003,C002,GROCERY,120.3 +T004,C002,FUEL,45.8 +T005,C003,TRAVEL,1540.0 +T006,C003,DINING,215.4 +T007,C004,GROCERY,88.2 +T008,C004,ONLINE,31.9 +T009,C005,RETAIL,540.0 +T010,C005,TRAVEL,260.0 +T011,C006,GROCERY,96.4 +T012,C006,FUEL,52.1 +T013,C001,TRAVEL,310.0 +T014,C003,RETAIL,899.9 +T015,C002,GROCERY,140.6 diff --git a/usecases/llm-langchain/dozer-config.yaml b/usecases/llm-langchain/dozer-config.yaml new file mode 100644 index 00000000..f9eba8b4 --- /dev/null +++ b/usecases/llm-langchain/dozer-config.yaml @@ -0,0 +1,71 @@ +app_name: llm-langchain +version: 1 +connections: + - name: bank_data + config: !LocalStorage + details: + path: data + tables: + - !Table + name: customers + config: !Csv + path: customers + extension: .csv + - !Table + name: transactions + config: !Csv + path: transactions + extension: .csv + - !Table + name: card_products + config: !Csv + path: card_products + extension: .csv + +sql: | + SELECT + customer_id, + category, + SUM(amount) as total_amount, + COUNT(transaction_id) as transaction_count + INTO customer_spending + FROM transactions + GROUP BY customer_id, category; + +sources: + - name: customers + table_name: customers + connection: !Ref bank_data + - name: transactions + table_name: transactions + connection: !Ref bank_data + - name: card_products + table_name: card_products + connection: !Ref bank_data + +endpoints: + - name: customers + path: /customers + table_name: customers + index: + primary_key: + - customer_id + - name: transactions + path: /transactions + table_name: transactions + index: + primary_key: + - transaction_id + - name: card_products + path: /card_products + table_name: card_products + index: + primary_key: + - card_id + - name: customer_spending + path: /customer_spending + table_name: customer_spending + index: + primary_key: + - customer_id + - category diff --git a/usecases/llm-langchain/requirements.txt b/usecases/llm-langchain/requirements.txt new file mode 100644 index 00000000..e608597c --- /dev/null +++ b/usecases/llm-langchain/requirements.txt @@ -0,0 +1,5 @@ +langchain +langchain-openai +langchain-community +chromadb +requests