3. Chat apps¶
Intermediate · 10 min read
Streamlit has chat elements built in. A full chatbot with history and streaming is about 30 lines.
3.1 The two chat elements¶
elements.py
import streamlit as st
with st.chat_message("user"):
st.write("What is a vector database?")
with st.chat_message("assistant"):
st.write("A database that stores embeddings and finds the most similar ones fast.")
st.code("index.query(vector=q, top_k=5)")
prompt = st.chat_input("Ask anything…") # pinned to the bottom of the page
if prompt:
st.write("You typed:", prompt)
st.chat_message(role)— a message bubble with an avatar;"user"and"assistant"get default icons, or passavatar="🤖". Anything inside thewithblock goes into the bubble.st.chat_input()— returns the text on the rerun after the user presses Enter, otherwiseNone.
3.2 A chatbot with memory (echo version)¶
The pattern every chat app follows:
chat_echo.py
import streamlit as st
st.title("Echo bot")
# 1. history lives in session_state
if "messages" not in st.session_state:
st.session_state.messages = [{"role": "assistant", "content": "Hi! Ask me anything."}]
# 2. redraw the whole conversation on every rerun
for m in st.session_state.messages:
with st.chat_message(m["role"]):
st.markdown(m["content"])
# 3. on new input: show it, save it, answer, save the answer
if prompt := st.chat_input("Your message"):
st.session_state.messages.append({"role": "user", "content": prompt})
with st.chat_message("user"):
st.markdown(prompt)
reply = f"You said: **{prompt}**"
with st.chat_message("assistant"):
st.markdown(reply)
st.session_state.messages.append({"role": "assistant", "content": reply})
The messages list is already in the format LLM APIs expect (role + content) — so making it real is
one function.
3.3 Streaming a real LLM reply — st.write_stream¶
st.write_stream takes a generator (or an OpenAI stream directly), shows tokens as they arrive, and
returns the full text so you can save it.
chat_openai.py
import streamlit as st
from openai import OpenAI
st.title("Chat with GPT")
client = OpenAI(api_key=st.secrets["OPENAI_API_KEY"]) # see "Secrets" in topic 4
with st.sidebar:
model = st.selectbox("Model", ["gpt-4o-mini", "gpt-4o"])
temperature = st.slider("Temperature", 0.0, 1.0, 0.3)
system = st.text_area("System prompt", "You are a helpful, concise assistant.")
if st.button("New chat"):
st.session_state.messages = []
st.rerun()
st.session_state.setdefault("messages", [])
for m in st.session_state.messages:
with st.chat_message(m["role"]):
st.markdown(m["content"])
if prompt := st.chat_input("Ask anything…"):
st.session_state.messages.append({"role": "user", "content": prompt})
with st.chat_message("user"):
st.markdown(prompt)
with st.chat_message("assistant"):
stream = client.chat.completions.create(
model=model, temperature=temperature, stream=True,
messages=[{"role": "system", "content": system}, *st.session_state.messages[-20:]],
)
reply = st.write_stream(stream) # renders tokens live, returns the full string
st.session_state.messages.append({"role": "assistant", "content": reply})
For any other provider, pass a generator of strings:
# no-run — inside the `with st.chat_message("assistant"):` block
import anthropic
client = anthropic.Anthropic(api_key=st.secrets["ANTHROPIC_API_KEY"])
def claude_tokens():
with client.messages.stream(model="claude-sonnet-5-5", max_tokens=1024, system=system,
messages=st.session_state.messages) as s:
yield from s.text_stream
reply = st.write_stream(claude_tokens())
# no-run — reads the SSE stream from the FastAPI notes
import json, httpx
def backend_tokens(message: str):
with httpx.stream("POST", "http://127.0.0.1:8000/chat/stream", timeout=60,
json={"session_id": st.session_state.session_id, "message": message}) as r:
for line in r.iter_lines():
if line.startswith("data: {"):
event = json.loads(line[6:])
if "token" in event:
yield event["token"]
reply = st.write_stream(backend_tokens(prompt))
3.4 Handling errors¶
An API failure shouldn't crash the page or leave a half-saved conversation:
chat_errors.py
import streamlit as st
def fake_stream(prompt):
yield "Thinking about "
yield prompt
if "fail" in prompt:
raise TimeoutError("provider timed out")
st.session_state.setdefault("messages", [])
for m in st.session_state.messages:
st.chat_message(m["role"]).markdown(m["content"])
if prompt := st.chat_input("Try a message containing 'fail'"):
st.chat_message("user").markdown(prompt)
with st.chat_message("assistant"):
try:
reply = st.write_stream(fake_stream(prompt))
except Exception as e:
st.error(f"Sorry, the model failed: {e}. Please try again.")
st.stop() # don't save a broken turn
st.session_state.messages += [{"role": "user", "content": prompt},
{"role": "assistant", "content": reply}]
3.5 Extras that make a demo feel finished¶
chat_extras.py
import streamlit as st
st.session_state.setdefault("messages", [
{"role": "assistant", "content": "Refunds are processed within 5 working days.",
"sources": ["policy.pdf · p3"], "tokens": 214},
])
for i, m in enumerate(st.session_state.messages):
with st.chat_message(m["role"]):
st.markdown(m["content"])
if m["role"] == "assistant":
if m.get("sources"):
with st.expander("Sources"):
for s in m["sources"]:
st.caption(s)
st.caption(f"{m.get('tokens', 0)} tokens")
feedback = st.feedback("thumbs", key=f"fb_{i}") # 👍 / 👎 → 1 / 0
if feedback is not None:
st.session_state[f"rating_{i}"] = feedback # log it for your evals
st.download_button("Download chat", data="\n\n".join(
f"{m['role']}: {m['content']}" for m in st.session_state.messages), file_name="chat.txt")
- Sources in an expander — users trust answers they can check.
- Feedback buttons — free evaluation data from real users.
- Token count / cost caption — keeps you honest about spend.
Practice¶
- Turn the echo bot into a real bot with your provider of choice, reading the key from
st.secrets. - Keep only the last 10 messages when calling the LLM, but show the full history on screen.
Next: Streamlit for GenAI — a full "chat with your PDF" app.