Skip to content

3. Chat apps

Intermediate · 10 min read

Streamlit has chat elements built in. A full chatbot with history and streaming is about 30 lines.

3.1 The two chat elements

elements.py
import streamlit as st

with st.chat_message("user"):
    st.write("What is a vector database?")

with st.chat_message("assistant"):
    st.write("A database that stores embeddings and finds the most similar ones fast.")
    st.code("index.query(vector=q, top_k=5)")

prompt = st.chat_input("Ask anything…")         # pinned to the bottom of the page
if prompt:
    st.write("You typed:", prompt)
  • st.chat_message(role) — a message bubble with an avatar; "user" and "assistant" get default icons, or pass avatar="🤖". Anything inside the with block goes into the bubble.
  • st.chat_input() — returns the text on the rerun after the user presses Enter, otherwise None.

3.2 A chatbot with memory (echo version)

The pattern every chat app follows:

chat_echo.py
import streamlit as st

st.title("Echo bot")

# 1. history lives in session_state
if "messages" not in st.session_state:
    st.session_state.messages = [{"role": "assistant", "content": "Hi! Ask me anything."}]

# 2. redraw the whole conversation on every rerun
for m in st.session_state.messages:
    with st.chat_message(m["role"]):
        st.markdown(m["content"])

# 3. on new input: show it, save it, answer, save the answer
if prompt := st.chat_input("Your message"):
    st.session_state.messages.append({"role": "user", "content": prompt})
    with st.chat_message("user"):
        st.markdown(prompt)

    reply = f"You said: **{prompt}**"
    with st.chat_message("assistant"):
        st.markdown(reply)
    st.session_state.messages.append({"role": "assistant", "content": reply})

The messages list is already in the format LLM APIs expect (role + content) — so making it real is one function.

3.3 Streaming a real LLM reply — st.write_stream

st.write_stream takes a generator (or an OpenAI stream directly), shows tokens as they arrive, and returns the full text so you can save it.

chat_openai.py
import streamlit as st
from openai import OpenAI

st.title("Chat with GPT")
client = OpenAI(api_key=st.secrets["OPENAI_API_KEY"])     # see "Secrets" in topic 4

with st.sidebar:
    model = st.selectbox("Model", ["gpt-4o-mini", "gpt-4o"])
    temperature = st.slider("Temperature", 0.0, 1.0, 0.3)
    system = st.text_area("System prompt", "You are a helpful, concise assistant.")
    if st.button("New chat"):
        st.session_state.messages = []
        st.rerun()

st.session_state.setdefault("messages", [])
for m in st.session_state.messages:
    with st.chat_message(m["role"]):
        st.markdown(m["content"])

if prompt := st.chat_input("Ask anything…"):
    st.session_state.messages.append({"role": "user", "content": prompt})
    with st.chat_message("user"):
        st.markdown(prompt)

    with st.chat_message("assistant"):
        stream = client.chat.completions.create(
            model=model, temperature=temperature, stream=True,
            messages=[{"role": "system", "content": system}, *st.session_state.messages[-20:]],
        )
        reply = st.write_stream(stream)          # renders tokens live, returns the full string
    st.session_state.messages.append({"role": "assistant", "content": reply})

For any other provider, pass a generator of strings:

# no-run — inside the `with st.chat_message("assistant"):` block
import anthropic
client = anthropic.Anthropic(api_key=st.secrets["ANTHROPIC_API_KEY"])

def claude_tokens():
    with client.messages.stream(model="claude-sonnet-5-5", max_tokens=1024, system=system,
                                messages=st.session_state.messages) as s:
        yield from s.text_stream

reply = st.write_stream(claude_tokens())
# no-run — reads the SSE stream from the FastAPI notes
import json, httpx

def backend_tokens(message: str):
    with httpx.stream("POST", "http://127.0.0.1:8000/chat/stream", timeout=60,
                      json={"session_id": st.session_state.session_id, "message": message}) as r:
        for line in r.iter_lines():
            if line.startswith("data: {"):
                event = json.loads(line[6:])
                if "token" in event:
                    yield event["token"]

reply = st.write_stream(backend_tokens(prompt))

3.4 Handling errors

An API failure shouldn't crash the page or leave a half-saved conversation:

chat_errors.py
import streamlit as st

def fake_stream(prompt):
    yield "Thinking about "
    yield prompt
    if "fail" in prompt:
        raise TimeoutError("provider timed out")

st.session_state.setdefault("messages", [])
for m in st.session_state.messages:
    st.chat_message(m["role"]).markdown(m["content"])

if prompt := st.chat_input("Try a message containing 'fail'"):
    st.chat_message("user").markdown(prompt)
    with st.chat_message("assistant"):
        try:
            reply = st.write_stream(fake_stream(prompt))
        except Exception as e:
            st.error(f"Sorry, the model failed: {e}. Please try again.")
            st.stop()                                     # don't save a broken turn
    st.session_state.messages += [{"role": "user", "content": prompt},
                                  {"role": "assistant", "content": reply}]

3.5 Extras that make a demo feel finished

chat_extras.py
import streamlit as st

st.session_state.setdefault("messages", [
    {"role": "assistant", "content": "Refunds are processed within 5 working days.",
     "sources": ["policy.pdf · p3"], "tokens": 214},
])

for i, m in enumerate(st.session_state.messages):
    with st.chat_message(m["role"]):
        st.markdown(m["content"])
        if m["role"] == "assistant":
            if m.get("sources"):
                with st.expander("Sources"):
                    for s in m["sources"]:
                        st.caption(s)
            st.caption(f"{m.get('tokens', 0)} tokens")
            feedback = st.feedback("thumbs", key=f"fb_{i}")      # 👍 / 👎 → 1 / 0
            if feedback is not None:
                st.session_state[f"rating_{i}"] = feedback        # log it for your evals

st.download_button("Download chat", data="\n\n".join(
    f"{m['role']}: {m['content']}" for m in st.session_state.messages), file_name="chat.txt")
  • Sources in an expander — users trust answers they can check.
  • Feedback buttons — free evaluation data from real users.
  • Token count / cost caption — keeps you honest about spend.

Practice

  • Turn the echo bot into a real bot with your provider of choice, reading the key from st.secrets.
  • Keep only the last 10 messages when calling the LLM, but show the full history on screen.

Next: Streamlit for GenAI — a full "chat with your PDF" app.