# Copyright © 2026 Pathway
import logging
import os
import streamlit as st
from dotenv import load_dotenv
from pathway.xpacks.llm.document_store import IndexingStatus
from pathway.xpacks.llm.question_answering import RAGClient
load_dotenv()
PATHWAY_HOST = os.environ.get("PATHWAY_HOST", "app")
PATHWAY_PORT = os.environ.get("PATHWAY_PORT", 8000)
st.set_page_config(
page_title="Pathway Live Data Framework RAG App", page_icon="favicon.ico"
)
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s %(name)s %(levelname)s %(message)s",
datefmt="%Y-%m-%d %H:%M:%S",
force=True,
)
logger = logging.getLogger("streamlit")
logger.setLevel(logging.INFO)
conn = RAGClient(url=f"http://{PATHWAY_HOST}:{PATHWAY_PORT}")
note = """
Ask a question"""
st.markdown(note, unsafe_allow_html=True)
st.markdown(
"""
""",
unsafe_allow_html=True,
)
question = st.text_input(label="", placeholder="Ask your question?")
def get_indexed_files(metadata_list: list[dict], opt_key: str) -> list:
"""Get all available options in a specific metadata key."""
only_indexed_files = [
file
for file in metadata_list
if file["_indexing_status"] == IndexingStatus.INDEXED
]
options = set(map(lambda x: x[opt_key], only_indexed_files))
return list(options)
def get_ingested_files(metadata_list: list[dict], opt_key: str) -> list:
"""Get all available options in a specific metadata key."""
not_indexed_files = [
file
for file in metadata_list
if file["_indexing_status"] == IndexingStatus.INGESTED
]
options = set(map(lambda x: x[opt_key], not_indexed_files))
return list(options)
logger.info("Requesting list_documents...")
document_meta_list = conn.list_documents(keys=[])
logger.info("Received response list_documents")
st.session_state["document_meta_list"] = document_meta_list
indexed_files = get_indexed_files(st.session_state["document_meta_list"], "path")
ingested_files = get_ingested_files(st.session_state["document_meta_list"], "path")
logo_htm = """
"""
with st.sidebar:
st.markdown(logo_htm, unsafe_allow_html=True)
st.info(
body="See the source code [here](https://github.com/pathwaycom/llm-app/tree/main/templates/question_answering_rag).", # noqa: E501
icon=":material/code:",
)
indexed_file_names = [i.split("/")[-1] for i in indexed_files]
ingested_file_names = [i.split("/")[-1] for i in ingested_files]
markdown_table = "| Indexed files |\n| --- |\n"
for file_name in indexed_file_names:
markdown_table += f"| {file_name} |\n"
if len(ingested_file_names) > 0:
markdown_table += "| Files being processed |\n| --- |\n"
for file_name in ingested_file_names:
markdown_table += f"| {file_name} |\n"
st.markdown(markdown_table, unsafe_allow_html=True)
st.button("⟳ Refresh", use_container_width=True)
css = """
"""
st.markdown(css, unsafe_allow_html=True)
if question:
logger.info(
{
"_type": "search_request_event",
"query": question,
}
)
with st.spinner("Retrieving response..."):
api_response = conn.answer(question, return_context_docs=True)
response = api_response["response"]
context_docs = api_response["context_docs"]
logger.info(
{
"_type": "search_response_event",
"query": question,
"response": type(response),
}
)
logger.info(type(response))
st.markdown(f"**Answering question:** {question}")
st.markdown(f"""{response}""")
with st.expander(label="Context documents"):
st.markdown("Documents sent to LLM as context:\n")
for i, doc in enumerate(context_docs):
st.markdown(
f"{i+1}. Path: {doc['metadata']['path']}\n ```\n{doc['text']}\n```"
)