From 1772ccad7fb84eda8aff83937bceab7a2c4455c5 Mon Sep 17 00:00:00 2001 From: Pavan Kumar Date: Fri, 2 Oct 2026 19:34:41 +0530 Subject: [PATCH] added new docs for rag --- README.md | 19 ++++++++++++++++ examples/rag_demo.py | 21 ++++++++++++------ fixtures/documents/01-overview.md | 9 ++++++++ fixtures/documents/02-loading-and-indexing.md | 9 ++++++++ fixtures/documents/03-chunking.md | 9 ++++++++ .../04-embeddings-and-vector-search.md | 9 ++++++++ fixtures/documents/05-mcp.md | 9 ++++++++ .../documents/06-evaluation-and-workflows.md | 9 ++++++++ fixtures/documents/mcp_basics.txt | 3 --- fixtures/documents/mcp_tools.txt | 3 --- fixtures/documents/rag_basics.txt | 3 --- src/rag/loaders.py | 4 +++- tests/test_loaders.py | 22 +++++++++++++++++++ 13 files changed, 112 insertions(+), 17 deletions(-) create mode 100644 fixtures/documents/01-overview.md create mode 100644 fixtures/documents/02-loading-and-indexing.md create mode 100644 fixtures/documents/03-chunking.md create mode 100644 fixtures/documents/04-embeddings-and-vector-search.md create mode 100644 fixtures/documents/05-mcp.md create mode 100644 fixtures/documents/06-evaluation-and-workflows.md delete mode 100644 fixtures/documents/mcp_basics.txt delete mode 100644 fixtures/documents/mcp_tools.txt delete mode 100644 fixtures/documents/rag_basics.txt diff --git a/README.md b/README.md index 07004d4..6fd5589 100644 --- a/README.md +++ b/README.md @@ -76,6 +76,25 @@ python examples/rag_demo.py This repository is offline-first and intentionally uses deterministic mock embeddings rather than paid external embedding providers. The examples and tests are designed to run entirely on local fixtures. No secret, credential, or hosted service is required for standard execution. +## Workshop Corpus + +The fixture corpus is intentionally built as a compact but realistic RAG workshop. Instead of three tiny text files, the project now includes six original Markdown documents that cover the main concepts of retrieval: + +- a project overview and RAG fundamentals +- document loading and indexing +- chunking and segmentation +- embeddings and vector search +- MCP tool contracts and structured tool discovery +- evaluation and end-to-end retrieval workflows + +These documents are designed to support questions such as: + +- "What is retrieval augmented generation?" +- "How does a retriever choose the best chunks?" +- "What does an MCP tool do?" + +The loader supports both `.txt` and `.md` files and preserves deterministic ordering, while the workshop corpus emphasizes the Markdown workflow that is common in documentation-heavy RAG use cases. + ## Architecture Explanation ### Documents diff --git a/examples/rag_demo.py b/examples/rag_demo.py index 34d6965..269cbd8 100644 --- a/examples/rag_demo.py +++ b/examples/rag_demo.py @@ -33,13 +33,20 @@ def main() -> None: chunks=chunks, ) - query = "MCP tools and retrieval basics" - results = retriever.retrieve(query, top_k=3) - print(f"Query: {query}") - print("Results:") - for index, result in enumerate(results, start=1): - print(f" {index}. {result.source} (score={result.score:.4f})") - print(f" {result.text[:120]}...") + queries = [ + "What is retrieval augmented generation?", + "What does an MCP tool do?", + "How does a vector store rank relevant chunks?", + ] + + for query in queries: + results = retriever.retrieve(query, top_k=3) + print(f"Query: {query}") + print("Results:") + for index, result in enumerate(results, start=1): + print(f" {index}. {result.source} (score={result.score:.4f})") + print(f" {result.text[:120]}...") + print() if __name__ == "__main__": diff --git a/fixtures/documents/01-overview.md b/fixtures/documents/01-overview.md new file mode 100644 index 0000000..5a8ab83 --- /dev/null +++ b/fixtures/documents/01-overview.md @@ -0,0 +1,9 @@ +# Retrieval-Augmented Generation Overview + +Retrieval-augmented generation, or RAG, combines a searchable knowledge base with a language model. Instead of asking the model to memorize everything in its weights, the system first retrieves the most relevant evidence from a corpus and then uses that evidence as context. + +A typical RAG pipeline starts with a document loader, which reads source files from disk or a remote store. The loader converts each source into a normalized document object with metadata such as filename, language, and source path. The next stage splits each document into chunks so that retrieval can focus on smaller, semantically meaningful passages. + +Once chunks are created, an embedding model maps them into a vector space. A FAISS index stores those vectors for efficient similarity search. At query time, the system embeds the user question, searches the index, and returns the most relevant chunks. The selected passages are then supplied to a downstream model for final reasoning or synthesis. + +This project is intentionally small, transparent, and offline-first. It teaches the mechanics behind retrieval without hiding them behind a large framework. diff --git a/fixtures/documents/02-loading-and-indexing.md b/fixtures/documents/02-loading-and-indexing.md new file mode 100644 index 0000000..af13218 --- /dev/null +++ b/fixtures/documents/02-loading-and-indexing.md @@ -0,0 +1,9 @@ +# Document Loading and Indexing + +Document loading is the first step in any RAG system. The loader reads text files, normalizes whitespace, and preserves metadata that will later be used for debugging and traceability. A realistic loader should be deterministic, support UTF-8 content, and ignore files that are not appropriate for retrieval. + +In this workshop, the loader accepts plain text and Markdown files and produces a `Document` object for each file. A normalized document record includes the document ID, the raw text, and metadata such as the filename and source path. The loader sorts files consistently so that the corpus order remains stable across repeated runs. + +After loading, the indexer prepares those documents for search. Each chunk is turned into an embedding, and those embeddings are inserted into a FAISS index. A vector store preserves the relationship between a chunk and its metadata, which allows the retriever to return the matching text and its source document. + +This separation of concerns is important. Loading extracts content, indexing prepares the data for search, and retrieval answers the user question by comparing the query embedding to the stored vectors. diff --git a/fixtures/documents/03-chunking.md b/fixtures/documents/03-chunking.md new file mode 100644 index 0000000..7586777 --- /dev/null +++ b/fixtures/documents/03-chunking.md @@ -0,0 +1,9 @@ +# Chunking Strategies + +Chunking is the process of splitting a document into smaller segments before indexing. A chunk should be large enough to preserve context, but small enough to stay semantically focused. The right size depends on the use case, but the engineering principle is consistent: short chunks reduce noise and improve retrieval precision. + +A basic chunker takes a document and a chunk size, then slices the text into contiguous blocks. Many systems also add overlap so that adjacent chunks share some context. Overlap is helpful when the relevant fact is split across boundaries, but too much overlap can create redundant retrieval results. + +Deterministic chunking matters because it makes the retrieval system stable and easy to test. In this project, chunk size and overlap are explicit parameters, and invalid combinations are rejected. That prevents accidental behavior where a chunker silently produces empty or irregular segments. + +Good chunking turns a document into a set of retrieval units. Each unit can be ranked independently against a query, which allows the system to identify the most relevant slice of information instead of retrieving an entire article. diff --git a/fixtures/documents/04-embeddings-and-vector-search.md b/fixtures/documents/04-embeddings-and-vector-search.md new file mode 100644 index 0000000..43a8aa7 --- /dev/null +++ b/fixtures/documents/04-embeddings-and-vector-search.md @@ -0,0 +1,9 @@ +# Embeddings and Vector Search + +Embeddings transform text into dense vectors so that similarity can be computed with mathematical operations instead of string comparisons. In a retrieval pipeline, each chunk receives an embedding that captures semantic relationships between words and phrases. + +A mock embedding provider is useful in education because it is deterministic and reproducible. It does not require external models or API calls. Instead, the provider derives a stable vector from the chunk text using a hashing-based approach. This makes the workshop easy to run locally and ensures tests remain repeatable. + +Once embeddings are created, a vector store such as FAISS can index the vectors. The store keeps both the vector and a reference to the chunk metadata, so the retriever can return the original text and source file. Query-time retrieval is then a nearest-neighbor search: convert the query into an embedding, compare it against the stored vectors, and select the best matches. + +The ranking function is measured by distance. Lower distance means a stronger match, and a retrieval layer can convert that to a score that is easier to explain in a result table. diff --git a/fixtures/documents/05-mcp.md b/fixtures/documents/05-mcp.md new file mode 100644 index 0000000..1cc2c4e --- /dev/null +++ b/fixtures/documents/05-mcp.md @@ -0,0 +1,9 @@ +# MCP Tool Contracts + +The Model Context Protocol defines a clean contract for connecting clients and servers. In practice, an MCP server exposes tools, resources, and prompts that other systems can inspect and invoke programmatically. This makes tool access predictable, structured, and machine-readable. + +A tool contract includes a name, a description, and input schema. The client can ask the server which tools are available before invoking one. That allows an agent to discover capabilities without hard-coding them in advance. In a retrieval workflow, a tool may accept a user query and return the top-k most relevant chunks. + +The retrieval tool in this project is intentionally thin. It validates the input query, invokes the retriever, and converts the returned results into a JSON-friendly payload. The MCP boundary is separated from the retrieval logic so that the same retrieval behavior can be reused outside a protocol implementation. + +This explicit boundary is helpful for debugging. The protocol defines the interface, while the underlying code still owns the actual search semantics. diff --git a/fixtures/documents/06-evaluation-and-workflows.md b/fixtures/documents/06-evaluation-and-workflows.md new file mode 100644 index 0000000..f5af928 --- /dev/null +++ b/fixtures/documents/06-evaluation-and-workflows.md @@ -0,0 +1,9 @@ +# Evaluation and Retrieval Workflows + +A robust RAG workflow is not just about retrieving a result. It is also about evaluating whether the right evidence was selected and whether the final response is grounded in that evidence. Evaluation can involve checking source coverage, measuring retrieval precision, and reviewing the final answer for unsupported claims. + +In a workshop, a small retrieval pipeline often uses hand-crafted queries to test the system. For example, a user may ask, "What is retrieval augmented generation?" or "What does an MCP tool do?" The expected behavior is that the retrieval layer surfaces the most relevant passages rather than a random assortment of semantically adjacent text. + +A good workflow keeps the retrieval stage deterministic. That makes it easy to compare the effects of different chunk sizes, overlap values, and top-k limits. It also reduces the chance that small changes in the environment will change the results in surprising ways. + +This project emphasizes a transparent, local workflow that is easy to reason about. The documents, embeddings, and retrieval results are all inspectable, which makes the lab suitable for teaching and deliberate experimentation. diff --git a/fixtures/documents/mcp_basics.txt b/fixtures/documents/mcp_basics.txt deleted file mode 100644 index 474c9f3..0000000 --- a/fixtures/documents/mcp_basics.txt +++ /dev/null @@ -1,3 +0,0 @@ -MCP stands for Model Context Protocol. It helps clients connect to servers and invoke tools. The protocol defines a clean contract for machine-readable interactions between agents and local tools. - -A client sends a request. A server exposes tools, resources, and prompts. The protocol keeps the interaction explicit and structured so the agent can reason about what is available. diff --git a/fixtures/documents/mcp_tools.txt b/fixtures/documents/mcp_tools.txt deleted file mode 100644 index 50459bb..0000000 --- a/fixtures/documents/mcp_tools.txt +++ /dev/null @@ -1,3 +0,0 @@ -Tool discovery is the process of learning which capabilities a server offers. The server describes each tool with metadata such as a name, description, and parameter schema. Clients can inspect those descriptions before invoking a tool. - -Tool invocation uses a structured payload. The client sends a query or task and the server returns a result that is easy for other systems to parse. This makes retrieval and direct function calls predictable. diff --git a/fixtures/documents/rag_basics.txt b/fixtures/documents/rag_basics.txt deleted file mode 100644 index 3ff08a8..0000000 --- a/fixtures/documents/rag_basics.txt +++ /dev/null @@ -1,3 +0,0 @@ -Retrieval-augmented generation combines a knowledge corpus with a retriever. A document loader reads source files. A chunker splits them into smaller segments. An embedding model converts each chunk into a vector representation. - -A vector store indexes those embeddings so a query can be compared against the corpus. Top-k retrieval selects the most relevant chunks. The result is then used as context for a downstream model or workflow. diff --git a/src/rag/loaders.py b/src/rag/loaders.py index 021cc53..9dd2468 100644 --- a/src/rag/loaders.py +++ b/src/rag/loaders.py @@ -37,13 +37,15 @@ def load_documents(directory: str | Path) -> list[Document]: continue content = file_path.read_text(encoding="utf-8") + relative_source = file_path.relative_to(path).as_posix() documents.append( Document( document_id=file_path.stem, text=content, metadata={ - "source": str(file_path), + "source": relative_source, "filename": file_path.name, + "document_type": suffix, }, ) ) diff --git a/tests/test_loaders.py b/tests/test_loaders.py index edcf742..94c80c3 100644 --- a/tests/test_loaders.py +++ b/tests/test_loaders.py @@ -40,3 +40,25 @@ def test_load_documents_utf8_and_unsupported_extensions(tmp_path: Path) -> None: assert [doc.filename for doc in documents] == ["hello.txt", "notes.md"] assert documents[0].text == "héllo\n" + assert documents[0].source == "hello.txt" + assert "/tmp" not in documents[0].source + + +def test_load_documents_markdown_files_are_supported(tmp_path: Path) -> None: + doc_dir = tmp_path / "workshop" + doc_dir.mkdir() + (doc_dir / "overview.md").write_text("# Overview\n\nRAG helps retrieval.\n", encoding="utf-8") + (doc_dir / "notes.txt").write_text("plain text\n", encoding="utf-8") + + documents = load_documents(doc_dir) + + assert [doc.filename for doc in documents] == ["notes.txt", "overview.md"] + assert all(doc.source.endswith((".txt", ".md")) for doc in documents) + + +def test_workshop_corpus_is_a_markdown_collection() -> None: + documents = load_documents(Path(__file__).resolve().parents[1] / "fixtures" / "documents") + + assert len(documents) == 6 + assert all(doc.filename.endswith(".md") for doc in documents) + assert all(doc.metadata["source"].endswith(".md") for doc in documents)