From 22c9403e26fa224ef6af0817355dfe044f5fffb8 Mon Sep 17 00:00:00 2001 From: paramthakkar123 Date: Sun, 12 Jul 2026 22:19:31 +0530 Subject: [PATCH 01/10] Add Document Ingestion module for curated docs and web search MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Introduces a new Ingestion module that pulls documentation into a RAG index from two kinds of sources: curated docs (JuliaHealth, OMOP CDM, OHDSI, FunSQL.jl) and live web search via a pluggable provider (DuckDuckGo). Includes HTTP/S fetch with HTML-to-text cleaning, a search provider interface, and an orchestration pipeline (ingest/ingest_to_index) that feeds directly into RAGTools for indexing. New files: - src/ingestion.jl — module entry point - src/ingestion/types.jl — SourceDocument, SearchResult - src/ingestion/curated.jl — CURATED_SOURCES registry - src/ingestion/fetch.jl — fetch_url, html_to_text, fetch_curated - src/ingestion/search.jl — AbstractSearchProvider, DuckDuckGoProvider - src/ingestion/pipeline.jl — ingest, ingest_to_index - test/IngestionTest.jl — unit tests for offline logic - docs/src/ingestion.md — full user-facing documentation Dependencies added: HTTP, URIs --- Project.toml | 4 + docs/make.jl | 1 + docs/src/index.md | 5 +- docs/src/ingestion.md | 248 ++++++++++++++++++++++++++++++++++++++ src/HealthLLM.jl | 11 ++ src/ingestion.jl | 62 ++++++++++ src/ingestion/curated.jl | 32 +++++ src/ingestion/fetch.jl | 111 +++++++++++++++++ src/ingestion/pipeline.jl | 97 +++++++++++++++ src/ingestion/search.jl | 61 ++++++++++ src/ingestion/types.jl | 37 ++++++ test/IngestionTest.jl | 51 ++++++++ test/runtests.jl | 1 + 13 files changed, 719 insertions(+), 2 deletions(-) create mode 100644 docs/src/ingestion.md create mode 100644 src/ingestion.jl create mode 100644 src/ingestion/curated.jl create mode 100644 src/ingestion/fetch.jl create mode 100644 src/ingestion/pipeline.jl create mode 100644 src/ingestion/search.jl create mode 100644 src/ingestion/types.jl create mode 100644 test/IngestionTest.jl diff --git a/Project.toml b/Project.toml index fe54725..f38ea99 100644 --- a/Project.toml +++ b/Project.toml @@ -4,6 +4,7 @@ version = "0.1.0" authors = ["ParamThakkar123 and TheCedarPrince "] [deps] +HTTP = "cd3eb016-35fb-5094-929b-558a96fad6f3" HuggingFaceHub = "d0076355-e2c0-48e6-a044-05906e51b7fc" JSON3 = "0f8b85d8-7281-11e9-16c2-39a750bddbf1" LibPQ = "194296ae-ab2e-5f79-8cd4-7183a0a5a0d1" @@ -13,8 +14,10 @@ RAGTools = "16ddad29-bbe8-45a7-857d-3d9514eb0023" Serialization = "9e88b42a-f829-5b0c-bbe9-9e923198166b" SparseArrays = "2f01184e-e22b-5df5-ae63-d93ebab69eaf" Statistics = "10745b16-79ce-11e8-11f9-7d13ad32a3b2" +URIs = "5c2747f8-b7ea-4ff2-ba2e-563bfd36b1d4" [compat] +HTTP = "1.11" HuggingFaceHub = "0.1.2" JSON3 = "1.14.3" LibPQ = "1.18.0" @@ -24,4 +27,5 @@ RAGTools = "0.7.0" Serialization = "1.10" SparseArrays = "1.10" Statistics = "1.10" +URIs = "1.6" julia = "1.10" diff --git a/docs/make.jl b/docs/make.jl index 8ad03b9..961bd37 100644 --- a/docs/make.jl +++ b/docs/make.jl @@ -15,6 +15,7 @@ makedocs(; pages=[ "Home" => "index.md", "Getting Started" => "getting-started.md", + "Document Ingestion" => "ingestion.md", ], ) diff --git a/docs/src/index.md b/docs/src/index.md index 70a8c4c..d59486e 100644 --- a/docs/src/index.md +++ b/docs/src/index.md @@ -8,9 +8,10 @@ HealthLLM provides a compact Julia interface for retrieval-augmented workflows o ## Package scope -The package centers on four areas: +The package centers on five areas: - collecting source files and writing combined corpora +- ingesting curated docs and web-search results into an index (see [Document Ingestion](ingestion.md)) - building retrieval indexes through `RAGTools` - generating retrieval-backed answers for query construction - storing embeddings in PostgreSQL with `pgvector` @@ -30,5 +31,5 @@ index = build_index_rag(RAGTools.SimpleIndexer(), files) More detailed setup, testing commands, and the end-to-end walkthrough are in [Getting Started](getting-started.md). ```@autodocs -Modules = [HealthLLM, HealthLLM.Utils, HealthLLM.Database, HealthLLM.Query] +Modules = [HealthLLM, HealthLLM.Utils, HealthLLM.Database, HealthLLM.Query, HealthLLM.Ingestion] ``` diff --git a/docs/src/ingestion.md b/docs/src/ingestion.md new file mode 100644 index 0000000..f969280 --- /dev/null +++ b/docs/src/ingestion.md @@ -0,0 +1,248 @@ +```@meta +CurrentModule = HealthLLM +``` + +# Document Ingestion + +The ingestion layer pulls documentation into a RAG index from two kinds of +sources and hands the result straight to [`generate_funsql_query`](@ref): + +1. **Curated docs** — a fixed, reviewed list of high-quality pages + (JuliaHealth, OMOP CDM, OHDSI, FunSQL.jl). See [`CURATED_SOURCES`](@ref). +2. **Web search** — a live query against a pluggable search backend for + anything outside the curated set. See [`AbstractSearchProvider`](@ref). + +``` +CURATED_SOURCES ─┐ + ├─▶ fetch_url ─▶ SourceDocument ─┐ +web_search ──────┘ ├─▶ ingest ─▶ ingest_to_index ─▶ RAG index + │ + (each hit's URL fetched) ────┘ +``` + +Everything is available from the single `using HealthLLM` entry point. + +!!! note "Prerequisites" + Building an index calls an embedding model, so register one first (for + example a local [Ollama](https://ollama.com) model with + `register_models("llama3.2", "nomic-embed-text")`). Fetching and search make + outbound HTTP requests, so these functions need network access. The default + search backend ([`DuckDuckGoProvider`](@ref)) is keyless. + +## Quick demo + +Curated docs plus a targeted web search, indexed and queried end to end: + +```julia +using HealthLLM + +register_models("llama3.2", "nomic-embed-text") + +index = ingest_to_index(; + sources = ["FunSQL.jl", "OMOP CDM"], + query = "OMOP condition_occurrence table columns", + embedder_kwargs = (model = "nomic-embed-text",), +) + +answer = generate_funsql_query( + index, + "nomic-embed-text", + "llama3.2", + "Context: {input_query}. Answer with a FunSQL query.", + "How do I join the person and visit_occurrence tables?", +) +``` + +The rest of this page tours each building block so you can compose your own +pipeline. + +## The building blocks + +### 1. Curated sources + +[`CURATED_SOURCES`](@ref) is a registry mapping a source name to a list of +candidate URLs. Raw Markdown endpoints are preferred; rendered HTML pages are +accepted and cleaned to text automatically. + +```julia +julia> collect(keys(CURATED_SOURCES)) +4-element Vector{String}: + "JuliaHealth" + "OMOP CDM" + "OHDSI" + "FunSQL.jl" + +julia> CURATED_SOURCES["FunSQL.jl"] +3-element Vector{String}: + "https://raw.githubusercontent.com/MechanicalRabbit/FunSQL.jl/master/docs/src/index.md" + "https://raw.githubusercontent.com/MechanicalRabbit/FunSQL.jl/master/docs/src/guide/index.md" + "https://raw.githubusercontent.com/MechanicalRabbit/FunSQL.jl/master/docs/src/examples/index.md" +``` + +Add your own source by pushing to the registry before ingesting: + +```julia +CURATED_SOURCES["MyDocs"] = ["https://example.org/docs/index.md"] +``` + +### 2. Fetching and cleaning a URL + +[`fetch_url`](@ref) downloads a URL and returns clean text. HTML is stripped to +readable text with [`html_to_text`](@ref); Markdown and plain text pass through +untouched. + +```julia +text = fetch_url("https://raw.githubusercontent.com/OHDSI/CommonDataModel/main/README.md") + +html_to_text("

Hello world

") # -> "Hello world" +``` + +### 3. Fetching the curated set + +[`fetch_curated`](@ref) walks the registry, tries each candidate URL, and returns +a flat vector of [`SourceDocument`](@ref)s. URLs that fail (404, timeout) are +skipped with a warning, so partial results still come through. + +```julia +docs = fetch_curated(["FunSQL.jl", "OMOP CDM"]) + +for d in docs + println(d.source, " <- ", d.url, " (", length(d.content), " chars)") +end +``` + +Each [`SourceDocument`](@ref) carries its `source`, `url`, `title`, and cleaned +`content`. + +### 4. Web search + +[`web_search`](@ref) runs a query against a search backend and returns +[`SearchResult`](@ref)s. With no provider argument it uses +[`default_search_provider`](@ref) (currently the keyless +[`DuckDuckGoProvider`](@ref)). + +```julia +hits = web_search("OMOP CDM person table columns"; max_results=3) + +for h in hits + println(h.title, " => ", h.url) +end +``` + +Search is pluggable: define a `struct MyProvider <: AbstractSearchProvider` and a +`web_search(::MyProvider, query; max_results)` method to add a backend — nothing +else in the pipeline changes. + +### 5. Gathering documents with `ingest` + +[`ingest`](@ref) combines curated fetching and web search into one +`Vector{SourceDocument}`. Use `sources` for curated names and `query` for a live +search; either can be omitted. + +```julia +# Curated only +docs = ingest(; sources = ["FunSQL.jl"]) + +# Search only (skip curated docs) +docs = ingest(; sources = String[], query = "FunSQL From Where Select example") + +# Both +docs = ingest(; sources = ["FunSQL.jl"], query = "FunSQL aggregate group example") +``` + +By default each search hit's page is fetched for full text +(`fetch_search_content = true`); set it to `false` to keep only the provider's +snippet. + +### 6. Building an index with `ingest_to_index` + +[`ingest_to_index`](@ref) ingests documents and builds a RAG index directly from +the in-memory text — no intermediate files. Each chunk keeps its origin URL as +its source for provenance. + +```julia +register_models("llama3.2", "nomic-embed-text") + +index = ingest_to_index(; + sources = ["FunSQL.jl", "OMOP CDM"], + query = "OMOP condition_occurrence table", + embedder_kwargs = (model = "nomic-embed-text",), +) +``` + +If you already have documents from [`ingest`](@ref), reuse them instead of +fetching again: + +```julia +docs = ingest(; sources = ["OHDSI"]) +index = ingest_to_index(; docs = docs, embedder_kwargs = (model = "nomic-embed-text",)) +``` + +## Full end-to-end demo + +Save this as `ingestion_demo.jl` and run it with `julia --project=. ingestion_demo.jl` +(requires network access and a running embedding model): + +```julia +using HealthLLM + +# 1. Register the chat + embedding models (here: local Ollama models). +register_models("llama3.2", "nomic-embed-text") + +# 2. Inspect what curated sources are available. +println("Curated sources: ", collect(keys(CURATED_SOURCES))) + +# 3. Gather documents: curated FunSQL + OMOP docs, plus a live web search. +docs = ingest(; + sources = ["FunSQL.jl", "OMOP CDM"], + query = "OMOP CDM condition_occurrence table definition", + max_results = 3, +) +println("\nGathered $(length(docs)) documents:") +for d in docs + println(" [$(d.source)] $(first(d.title, 60)) ($(length(d.content)) chars)") +end + +# 4. Build a RAG index straight from the gathered text. +index = ingest_to_index(; docs = docs, embedder_kwargs = (model = "nomic-embed-text",)) + +# 5. Ask a question against the freshly ingested docs. +answer = generate_funsql_query( + index, + "nomic-embed-text", + "llama3.2", + "Context: {input_query}. Answer with a concise FunSQL query.", + "Write a FunSQL query that counts conditions per person.", +) +println("\nAnswer:\n", answer) +``` + +Expected shape of the output (content varies with the live sources and model): + +``` +Curated sources: ["JuliaHealth", "OMOP CDM", "OHDSI", "FunSQL.jl"] + +Gathered 8 documents: + [FunSQL.jl] FunSQL.jl (280 chars) + [FunSQL.jl] Guide (49284 chars) + ... + [web-search] OMOP CDM v5.4 (164111 chars) + +Answer: +From(:condition_occurrence) |> Group(Get.person_id) |> Select(...) +``` + +## Notes and tips + +- **Resilience.** Curated entries may list several candidate URLs; unreachable + ones are skipped, so a single dead link never fails the whole run. +- **Provenance.** Index chunks record their origin URL, so retrieved context is + traceable back to the source page. +- **Rate limits.** The keyless [`DuckDuckGoProvider`](@ref) is rate-limited and + best-effort. For heavy use, add a keyed provider via the + [`AbstractSearchProvider`](@ref) interface. +- **Offline pieces.** [`html_to_text`](@ref) needs no network and is handy to + test in isolation. + +See the [API reference](index.md) for full docstrings of every function above. +``` diff --git a/src/HealthLLM.jl b/src/HealthLLM.jl index 6213c08..7e3a91a 100644 --- a/src/HealthLLM.jl +++ b/src/HealthLLM.jl @@ -10,14 +10,25 @@ using Statistics include("utils.jl") include("database.jl") include("query.jl") +include("ingestion.jl") import .Utils: collect_files_with_extensions, write_combined_file, register_models, load_huggingface_model, HuggingFaceLoadResult, build_index_rag import .Database: store_embeddings_pgvector, validate_embeddings_inputs import .Query: generate_funsql_query +import .Ingestion: SourceDocument, SearchResult, + AbstractSearchProvider, DuckDuckGoProvider, + default_search_provider, web_search, + CURATED_SOURCES, fetch_url, html_to_text, fetch_curated, + ingest, ingest_to_index export PromptingTools, RAGTools export collect_files_with_extensions, write_combined_file, generate_funsql_query, build_index_rag, store_embeddings_pgvector, validate_embeddings_inputs, register_models, load_huggingface_model, HuggingFaceLoadResult +export SourceDocument, SearchResult, + AbstractSearchProvider, DuckDuckGoProvider, + default_search_provider, web_search, + CURATED_SOURCES, fetch_url, html_to_text, fetch_curated, + ingest, ingest_to_index end diff --git a/src/ingestion.jl b/src/ingestion.jl new file mode 100644 index 0000000..6669957 --- /dev/null +++ b/src/ingestion.jl @@ -0,0 +1,62 @@ +""" + Ingestion + +Pull documentation into a RAG index from two kinds of sources: + +1. **Curated docs** — a fixed, reviewed list of high-quality pages + (JuliaHealth, OMOP CDM, OHDSI, FunSQL.jl). See [`CURATED_SOURCES`](@ref). +2. **Web search** — a live query against a pluggable search backend for + anything not in the curated set. See [`AbstractSearchProvider`](@ref). + +## Data flow + + CURATED_SOURCES ─┐ + ├─▶ fetch_url ─▶ SourceDocument ─┐ + web_search ──────┘ ├─▶ ingest ─▶ ingest_to_index ─▶ RAG index + │ + (each hit's URL fetched) ────┘ + +## Layout (one concern per file, in include order) + +| File | Responsibility | +|---------------------------|--------------------------------------------------| +| `ingestion/types.jl` | `SourceDocument`, `SearchResult` data types | +| `ingestion/curated.jl` | `CURATED_SOURCES` registry | +| `ingestion/fetch.jl` | HTTP fetch + HTML→text (`fetch_url`, `html_to_text`, `fetch_curated`) | +| `ingestion/search.jl` | Search providers (`web_search`, `DuckDuckGoProvider`) | +| `ingestion/pipeline.jl` | Orchestration (`ingest`, `ingest_to_index`) | + +## Quickstart + +```julia +using HealthLLM + +register_models("llama3.2", "nomic-embed-text") +index = ingest_to_index(; sources=["FunSQL.jl", "OMOP CDM"], + query="OMOP condition_occurrence table", + embedder_kwargs=(model="nomic-embed-text",)) +answer = generate_funsql_query(index, "nomic-embed-text", "llama3.2", + "Context: {input_query}", "How do I join person and visit?") +``` +""" +module Ingestion + +using HTTP +using URIs +using RAGTools +using ..Utils + +export SourceDocument, SearchResult, + AbstractSearchProvider, DuckDuckGoProvider, + default_search_provider, web_search, + CURATED_SOURCES, fetch_url, html_to_text, fetch_curated, + ingest, ingest_to_index + +# Included in dependency order: later files use the types/functions above them. +include("ingestion/types.jl") +include("ingestion/curated.jl") +include("ingestion/fetch.jl") +include("ingestion/search.jl") +include("ingestion/pipeline.jl") + +end diff --git a/src/ingestion/curated.jl b/src/ingestion/curated.jl new file mode 100644 index 0000000..dab8f7f --- /dev/null +++ b/src/ingestion/curated.jl @@ -0,0 +1,32 @@ +""" + CURATED_SOURCES + +Registry of curated documentation sources as `name => Vector{url}`. + +Each entry lists one or more URLs for a documentation source. Raw Markdown +endpoints (e.g. `raw.githubusercontent.com`) are preferred because they need no +HTML cleaning; rendered pages are also accepted and stripped to text on fetch. + +URLs that fail to fetch (404, timeout, network error) are skipped, so entries +may safely list several candidate locations for robustness. Extend or override +this registry to point at your own curated set. +""" +const CURATED_SOURCES = Dict{String,Vector{String}}( + "JuliaHealth" => [ + "https://raw.githubusercontent.com/JuliaHealth/juliahealth.github.io/main/JuliaHealthBlog/posts/juliahealth-ecosystem/ecosystem.qmd", + "https://juliahealth.org/", + ], + "OMOP CDM" => [ + "https://raw.githubusercontent.com/OHDSI/CommonDataModel/main/README.md", + "https://ohdsi.github.io/CommonDataModel/cdm54.html", + ], + "OHDSI" => [ + "https://raw.githubusercontent.com/OHDSI/TheBookOfOhdsi/master/StandardizedVocabularies.Rmd", + "https://www.ohdsi.org/data-standardization/", + ], + "FunSQL.jl" => [ + "https://raw.githubusercontent.com/MechanicalRabbit/FunSQL.jl/master/docs/src/index.md", + "https://raw.githubusercontent.com/MechanicalRabbit/FunSQL.jl/master/docs/src/guide/index.md", + "https://raw.githubusercontent.com/MechanicalRabbit/FunSQL.jl/master/docs/src/examples/index.md", + ], +) diff --git a/src/ingestion/fetch.jl b/src/ingestion/fetch.jl new file mode 100644 index 0000000..5bfd72f --- /dev/null +++ b/src/ingestion/fetch.jl @@ -0,0 +1,111 @@ +const _DEFAULT_HEADERS = ["User-Agent" => "HealthLLM.jl-ingestion/0.1"] + +""" + html_to_text(html) -> String + +Strip HTML markup to readable plain text without an external parser. + +Removes `" => " ") + s = replace(s, r"(?is)" => " ") + s = replace(s, r"(?is)" => " ") + s = replace(s, r"(?s)" => " ") + s = replace(s, r"(?i)" => "\n") + s = replace(s, r"(?i)" => "\n") + s = replace(s, r"<[^>]+>" => " ") + for (pat, rep) in ( + " " => " ", "&" => "&", "<" => "<", ">" => ">", + """ => "\"", "'" => "'", "'" => "'", "—" => "—", + "–" => "–", "…" => "…", + ) + s = replace(s, pat => rep) + end + s = replace(s, r"&#(\d+);" => m -> string(Char(parse(Int, m[3:end-1])))) + s = replace(s, r"[ \t]+" => " ") + s = replace(s, r"\n[ \t]+" => "\n") + s = replace(s, r"\n{3,}" => "\n\n") + return strip(s) +end + +_looks_like_markup(url, ctype) = + occursin("html", lowercase(ctype)) || + (isempty(ctype) && !occursin(r"\.(md|markdown|txt|rst|csv|json)$"i, url)) + +""" + fetch_url(url; timeout=30, max_bytes=5_000_000) -> String + +Fetch `url` over HTTP(S) and return cleaned text. + +Content served as HTML is run through [`html_to_text`](@ref); Markdown/plain +text is returned as-is. Responses larger than `max_bytes` are truncated. Throws +on network errors or non-2xx status. + +# Keywords +- `timeout=30`: Per-request timeout in seconds. +- `max_bytes=5_000_000`: Byte cap on the response body before cleaning. +""" +function fetch_url(url::AbstractString; timeout::Real=30, max_bytes::Integer=5_000_000) + resp = HTTP.get(String(url); headers=_DEFAULT_HEADERS, readtimeout=timeout, + redirect=true, retry=false, status_exception=true) + ctype = HTTP.header(resp, "Content-Type", "") + body = String(resp.body) + length(body) > max_bytes && (body = body[1:max_bytes]) + return _looks_like_markup(url, ctype) ? html_to_text(body) : strip(body) +end + +function _title_from(content::AbstractString, url::AbstractString) + m = match(r"(?m)^#{1,6}\s+(.+)$", content) + m !== nothing && return strip(m[1]) + for line in eachline(IOBuffer(content)) + !isempty(strip(line)) && return first(strip(line), 120) + end + return String(url) +end + +""" + fetch_curated(names=keys(CURATED_SOURCES); timeout=30, min_length=200) -> Vector{SourceDocument} + +Fetch the curated documentation sources named in `names` from [`CURATED_SOURCES`](@ref). + +For each source, its candidate URLs are tried in order and every one that +fetches successfully and yields at least `min_length` characters is kept. URLs +that error are skipped with a warning. Returns a flat vector of +[`SourceDocument`](@ref). + +# Example + +```julia +docs = fetch_curated(["FunSQL.jl", "OMOP CDM"]) +``` +""" +function fetch_curated(names=keys(CURATED_SOURCES); timeout::Real=30, min_length::Integer=200) + docs = SourceDocument[] + for name in names + haskey(CURATED_SOURCES, name) || + (@warn "Unknown curated source, skipping: $name"; continue) + for url in CURATED_SOURCES[name] + try + content = fetch_url(url; timeout=timeout) + if length(content) >= min_length + push!(docs, SourceDocument(name, url, _title_from(content, url), content)) + else + @debug "Skipping short/empty content ($(length(content)) chars) from $name: $url" + end + catch err + @warn "Failed to fetch curated URL from $name, skipping: $url ($err)" + end + end + end + return docs +end diff --git a/src/ingestion/pipeline.jl b/src/ingestion/pipeline.jl new file mode 100644 index 0000000..91060ba --- /dev/null +++ b/src/ingestion/pipeline.jl @@ -0,0 +1,97 @@ +# The orchestration layer: gather documents (curated + search) and, optionally, +# turn them straight into a RAG index. + +""" + ingest(; sources=keys(CURATED_SOURCES), query=nothing, provider=default_search_provider(), + max_results=5, fetch_search_content=true, timeout=30, min_length=200) -> Vector{SourceDocument} + +Gather documents from curated sources and/or a live web search into a single +vector of [`SourceDocument`](@ref)s. + +# Keywords +- `sources`: Curated source names to fetch (see [`CURATED_SOURCES`](@ref)). Pass + `String[]`/`nothing` to skip curated fetching and search only. +- `query`: If given, also run a web search for `query` via `provider`. +- `provider`: Search backend (default: [`default_search_provider`](@ref)). +- `max_results`: Max web-search hits to keep. +- `fetch_search_content`: When `true`, fetch each result URL for full text; when + `false`, use only the provider-supplied snippet/content. +- `timeout`, `min_length`: Passed through to fetching/filtering. + +# Example + +```julia +# Curated docs plus a targeted web search +docs = ingest(; sources=["FunSQL.jl"], query="FunSQL From Where Select example") +``` +""" +function ingest(; sources=keys(CURATED_SOURCES), query::Union{Nothing,AbstractString}=nothing, + provider::AbstractSearchProvider=default_search_provider(), max_results::Integer=5, + fetch_search_content::Bool=true, timeout::Real=30, min_length::Integer=200) + + docs = SourceDocument[] + + # 1. Curated docs (skipped when `sources` is empty/nothing). + if sources !== nothing && !isempty(collect(sources)) + append!(docs, fetch_curated(sources; timeout=timeout, min_length=min_length)) + end + + # 2. Live web search (only when a `query` is given). + if query !== nothing + for hit in web_search(provider, query; max_results=max_results) + content = hit.content + if fetch_search_content && !isempty(hit.url) + try + content = fetch_url(hit.url; timeout=timeout) + catch err + @warn "Failed to fetch search result, using snippet: $(hit.url) ($err)" + end + end + isempty(strip(content)) && continue + push!(docs, SourceDocument("web-search", hit.url, hit.title, content)) + end + end + + return docs +end + +""" + ingest_to_index(cfg=RAGTools.SimpleIndexer(); sources=keys(CURATED_SOURCES), query=nothing, + provider=default_search_provider(), max_results=5, fetch_search_content=true, + timeout=30, min_length=200, embedder_kwargs=NamedTuple(), docs=nothing) -> index + +Ingest curated docs and/or web-search results and build a RAG index directly from +the in-memory text — no intermediate files. + +Documents are chunked with `RAGTools.TextChunker()` and their URLs are recorded as +chunk sources for provenance. Pass a precomputed `docs` vector (from [`ingest`](@ref)) +to reuse it instead of fetching again. The returned index is ready for +`generate_funsql_query`. + +# Example + +```julia +register_models("llama3.2", "nomic-embed-text") +index = ingest_to_index(; sources=["FunSQL.jl", "OMOP CDM"], + query="OMOP condition_occurrence table", + embedder_kwargs=(model="nomic-embed-text",)) +answer = generate_funsql_query(index, "nomic-embed-text", "llama3.2", + "Context: {input_query}", "How do I join person and visit?") +``` +""" +function ingest_to_index(cfg=RAGTools.SimpleIndexer(); + docs::Union{Nothing,AbstractVector{SourceDocument}}=nothing, + embedder_kwargs=NamedTuple(), kwargs...) + + documents = docs === nothing ? ingest(; kwargs...) : docs + isempty(documents) && + throw(ArgumentError("No documents ingested; nothing to index. Check sources/query and network.")) + + contents = [d.content for d in documents] + provenance = [isempty(d.url) ? d.source : d.url for d in documents] + + return RAGTools.build_index(cfg, contents; + chunker=RAGTools.TextChunker(), + chunker_kwargs=(sources=provenance,), + embedder_kwargs=embedder_kwargs) +end diff --git a/src/ingestion/search.jl b/src/ingestion/search.jl new file mode 100644 index 0000000..b5589b0 --- /dev/null +++ b/src/ingestion/search.jl @@ -0,0 +1,61 @@ +# Pluggable web-search backends. One provider is wired up today (DuckDuckGo); +# add another by defining a struct <: AbstractSearchProvider and a `web_search` +# method for it — nothing else in the pipeline needs to change. + +""" + AbstractSearchProvider + +Interface for web-search backends. Implement [`web_search`](@ref) for a concrete +provider to plug it into ingestion. Only [`DuckDuckGoProvider`](@ref) is wired up +for now; add more providers as subtypes later. +""" +abstract type AbstractSearchProvider end + +""" + DuckDuckGoProvider() + +Keyless web search via the DuckDuckGo HTML endpoint. No API key required. +The endpoint is rate-limited and results are best-effort scraped from the HTML. +""" +struct DuckDuckGoProvider <: AbstractSearchProvider end + +""" + default_search_provider() -> AbstractSearchProvider + +Return the default search backend. Currently [`DuckDuckGoProvider`](@ref) (keyless). +""" +default_search_provider() = DuckDuckGoProvider() + +""" + web_search(query; max_results=5) -> Vector{SearchResult} + web_search(provider, query; max_results=5) -> Vector{SearchResult} + +Run `query` against `provider` and return up to `max_results` [`SearchResult`](@ref)s. +`provider` defaults to [`default_search_provider`](@ref). + +# Example + +```julia +hits = web_search("OMOP CDM person table columns"; max_results=3) +``` +""" +web_search(query::AbstractString; kwargs...) = web_search(default_search_provider(), query; kwargs...) + +function web_search(::DuckDuckGoProvider, query::AbstractString; max_results::Integer=5) + url = string("https://html.duckduckgo.com/html/?q=", URIs.escapeuri(String(query))) + resp = HTTP.get(url; headers=_DEFAULT_HEADERS, readtimeout=30, status_exception=true) + body = String(resp.body) + results = SearchResult[] + # Each hit is an title element. + for m in eachmatch(r"class=\"result__a\"[^>]*href=\"(.*?)\"[^>]*>(.*?)"s, body) + href, title = m[1], html_to_text(m[2]) + # DuckDuckGo wraps the real destination in a redirect: .../l/?uddg=. + real = href + rm = match(r"uddg=([^&]+)", href) + rm !== nothing && (real = URIs.unescapeuri(rm[1])) + startswith(real, "//") && (real = "https:" * real) + push!(results, SearchResult(String(title), String(real), "", NaN)) + length(results) >= max_results && break + end + return results +end diff --git a/src/ingestion/types.jl b/src/ingestion/types.jl new file mode 100644 index 0000000..afa642c --- /dev/null +++ b/src/ingestion/types.jl @@ -0,0 +1,37 @@ +# Data types shared across the ingestion pipeline. + +""" + SourceDocument + +A single ingested document with provenance. + +# Fields +- `source::String`: Logical source name (e.g. `"FunSQL.jl"`, `"web-search"`). +- `url::String`: URL the content came from. +- `title::String`: Human-readable title (best effort; may be empty). +- `content::String`: Cleaned plain-text content. +""" +struct SourceDocument + source::String + url::String + title::String + content::String +end + +""" + SearchResult + +A single web-search hit returned by an [`AbstractSearchProvider`](@ref). + +# Fields +- `title::String`: Result title. +- `url::String`: Result URL. +- `content::String`: Snippet or extracted content (may be empty depending on provider). +- `score::Float64`: Provider-supplied relevance score, or `NaN` if unavailable. +""" +struct SearchResult + title::String + url::String + content::String + score::Float64 +end diff --git a/test/IngestionTest.jl b/test/IngestionTest.jl new file mode 100644 index 0000000..8410c76 --- /dev/null +++ b/test/IngestionTest.jl @@ -0,0 +1,51 @@ +using HealthLLM +using HealthLLM.Ingestion: SourceDocument, SearchResult, + AbstractSearchProvider, DuckDuckGoProvider, + default_search_provider, web_search, + CURATED_SOURCES, html_to_text, _title_from + +# These tests exercise the offline/pure logic only. Live fetching and search +# depend on the network and provider credentials and are not run in CI. + +@testset "Ingestion" begin + @testset "html_to_text" begin + @test html_to_text("

Hello world

") == "Hello world" + @test html_to_text("
a
b
") == "a\nb" + # script/style content is dropped + @test !occursin("alert", html_to_text("text")) + @test occursin("text", html_to_text("text")) + # entity decoding + @test occursin("&", html_to_text("A & B")) + @test occursin("<", html_to_text("<tag>")) + @test html_to_text("AB") == "AB" + # whitespace collapse + @test !occursin("\n\n\n", html_to_text("a



b")) + end + + @testset "curated registry" begin + @test haskey(CURATED_SOURCES, "FunSQL.jl") + @test haskey(CURATED_SOURCES, "OMOP CDM") + @test haskey(CURATED_SOURCES, "OHDSI") + @test haskey(CURATED_SOURCES, "JuliaHealth") + @test all(!isempty, values(CURATED_SOURCES)) + @test all(u -> startswith(u, "http"), Iterators.flatten(values(CURATED_SOURCES))) + end + + @testset "_title_from" begin + @test _title_from("# Big Title\n\nbody", "u") == "Big Title" + @test _title_from("no heading here\nmore", "u") == "no heading here" + @test _title_from(" ", "http://x") == "http://x" + end + + @testset "search provider" begin + @test DuckDuckGoProvider() isa AbstractSearchProvider + @test default_search_provider() isa DuckDuckGoProvider + end + + @testset "SourceDocument / SearchResult constructors" begin + d = SourceDocument("s", "u", "t", "c") + @test d.source == "s" && d.content == "c" + r = SearchResult("t", "u", "c", 0.5) + @test r.score == 0.5 + end +end diff --git a/test/runtests.jl b/test/runtests.jl index 2b8d539..0a5db14 100644 --- a/test/runtests.jl +++ b/test/runtests.jl @@ -11,6 +11,7 @@ ti = time() include("hf_model_test.jl") include("hf_load_tests.jl") include("FunSQLTest.jl") + include("IngestionTest.jl") end ti = time() - ti From 7ece173dd6053be6d690672efb114c5ca4055b2b Mon Sep 17 00:00:00 2001 From: paramthakkar123 Date: Sat, 18 Jul 2026 09:57:11 +0530 Subject: [PATCH 02/10] Add document chunking strategies with provenance tracking --- src/ingestion.jl | 3 + src/ingestion/chunk.jl | 356 ++++++++++++++++++++++++++++++++++++++ src/ingestion/pipeline.jl | 37 +++- test/IngestionTest.jl | 121 ++++++++++++- 4 files changed, 508 insertions(+), 9 deletions(-) create mode 100644 src/ingestion/chunk.jl diff --git a/src/ingestion.jl b/src/ingestion.jl index 6669957..f71f897 100644 --- a/src/ingestion.jl +++ b/src/ingestion.jl @@ -50,6 +50,8 @@ export SourceDocument, SearchResult, AbstractSearchProvider, DuckDuckGoProvider, default_search_provider, web_search, CURATED_SOURCES, fetch_url, html_to_text, fetch_curated, + Chunk, AbstractChunkStrategy, RecursiveChunk, HeaderChunk, RecordChunk, FixedSizeChunk, + chunk, chunk_document, chunk_provenance, default_strategy, load_funsql_examples, ingest, ingest_to_index # Included in dependency order: later files use the types/functions above them. @@ -57,6 +59,7 @@ include("ingestion/types.jl") include("ingestion/curated.jl") include("ingestion/fetch.jl") include("ingestion/search.jl") +include("ingestion/chunk.jl") include("ingestion/pipeline.jl") end diff --git a/src/ingestion/chunk.jl b/src/ingestion/chunk.jl new file mode 100644 index 0000000..9e3889f --- /dev/null +++ b/src/ingestion/chunk.jl @@ -0,0 +1,356 @@ +using PromptingTools: recursive_splitter +using JSON3 + +""" + Chunk + +One retrieval chunk plus grounding metadata. + +# Fields +- `text::String`: The chunk text. +- `metadata::Dict{Symbol,Any}`: Grounding info such as `:heading` (the parent + Markdown/section heading, e.g. the OMOP table a chunk describes), `:group` + (FunSQL example group), `:source`, `:url`, and `:title`. Populated by the + strategy and enriched with the document's provenance by [`chunk_document`](@ref). +""" +struct Chunk + text::String + metadata::Dict{Symbol,Any} +end +Chunk(text::AbstractString) = Chunk(String(text), Dict{Symbol,Any}()) + +""" + AbstractChunkStrategy + +Supertype for chunking strategies. A strategy turns one document's text into a +vector of [`Chunk`](@ref)s via [`chunk`](@ref)`(strategy, text)`. + +Concrete strategies: +- [`RecursiveChunk`](@ref) — size-bounded recursive split (default for prose). +- [`HeaderChunk`](@ref) — one chunk per Markdown heading section (OMOP tables). +- [`RecordChunk`](@ref) — one chunk per JSONL record (FunSQL examples). +- [`FixedSizeChunk`](@ref) — fixed character windows with overlap. +""" +abstract type AbstractChunkStrategy end + +""" + chunk(strategy::AbstractChunkStrategy, text::AbstractString) -> Vector{Chunk} + +Split `text` into atomic [`Chunk`](@ref)s according to `strategy`. Fenced code +blocks are never split mid-expression, and each chunk carries the parent heading +it came from where the strategy can determine one. Empty/whitespace-only chunks +are dropped. +""" +function chunk end + +const _RECURSIVE_SEPARATORS = ["\n## ", "\n### ", "\n\n", "\n", ". ", " "] + +""" + RecursiveChunk(; max_length=1024, separators=_RECURSIVE_SEPARATORS) + +Split text into heading sections, then split each section to at or below +`max_length` characters, preferring the earliest separator that fits. Fenced code +blocks (```` ``` ````/`~~~`) are kept whole regardless of length so a FunSQL +snippet is never cut mid-expression. Each resulting chunk records its parent +`:heading`. The right default for prose docs (FunSQL guide, OHDSI book, +JuliaHealth), where a size cut degrades gracefully. +""" +Base.@kwdef struct RecursiveChunk <: AbstractChunkStrategy + max_length::Int = 1024 + separators::Vector{String} = _RECURSIVE_SEPARATORS +end + +function chunk(s::RecursiveChunk, text::AbstractString) + out = Chunk[] + for (heading, body) in _sections(String(text)) + for piece in _split_preserving_code(body, s.separators, s.max_length) + _push_chunk!(out, piece, heading) + end + end + return out +end + +""" + HeaderChunk(; min_level=1, max_level=6, max_length=4096) + +Split Markdown on ATX heading boundaries (`#`..`######`), emitting one chunk per +heading together with the body beneath it, up to the next heading of the same or +higher level. This keeps an OMOP table definition (its name heading plus the full +column/foreign-key list) in a single chunk, tagged with `:heading`. + +A section longer than `max_length` characters is further split with the same +code-preserving splitter as [`RecursiveChunk`](@ref), so no single chunk blows +past embedding limits and no code block is broken. Text before the first heading +becomes its own leading chunk. Heading lines inside fenced code blocks are +ignored. `min_level`/`max_level` bound which heading depths start a new section +(e.g. `min_level=2` ignores the document title `#` and breaks on `##` headings). +""" +Base.@kwdef struct HeaderChunk <: AbstractChunkStrategy + min_level::Int = 1 + max_level::Int = 6 + max_length::Int = 4096 +end + +function chunk(s::HeaderChunk, text::AbstractString) + out = Chunk[] + for (heading, body) in _sections(String(text); min_level=s.min_level, max_level=s.max_level) + pieces = length(body) > s.max_length ? + _split_preserving_code(body, _RECURSIVE_SEPARATORS, s.max_length) : [body] + for piece in pieces + _push_chunk!(out, piece, heading) + end + end + return out +end + +""" + RecordChunk(; fields=["query", "description", "sql_query", "response"], + labels=["Query", "Description", "SQL", "FunSQL"]) + +Treat the input as JSON Lines (one JSON object per line) and emit exactly one +chunk per record — the atomic retrieval unit for the FunSQL example dataset in +`FunSQLQueries/`. + +`fields` are pulled from each record in order and rendered as +`\"