Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion AGENTS.md
Original file line number Diff line number Diff line change
Expand Up @@ -21,7 +21,7 @@ Memgraph graph.
| `integrations/langchain-memgraph/` | LangChain graph store, QA chain, toolkit. |
| `integrations/mcp-memgraph/` | MCP server exposing Memgraph to LLMs. |
| `integrations/lightrag-memgraph/` | LightRAG storage backends (KV/vector/doc-status/graph) on Memgraph. |
| `unstructured2graph/` | Chunks unstructured input (files/URLs/text) and hands chunks to LightRAG for entity extraction. Outside the Context Graph family but shares its testing conventions. |
| `unstructured2graph/` | Chunks unstructured input (files/URLs/text) and runs entity/relation extraction through a pluggable `ExtractionBackend` (`LightRAGBackend` by default; `GLiNER2Backend` for local, LLM-free extraction, manual-install only -- see its module docstring). Outside the Context Graph family but shares its testing conventions. |
| `agents/sql2graph/` | MySQL/Postgres → Memgraph migration agent. Has its own `uv.lock`/`.python-version`; run it with `cd agents/sql2graph && uv run main.py`. |
| `context-graph/` | The Context Graph family — see below. |
| `scripts/dev-memgraph.sh` | Local dev lifecycle: exploration Memgraph + isolated test Memgraph for the context-graph family and unstructured2graph. |
Expand Down
4 changes: 2 additions & 2 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -79,7 +79,7 @@ Transform PDFs, URLs, and documents into queryable knowledge graphs:
import asyncio
from memgraph_toolbox.api.memgraph import Memgraph
from lightrag_memgraph import MemgraphLightRAGWrapper
from unstructured2graph import from_unstructured
from unstructured2graph import LightRAGBackend, from_unstructured


async def main():
Expand All @@ -92,7 +92,7 @@ async def main():
await from_unstructured(
sources=["https://example.com/doc.pdf", "./local_file.md"],
memgraph=memgraph,
lightrag_wrapper=lightrag,
extraction_backend=LightRAGBackend(lightrag),
link_chunks=True,
enforce_ontology=True, # promote entity_type to real labels (:Person, :Organization, ...)
)
Expand Down
4 changes: 2 additions & 2 deletions context-graph/sessions-graph/src/sessions_graph/core.py
Original file line number Diff line number Diff line change
Expand Up @@ -345,7 +345,7 @@ async def reconcile_session(
actions_graph = _ActionsGraph(self._db)

try:
from unstructured2graph import from_texts
from unstructured2graph import LightRAGBackend, from_texts
except ImportError as exc:
msg = "unstructured2graph is required for reconcile_session; install sessions-graph[reconciliation]"
raise ImportError(msg) from exc
Expand Down Expand Up @@ -390,7 +390,7 @@ async def reconcile_session(
grouped_chunks = await from_texts(
[combined_text],
memgraph=self._db,
lightrag_wrapper=lightrag_wrapper,
extraction_backend=LightRAGBackend(lightrag_wrapper),
entity_workspace=entity_workspace,
promote_labels=promote_labels,
enforce_ontology=enforce_ontology,
Expand Down
14 changes: 14 additions & 0 deletions integrations/lightrag-memgraph/src/lightrag_memgraph/core.py
Original file line number Diff line number Diff line change
Expand Up @@ -101,6 +101,20 @@ def get_lightrag(self) -> LightRAG:
raise RuntimeError("LightRAG not initialized. Call initialize() first.")
return self.rag

@property
def workspace(self) -> str:
"""The Memgraph label LightRAG writes every extracted entity under.

Exposed directly (rather than making callers reach through
get_lightrag().chunk_entity_relation_graph) since it's the one piece
of LightRAG's internal state consumers outside this package actually
need -- e.g. unstructured2graph's LightRAGBackend.workspace_label.

Raises:
RuntimeError: if initialize() hasn't been called yet.
"""
return self.get_lightrag().chunk_entity_relation_graph.workspace

# https://github.com/HKUDS/LightRAG/blob/main/lightrag/lightrag.py
async def ainsert(self, **kwargs) -> None:
"""
Expand Down
17 changes: 16 additions & 1 deletion integrations/lightrag-memgraph/tests/test_core.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,7 @@
from __future__ import annotations

import os
from unittest.mock import AsyncMock, patch
from unittest.mock import AsyncMock, MagicMock, patch

import pytest
from lightrag.llm.openai import gpt_4o_mini_complete, openai_embed
Expand Down Expand Up @@ -168,3 +168,18 @@ async def test_afinalize_resets_shared_data_even_if_finalize_storages_raises():
await wrapper.afinalize()

mock_finalize_share_data.assert_called_once()


def test_workspace_reads_chunk_entity_relation_graph_workspace():
wrapper = MemgraphLightRAGWrapper()
wrapper.rag = MagicMock()
wrapper.rag.chunk_entity_relation_graph.workspace = "tenant-42"

assert wrapper.workspace == "tenant-42"


def test_workspace_raises_before_initialize():
wrapper = MemgraphLightRAGWrapper()

with pytest.raises(RuntimeError, match="not initialized"):
_ = wrapper.workspace
10 changes: 10 additions & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -36,6 +36,16 @@ constraint-dependencies = [
"python-multipart>=0.0.22",
"requests>=2.33.0",
"starlette>=0.49.1",
# Fixes CVE-2026-1839, an RCE in Trainer's checkpoint loading -- see
# GHSA-69w3-r845-3855. gliner2[local] (used by unstructured2graph's
# optional GLiNER2Backend) hard-pins transformers<5, which conflicts with
# this floor -- so gliner2 is deliberately NOT a pyproject.toml extra of
# unstructured2graph (would force uv's workspace-wide lock to satisfy
# both constraints at once, which is impossible). It's a manual
# `pip install 'gliner2[local]>=2.0.0'` in your own environment instead,
# isolated from this floor rather than weakening it. See
# unstructured2graph/src/unstructured2graph/gliner2_backend.py's module
# docstring.
"transformers>=5.0.0rc3",
"urllib3>=2.6.3",
"wheel>=0.46.2",
Expand Down
87 changes: 73 additions & 14 deletions unstructured2graph/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,7 @@ Convert unstructured documents into knowledge graphs within [Memgraph](https://m
**unstructured2graph** enables you to transform any unstructured data (PDFs, URLs, documents) into a graph database, powering Graph Retrieval-Augmented Generation (GraphRAG) applications. It combines:

- **[Unstructured](https://github.com/Unstructured-IO/unstructured)** - Parse and chunk diverse document formats
- **[LightRAG](https://github.com/HKUDS/LightRAG)** - Extract entities and relationships using LLMs
- A pluggable extraction backend - **[LightRAG](https://github.com/HKUDS/LightRAG)** (LLM-based) or **[GLiNER2](https://github.com/fastino-ai/GLiNER2)** (local, LLM-free) - to extract entities and relationships
- **[Memgraph](https://memgraph.com/)** - Store and query your knowledge graph

## Installation
Expand All @@ -26,13 +26,37 @@ For full document support (PDF, DOCX, etc.):
pip install -e ".[all-docs]"
```

For the local, LLM-free GLiNER2 extraction backend, install `gliner2` manually
(deliberately *not* a `pyproject.toml` extra of this package -- `gliner2[local]`
hard-pins `transformers<5`, which conflicts with this monorepo's workspace-wide
`transformers>=5.0.0rc3` security floor; see `gliner2_backend.py`'s module
docstring for the full reasoning):

```bash
pip install 'gliner2[local]>=2.0.0'
```

## Choosing an extraction backend

Entity/relation extraction is pluggable behind an `ExtractionBackend`. Two are provided:

| | `LightRAGBackend` (default choice) | `GLiNER2Backend` |
|---|---|---|
| **Extraction** | LLM-based (via LightRAG) | Local model, no LLM |
| **Cost / network** | Per-call LLM cost, needs `OPENAI_API_KEY` (or another configured LLM) | Free, fully offline after the model download |
| **Cross-chunk coreference** | Yes — LightRAG's LLM normalizes mentions, so "Apple" in two chunks can merge into one entity | No — entity identity is scoped to `(chunk, entity_type, normalized text)`; the same entity mentioned in two chunks becomes two nodes |
| **Relation edges** | One generic `:DIRECTED` edge type (LightRAG's own convention) | One Cypher edge type per relation label (e.g. `:works_for`), drawn from the ontology's `relation_types` |
| **Relation vocabulary** | Open-ended (whatever the LLM extracts) | Closed — only extracts relations named in `Ontology.relation_types` |

Both write entities under a **workspace** label with an `entity_type` property and a `file_path` property equal to the source chunk's hash — the rest of the pipeline (chunk-to-entity linking, ontology-gated label promotion) works identically regardless of backend.

## Quick Start

```python
import asyncio
from memgraph_toolbox.api.memgraph import Memgraph
from lightrag_memgraph import MemgraphLightRAGWrapper
from unstructured2graph import from_unstructured
from unstructured2graph import LightRAGBackend, from_unstructured


async def main():
Expand All @@ -45,7 +69,7 @@ async def main():
await from_unstructured(
sources=["https://example.com/doc.pdf", "./local_file.md"],
memgraph=memgraph,
lightrag_wrapper=lightrag,
extraction_backend=LightRAGBackend(lightrag),
link_chunks=True, # create NEXT relationships between chunks
enforce_ontology=True, # promote entity_type to real labels (:Person, :Organization, ...)
)
Expand All @@ -57,6 +81,26 @@ asyncio.run(main())

The `Chunk.hash` uniqueness constraint is created for you inside `from_unstructured()` / `from_texts()` — no manual index step is needed.

### Using GLiNER2 instead (local, no LLM)

```python
from memgraph_toolbox.api.memgraph import Memgraph
from unstructured2graph import from_unstructured
from unstructured2graph.gliner2_backend import GLiNER2Backend

memgraph = Memgraph(user_agent="unstructured2graph")
backend = GLiNER2Backend() # downloads fastino/gliner2.5-base-v1 on first use

await from_unstructured(
sources=["./local_file.md"],
memgraph=memgraph,
extraction_backend=backend,
enforce_ontology=True,
)
```

`GLiNER2Backend` is not imported by `unstructured2graph`'s top-level package (which never requires the optional `gliner2` dependency) — import it from `unstructured2graph.gliner2_backend` directly.

### Ingesting raw text

For in-memory strings (no file or URL), use `from_texts`. It returns one `Chunk` group per input string, so you can trace an output chunk back to the text that produced it:
Expand All @@ -67,7 +111,7 @@ from unstructured2graph import from_texts
grouped = await from_texts(
texts=["Ada Lovelace collaborated with Charles Babbage in London."],
memgraph=memgraph,
lightrag_wrapper=lightrag,
extraction_backend=LightRAGBackend(lightrag),
enforce_ontology=True,
)
```
Expand All @@ -83,7 +127,7 @@ grouped = await from_texts(

## Entity typing / ontology

LightRAG writes every extracted entity under a single **workspace** label (default `base`) with its type only as an `entity_type` *property* — so out of the box you get `(:base {entity_type: "person"})`, not `(:Person)`. unstructured2graph can promote that type into a real Memgraph label. Two independent, opt-in flags on `from_unstructured()` / `from_texts()` (both default `False`):
An extraction backend writes every extracted entity under a single **workspace** label (default `base` for LightRAG, `gliner2` for GLiNER2) with its type only as an `entity_type` *property* — so out of the box you get `(:base {entity_type: "person"})`, not `(:Person)`. unstructured2graph can promote that type into a real Memgraph label. Two independent, opt-in flags on `from_unstructured()` / `from_texts()` (both default `False`):

| Flag | Behavior |
|---|---|
Expand All @@ -100,13 +144,16 @@ entity_types:
description: Human individuals, real or fictional
- label: Organization
description: Companies, institutions, government bodies, groups
relation_types: # optional -- read directly by GLiNER2Backend; no LightRAG equivalent
- label: works_for
description: Employment relationship between a person and an organization
```

```python
await from_unstructured(
sources=["./local_file.pdf"],
memgraph=memgraph,
lightrag_wrapper=lightrag,
extraction_backend=LightRAGBackend(lightrag),
enforce_ontology=True,
ontology_path="my_ontology.yaml", # omit to use the bundled default
)
Expand All @@ -125,14 +172,17 @@ await lightrag.initialize(
await from_unstructured(..., enforce_ontology=True, ontology_path="my_ontology.yaml")
```

`GLiNER2Backend` doesn't need this step — passing it the same `Ontology` (via its `ontology=` constructor argument) already steers extraction directly, entity types and relation types both.

## Key Features

| Feature | Description |
| ------------------------ | ----------------------------------------------------------------- |
| **Multi-format parsing** | PDFs, URLs, HTML, Markdown, DOCX, and more via Unstructured |
| **Automatic chunking** | Smart document chunking with configurable options |
| **Entity extraction** | LLM-powered entity and relationship extraction via LightRAG |
| **Pluggable extraction** | LLM-powered extraction via LightRAG, or local/offline via GLiNER2 |
| **Typed entities** | Promote `entity_type` to real labels (`:Person`, ...), optionally gated by an ontology |
| **Typed relations** | GLiNER2Backend writes relations as per-label edges (e.g. `:works_for`), gated by the same ontology |
| **Vector search** | Built-in support for embedding generation and vector indices |
| **GraphRAG queries** | Combine vector search with graph traversal for enhanced retrieval |

Expand All @@ -143,22 +193,29 @@ await from_unstructured(..., enforce_ontology=True, ontology_path="my_ontology.y
- `parse_source(source, partition_kwargs=None)` — parse a single file or URL into a list of `Chunk`s
- `parse_text(text, partition_kwargs=None)` — chunk a raw in-memory string (no file/URL involved)
- `make_chunks(sources, partition_kwargs=None)` — process multiple sources into `ChunkedDocument` objects
- `from_unstructured(sources, memgraph, lightrag_wrapper=None, only_chunks=False, link_chunks=False, entity_workspace=None, partition_kwargs=None, promote_labels=False, enforce_ontology=False, ontology_path=None)` — full ingestion for files/URLs; returns `list[list[Chunk]]`, one group per source
- `from_texts(texts, memgraph, lightrag_wrapper=None, only_chunks=False, entity_workspace=None, promote_labels=False, enforce_ontology=False, ontology_path=None)` — full ingestion for raw strings; returns `list[list[Chunk]]`, one group per input text (no `link_chunks`/`partition_kwargs`)
- `from_unstructured(sources, memgraph, extraction_backend=None, only_chunks=False, link_chunks=False, entity_workspace=None, partition_kwargs=None, promote_labels=False, enforce_ontology=False, ontology_path=None)` — full ingestion for files/URLs; returns `list[list[Chunk]]`, one group per source
- `from_texts(texts, memgraph, extraction_backend=None, only_chunks=False, entity_workspace=None, promote_labels=False, enforce_ontology=False, ontology_path=None)` — full ingestion for raw strings; returns `list[list[Chunk]]`, one group per input text (no `link_chunks`/`partition_kwargs`)

### Extraction backends

- `ExtractionBackend` — the protocol both backends below satisfy; `workspace_label` (the Memgraph label entities are written under) and `async aingest_chunk(memgraph, chunk)`
- `LightRAGBackend(wrapper)` — wraps an initialized `MemgraphLightRAGWrapper`
- `unstructured2graph.gliner2_backend.GLiNER2Backend(model_name=..., ontology=None, workspace="gliner2", model=None, entity_confidence_threshold=None, relation_confidence_threshold=None)` — local GLiNER2 model; requires `gliner2` installed manually (see Installation above), not exported from the top-level package

### Ontology

- `load_ontology(path)` → `Ontology` — parse an ontology YAML file
- `Ontology`, `EntityType` — the vocabulary types; `Ontology.addon_params()` renders LightRAG extraction guidance
- `DEFAULT_ONTOLOGY`, `DEFAULT_ONTOLOGY_PATH` — the bundled default vocabulary
- `Ontology`, `EntityType`, `RelationType` — the vocabulary types; `Ontology.addon_params()` renders LightRAG extraction guidance (entity types only)
- `DEFAULT_ONTOLOGY`, `DEFAULT_ONTOLOGY_PATH` — the bundled default vocabulary (entity-only)
- `promote_entity_types_to_labels(memgraph, workspace_label, ontology)` — the ontology-gated promotion (what `enforce_ontology` calls)
- `promote_all_entity_types_to_labels(memgraph, workspace_label)` — unrestricted promotion (what `promote_labels` calls)

### Graph Operations

- `create_nodes_from_list(memgraph, nodes, label, batch_size, merge_key=None)` — batch insert; pass `merge_key` to upsert (`MERGE`) instead of `CREATE`
- `connect_chunks_to_entities(memgraph, chunk_label, entity_label)` — link entities to source chunks (`entity_label` is the LightRAG workspace label, e.g. `base`)
- `connect_chunks_to_entities(memgraph, chunk_label, entity_label)` — link entities to source chunks (`entity_label` is the extraction backend's workspace label, e.g. `base` or `gliner2`)
- `link_nodes_in_order(memgraph, find_label, find_property, from_to_dicts, create_edge_type)` — create sequential relationships between nodes
- `upsert_typed_relationships(memgraph, node_label, match_key, relationships_by_type)` — upsert relationships under distinct Cypher relationship types (what `GLiNER2Backend` uses for typed relation edges)
- `create_vector_search_index(memgraph, label, property, dimension=384, index_name="vs_name")` — create a vector index for similarity search
- `compute_embeddings(memgraph, label)` — generate embeddings for nodes

Expand All @@ -173,10 +230,12 @@ For detailed usage examples and getting started guides, check out the official d
- Python 3.10+
- Memgraph database instance

### LLM API Key
### LLM API Key (LightRAGBackend only)

This library uses LightRAG for entity and relationship extraction, which requires an LLM API key. Set your OpenAI API key as an environment variable:
`LightRAGBackend` uses LightRAG for entity and relationship extraction, which requires an LLM API key. Set your OpenAI API key as an environment variable:

```bash
export OPENAI_API_KEY="your-api-key"
```

`GLiNER2Backend` needs no API key — it runs entirely locally.
40 changes: 40 additions & 0 deletions unstructured2graph/examples/gliner2_loading.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,40 @@
import asyncio
import logging
import os

import sources as SOURCES

from memgraph_toolbox.api.memgraph import Memgraph
from unstructured2graph import from_unstructured
from unstructured2graph.gliner2_backend import GLiNER2Backend

SCRIPT_DIR = os.path.dirname(os.path.realpath(__file__))


async def from_unstructured_with_gliner2():
"""Same ingestion as loading.py, but with a local, LLM-free extraction
backend -- no OPENAI_API_KEY needed. Requires `gliner2` installed
manually (not a pyproject.toml extra -- see gliner2_backend.py's module
docstring): pip install 'gliner2[local]>=2.0.0'.
"""
memgraph = Memgraph(user_agent="unstructured2graph")
memgraph.query("MATCH (n) DETACH DELETE n;")

# DEFAULT_ONTOLOGY (used when omitted) is entity-only; pass an ontology
# with relation_types to also extract typed relationship edges.
backend = GLiNER2Backend()

await from_unstructured(
SOURCES.MEMGRAPH_DOCS_GITHUB_LATEST_RAW,
memgraph,
backend,
only_chunks=False,
link_chunks=True,
enforce_ontology=True, # promote entity_type to real labels (:Person, :Organization, ...)
)


if __name__ == "__main__":
logging.basicConfig(level=logging.INFO)

asyncio.run(from_unstructured_with_gliner2())
4 changes: 2 additions & 2 deletions unstructured2graph/examples/loading.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,7 @@

from lightrag_memgraph import MemgraphLightRAGWrapper
from memgraph_toolbox.api.memgraph import Memgraph
from unstructured2graph import from_unstructured
from unstructured2graph import LightRAGBackend, from_unstructured

SCRIPT_DIR = os.path.dirname(os.path.realpath(__file__))
LIGHTRAG_DIR = os.path.join(SCRIPT_DIR, "..", "lightrag_storage.out")
Expand All @@ -33,7 +33,7 @@ async def from_unstructured_with_prep():
await from_unstructured(
SOURCES.MEMGRAPH_DOCS_GITHUB_LATEST_RAW,
memgraph,
lightrag_wrapper,
LightRAGBackend(lightrag_wrapper),
only_chunks=False,
link_chunks=True,
enforce_ontology=True, # promote entity_type to real labels (:Person, :Organization, ...)
Expand Down
Loading
Loading