참고소스 수정본

This commit is contained in:
LASTA_DEV01\lasta
2026-05-12 19:40:31 +09:00
parent 0f34a451fc
commit 2e9204243d
8708 changed files with 3259488 additions and 869 deletions

View File

@@ -0,0 +1,30 @@
### Usage Instructions
You will need both a Pinecone vector database and a Neo4j database to use this retriever.
### Writing Test Data
Update `NEO4J_AUTH`, `NEO4J_URL`, and `PC_API_KEY` variables in the `tests/e2e/pinecone_e2e/populate_dbs.py` script then run this from the project root to write test data to both dbs.
```
uv run python -m tests/e2e/pinecone_e2e/populate_dbs.py
```
### Install Pinecone client
You need to install the `pinecone-client` package to use this retriever.
```bash
pip install pinecone-client
```
### Search
Update the `NEO4J_AUTH`, `NEO4J_URL`, and `PC_API_KEY` variables in each file then run one of the following from the project root to test the retriever.
```
# Search by vector
uv run python -m examples.customize.retrievers.external.pinecone.vector_search
# Search by text, with embeddings generated locally
uv run python -m examples.customize.retrievers.external.pinecone.text_search
```

View File

@@ -0,0 +1,40 @@
"""This example demonstrates how to use PineconeNeo4jRetriever, ie vectors are
stored in the Pinecone database.
See the [README](./README.md) for more
information about how spin up a Pinecone and Neo4j databases if needed.
In this example, search is performed from a text. Embeddings are computed
using OpenAI models. See [../../embeddings/](../../embeddings/) for examples
using other supported embedders.
"""
from neo4j import GraphDatabase
from neo4j_graphrag.embeddings import OpenAIEmbeddings
from neo4j_graphrag.retrievers import PineconeNeo4jRetriever
from pinecone import Pinecone
NEO4J_AUTH = ("neo4j", "password")
NEO4J_URL = "neo4j://localhost:7687"
PC_API_KEY = "API_KEY"
def main() -> None:
with GraphDatabase.driver(NEO4J_URL, auth=NEO4J_AUTH) as neo4j_driver:
pc_client = Pinecone(PC_API_KEY)
embedder = OpenAIEmbeddings()
retriever = PineconeNeo4jRetriever(
driver=neo4j_driver,
client=pc_client,
index_name="jeopardy",
id_property_neo4j="id",
embedder=embedder,
)
res = retriever.search(query_text="biology", top_k=2)
print(res)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,39 @@
"""This example demonstrates how to use PineconeNeo4jRetriever, ie vectors are
stored in the Pinecone database.
See the [README](./README.md) for more
information about how spin up a Pinecone and Neo4j databases if needed.
In this example, search is performed from an already computed vector.
"""
from neo4j import GraphDatabase
from neo4j_graphrag.embeddings.sentence_transformers import (
SentenceTransformerEmbeddings,
)
from neo4j_graphrag.retrievers import PineconeNeo4jRetriever
from pinecone import Pinecone
NEO4J_AUTH = ("neo4j", "password")
NEO4J_URL = "neo4j://localhost:7687"
PC_API_KEY = "API_KEY"
def main() -> None:
with GraphDatabase.driver(NEO4J_URL, auth=NEO4J_AUTH) as neo4j_driver:
pc_client = Pinecone(PC_API_KEY)
embedder = SentenceTransformerEmbeddings(model="all-MiniLM-L6-v2")
retriever = PineconeNeo4jRetriever(
driver=neo4j_driver,
client=pc_client,
index_name="jeopardy",
id_property_neo4j="id",
embedder=embedder,
)
res = retriever.search(query_text="biology", top_k=2)
print(res)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,31 @@
### Start services locally
Run the following command to spin up Neo4j and Qdrant containers.
```bash
docker compose -f tests/e2e/docker-compose.yml up
```
### Write data (once)
Run this from the project root to write data to both Neo4J and Qdrant.
```bash
uv run python -m examples.customize.retrievers.external.qdrant.populate_dbs
```
### Install Qdrant client
```bash
pip install qdrant-client
```
### Search
```bash
# search by vector
uv run python -m examples.customize.retrievers.external.qdrant.vector_search
# search by text, with embeddings generated locally
uv run python -m examples.customize.retrievers.external.qdrant.text_search
```

View File

@@ -0,0 +1,16 @@
version: "3.9"
services:
neo4j:
image: neo4j:5.24-enterprise
ports:
- 7687:7687
- 7474:7474
environment:
NEO4J_AUTH: neo4j/password
NEO4J_ACCEPT_LICENSE_AGREEMENT: "eval"
NEO4J_PLUGINS: "[\"apoc\"]"
qdrant:
image: qdrant/qdrant
ports:
- 6333:6333

View File

@@ -0,0 +1,613 @@
# Copyright (c) "Neo4j"
# Neo4j Sweden AB [https://neo4j.com]
# #
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
# #
# https://www.apache.org/licenses/LICENSE-2.0
# #
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
from __future__ import annotations
import hashlib
import json
from typing import Any, Literal
import neo4j
from neo4j import GraphDatabase
from neo4j_graphrag.indexes import create_vector_index, drop_index_if_exists
try:
from qdrant_client import QdrantClient, models
except ImportError as e:
missing_module = str(e).split("'")[1]
if missing_module == "qdrant_client":
raise ImportError(
"The 'qdrant-client' package is missing. Please install it by running "
"`uv sync --extra qdrant` or `pip install qdrant-client`, or follow the instructions "
"in the Qdrant examples section of the README at https://github.com/neo4j/neo4j-graphrag-python/"
) from e
else:
raise
# biology
EMBEDDING_BIOLOGY = [
-0.04312172904610634,
0.027684900909662247,
-0.03451697900891304,
0.033050164580345154,
-0.02548302337527275,
-0.02422613650560379,
0.06078026071190834,
0.05728934705257416,
0.026537826284766197,
0.06596288084983826,
-0.011962010525166988,
-0.03995208814740181,
-0.08753148466348648,
0.048919592052698135,
-0.12121172994375229,
0.0046921405009925365,
-0.11684603989124298,
0.011236454360187054,
-0.08468407392501831,
0.00615630391985178,
-0.01894194260239601,
0.07591243833303452,
0.010966621339321136,
-0.0035629894118756056,
-0.07222436368465424,
-0.0038335833232849836,
-0.013435023836791515,
-0.01072753593325615,
-0.019481584429740906,
-0.08576139807701111,
-0.005984509363770485,
0.06511891633272171,
0.047047052532434464,
-0.008832206949591637,
-0.02516607940196991,
0.030757833272218704,
0.01017201878130436,
-0.035205237567424774,
0.0754496157169342,
0.049298327416181564,
-0.008700424805283546,
-0.042502958327531815,
-0.007827227003872395,
0.022037239745259285,
0.01526452787220478,
0.057564627379179,
0.01117363478988409,
-0.022880272939801216,
0.002632720861583948,
-0.06985171139240265,
-0.04440313205122948,
-0.02955677919089794,
-0.12585243582725525,
0.02834390103816986,
0.047075871378183365,
0.039754725992679596,
-0.062267716974020004,
-0.02022026665508747,
-0.02561432123184204,
-0.06196143478155136,
0.06283645331859589,
-0.010700947605073452,
-0.009793519042432308,
0.0676884651184082,
0.07930963486433029,
-0.05287032946944237,
0.005661565810441971,
0.04567752033472061,
0.01226893998682499,
0.0024454575031995773,
0.03628166392445564,
-0.01688201352953911,
-0.0009241311927326024,
0.10099577158689499,
0.0851878672838211,
-0.0056961155496537685,
-0.012224642559885979,
0.03202442824840546,
0.07562054693698883,
-0.030286865308880806,
0.031771108508110046,
-0.07060682028532028,
-0.01805405505001545,
0.03888492286205292,
0.013361049816012383,
-0.025815952569246292,
0.06777660548686981,
0.05500587448477745,
-0.09657344222068787,
0.06346127390861511,
-0.03499232605099678,
-0.06173454597592354,
0.03703204542398453,
-0.008920160122215748,
-0.1170671284198761,
0.05101491138339043,
0.02324659377336502,
-0.11062400788068771,
0.05613391101360321,
0.1953216791152954,
-0.046223606914281845,
0.0207725428044796,
-0.037569694221019745,
0.09747911989688873,
0.03182050585746765,
-0.09000813961029053,
-0.03741106390953064,
-0.001107109128497541,
0.05828772485256195,
0.03561848774552345,
-0.012867141515016556,
0.04318442568182945,
-0.007168568670749664,
0.10120371729135513,
0.015526260249316692,
0.046518243849277496,
0.09026871621608734,
-0.025434499606490135,
0.07185306400060654,
0.006204536650329828,
-0.025380434468388557,
-0.03805668652057648,
-0.046537771821022034,
-0.028066400438547134,
-0.04620446264743805,
-0.04175203666090965,
-0.05594577267765999,
-7.522815605061209e-33,
0.0032660849392414093,
-0.0987364873290062,
0.048021912574768066,
0.026744935661554337,
-0.04386494308710098,
0.033061958849430084,
-0.0033691301941871643,
-0.09897749125957489,
-0.0019302349537611008,
0.011837359517812729,
-0.05590499937534332,
-0.010936262086033821,
0.0011399721261113882,
0.043522097170352936,
0.024741550907492638,
0.030598435550928116,
-0.10856965184211731,
0.09524074196815491,
-0.011383255943655968,
0.0199135635048151,
-0.055631425231695175,
0.02365952730178833,
-0.02242668904364109,
-0.046286772936582565,
0.0020421564113348722,
-0.029465770348906517,
-0.0544230192899704,
-0.08019289374351501,
0.055107232183218,
-0.010339600965380669,
0.016182616353034973,
-0.019025344401597977,
-0.06670568138360977,
0.0005314883892424405,
0.01192407961934805,
-0.07481196522712708,
0.04253043606877327,
-0.05555248260498047,
-0.007614267058670521,
-0.012775871902704239,
-0.0022697255481034517,
-0.03448570892214775,
0.036854878067970276,
-0.04938317835330963,
0.08296490460634232,
0.03675423562526703,
-0.003109127515926957,
0.00267212837934494,
-0.06396905332803726,
0.019167274236679077,
-0.020589588209986687,
-0.017604926601052284,
0.10695572197437286,
-0.09441392868757248,
0.021013183519244194,
0.04480829834938049,
-0.027837105095386505,
-0.0011123925214633346,
-0.08835164457559586,
0.017125461250543594,
0.06728456914424896,
0.12567774951457977,
-0.06033247709274292,
0.03521595522761345,
0.07578440755605698,
-0.03819991648197174,
-0.09275069832801819,
-0.09814322739839554,
0.021795116364955902,
0.025850528851151466,
-0.0532839410007,
-0.054003648459911346,
-0.026674047112464905,
-0.10661919414997101,
0.004801975563168526,
0.016014324501156807,
0.0004556076892185956,
0.01613032817840576,
-0.09400207549333572,
0.012587584555149078,
0.01425306685268879,
-0.03173312172293663,
-0.04417256638407707,
-0.034732427448034286,
-0.03740209341049194,
0.07315755635499954,
0.022384824231266975,
-0.0658881664276123,
5.2405772294150665e-05,
0.019976867362856865,
-0.054833319038152695,
-0.04896378144621849,
-0.013568416237831116,
-0.011525219306349754,
0.000531611789483577,
3.594654745316847e-33,
0.006438549142330885,
-0.07441601902246475,
-0.04541584476828575,
0.04658213630318642,
0.05004657804965973,
0.06633730232715607,
-0.009470561519265175,
-0.020702792331576347,
0.003479442559182644,
0.017732439562678337,
0.0058183688670396805,
0.0337534099817276,
0.06103092059493065,
-0.010244826786220074,
-0.049173060804605484,
-0.039285846054553986,
-0.019474590197205544,
-0.005957481451332569,
-0.007637818343937397,
-0.001945863594301045,
-0.10105801373720169,
0.07999070733785629,
-0.002908281981945038,
-0.055638041347265244,
-0.013490298762917519,
0.07804659008979797,
-0.017962178215384483,
0.05879015102982521,
-0.02929818443953991,
0.010783063247799873,
-0.017044004052877426,
0.03597024083137512,
-0.010958723723888397,
-0.020666809752583504,
0.030345136299729347,
0.08541561663150787,
0.06714903563261032,
-0.004256050102412701,
0.05138356238603592,
-0.061394669115543365,
0.06958435475826263,
0.029510458931326866,
0.006212680600583553,
0.05935342609882355,
0.06217789649963379,
0.03869514539837837,
0.016871990635991096,
0.09106622636318207,
0.01593017764389515,
0.03309919685125351,
3.424335227464326e-05,
0.025418603792786598,
-0.0019248499302193522,
-0.04490964487195015,
0.05094093829393387,
-0.007762531749904156,
0.0022712810896337032,
-0.048982247710227966,
0.04316263645887375,
0.03385719656944275,
-0.02204400859773159,
-0.008660282008349895,
-0.0358533076941967,
0.11571227014064789,
-0.11273631453514099,
0.06849292665719986,
-0.03595784679055214,
0.04161537066102028,
-0.013967745006084442,
-0.004196549765765667,
0.04396875202655792,
0.07947150617837906,
-0.038232240825891495,
-0.004747415892779827,
-0.012230468913912773,
0.039619915187358856,
-0.11451079696416855,
0.0012758959783241153,
-0.007560120429843664,
0.015182306058704853,
-0.039508406072854996,
-0.06489536166191101,
-0.003909661900252104,
0.0037100922781974077,
0.010909723117947578,
-0.07211877405643463,
0.024324269965291023,
0.04425453022122383,
0.02042572572827339,
-0.07636857777833939,
-0.03965488821268082,
-0.04338192194700241,
-0.10347922146320343,
-0.04393737390637398,
-0.01368421409279108,
-1.568378671379378e-08,
0.10027588903903961,
-0.043567851185798645,
0.050843071192502975,
-0.08256983011960983,
0.02588729001581669,
0.056297384202480316,
-0.010702059604227543,
0.08518678694963455,
0.02700129710137844,
0.07154033333063126,
-0.05953506380319595,
0.0778793916106224,
0.036633629351854324,
0.08145243674516678,
0.07252412289381027,
-0.002973137190565467,
-0.013765356503427029,
-0.046995650976896286,
-0.0247399490326643,
-0.027804242447018623,
-0.025088943541049957,
-0.01294120866805315,
-0.03852837160229683,
0.11392410844564438,
0.030737342312932014,
-0.065057672560215,
0.06248151883482933,
-0.01869017630815506,
0.02368609979748726,
-0.0654701292514801,
0.058893367648124695,
0.06793640553951263,
0.008447104133665562,
0.015255128964781761,
-0.04546090587973595,
-0.04216332733631134,
0.02343529835343361,
-0.07064592838287354,
0.01851677894592285,
-0.006000120658427477,
-0.025140153244137764,
0.003474486293271184,
-0.01600506715476513,
0.0020763161592185497,
0.010104725137352943,
0.0021877381950616837,
0.07691455632448196,
0.008766746148467064,
-0.012105572037398815,
-0.047985345125198364,
-0.012077542021870613,
-0.02539162151515484,
0.01326005533337593,
-0.05451769009232521,
-0.08339792490005493,
0.052714310586452484,
-0.011956708505749702,
-0.01412263885140419,
-0.05299930274486542,
0.0125426622107625,
0.12013623863458633,
0.09160523116588593,
0.12442110478878021,
-0.03318800404667854,
]
def populate_neo4j(
neo4j_driver: neo4j.Driver,
neo4j_objs: dict[str, Any],
should_create_vector_index: bool = False,
) -> neo4j.EagerResult:
question_nodes = list(
filter(lambda x: x["label"] == "Question", neo4j_objs["nodes"])
)
answer_nodes = list(filter(lambda x: x["label"] == "Answer", neo4j_objs["nodes"]))
category_nodes = list(
filter(lambda x: x["label"] == "Category", neo4j_objs["nodes"])
)
belongs_to_relationships = list(
filter(lambda x: x["type"] == "BELONGS_TO", neo4j_objs["relationships"])
)
has_answer_relationships = list(
filter(lambda x: x["type"] == "HAS_ANSWER", neo4j_objs["relationships"])
)
question_nodes_cypher = "UNWIND $nodes as node MERGE (n:Question {id: node.properties.id}) ON CREATE SET n = node.properties"
answer_nodes_cypher = "UNWIND $nodes as node MERGE (n:Answer {id: node.properties.id}) ON CREATE SET n = node.properties"
category_nodes_cypher = (
"UNWIND $nodes as node MERGE (n:Category {id: node.id}) ON CREATE SET n = node"
)
belongs_to_relationships_cypher = "UNWIND $relationships as rel MATCH (q:Question {id: rel.start_node_id}), (c:Category {id: rel.end_node_id}) MERGE (q)-[r:BELONGS_TO]->(c)"
has_answer_relationships_cypher = "UNWIND $relationships as rel MATCH (q:Question {id: rel.start_node_id}), (a:Answer {id: rel.end_node_id}) MERGE (q)-[r:HAS_ANSWER]->(a)"
neo4j_driver.execute_query(question_nodes_cypher, {"nodes": question_nodes})
neo4j_driver.execute_query(answer_nodes_cypher, {"nodes": answer_nodes})
neo4j_driver.execute_query(category_nodes_cypher, {"nodes": category_nodes})
neo4j_driver.execute_query(
belongs_to_relationships_cypher, {"relationships": belongs_to_relationships}
)
res = neo4j_driver.execute_query(
has_answer_relationships_cypher, {"relationships": has_answer_relationships}
)
if should_create_vector_index:
vector_index_name = "vector-index-name"
drop_index_if_exists(neo4j_driver, vector_index_name)
# Create a vector index
create_vector_index(
neo4j_driver,
vector_index_name,
label="Question",
embedding_property="vector",
dimensions=384,
similarity_fn="cosine",
)
return res
def build_data_objects(
q_vector_fmt: Literal["neo4j", "qdrant"],
) -> tuple[dict[str, Any], list[Any]]:
# read file from disk
# this file is from https://github.com/weaviate-tutorials/quickstart/tree/main/data
# MIT License
file_name = "tests/e2e/data/jeopardy_tiny_with_vectors_all-MiniLM-L6-v2.json"
with open(file_name, "r") as f:
data = json.load(f)
question_objs: list[Any] = []
neo4j_objs: dict[str, list[Any]] = {"nodes": [], "relationships": []}
# only unique categories and IDs for them
unique_categories_list = list(set([c["Category"] for c in data]))
unique_categories = [
{"label": "Category", "name": c, "id": c} for c in unique_categories_list
]
neo4j_objs["nodes"] += unique_categories
for i, d in enumerate(data):
id_ = hashlib.md5(d["Question"].encode()).hexdigest()
question_properties = {
"id": f"question_{id_}",
"question": d["Question"],
}
if q_vector_fmt == "neo4j":
# Store the vector directly on the question node for Neo4j
question_properties["vector"] = d["vector"]
# Add the question node
neo4j_objs["nodes"].append(
{
"label": "Question",
"properties": question_properties,
}
)
# Add the answer node
neo4j_objs["nodes"].append(
{
"label": "Answer",
"properties": {
"id": f"answer_{id_}",
"answer": d["Answer"],
},
}
)
# Add relationships
neo4j_objs["relationships"].append(
{
"start_node_id": f"question_{id_}",
"end_node_id": f"answer_{id_}",
"type": "HAS_ANSWER",
"properties": {},
}
)
neo4j_objs["relationships"].append(
{
"start_node_id": f"question_{id_}",
"end_node_id": d["Category"],
"type": "BELONGS_TO",
"properties": {},
}
)
# If Qdrant, we build PointStruct objects
if q_vector_fmt == "qdrant":
question_objs.append(
models.PointStruct(
id=i,
payload={"neo4j_id": f"question_{id_}"},
vector=d["vector"],
)
)
elif q_vector_fmt == "neo4j":
# For Neo4j, the vector is already added to question_properties
pass
else:
raise ValueError("q_vector_fmt must be either 'neo4j' or 'qdrant'")
return neo4j_objs, question_objs
def populate_dbs(
neo4j_driver: neo4j.Driver,
qdrant_client: QdrantClient,
collection_name: str = "Jeopardy",
) -> None:
"""
Populates both Neo4j and Qdrant with the Jeopardy data set.
"""
neo4j_objects, question_objs = build_data_objects("qdrant")
# Recreate the Qdrant collection
if qdrant_client.collection_exists(collection_name):
qdrant_client.delete_collection(collection_name)
qdrant_client.create_collection(
collection_name=collection_name,
vectors_config=models.VectorParams(size=384, distance=models.Distance.COSINE),
)
_populate_qdrant(qdrant_client, question_objs, collection_name)
populate_neo4j(neo4j_driver, neo4j_objects)
def _populate_qdrant(
client: QdrantClient, question_objs: list[Any], collection_name: str
) -> None:
"""
Inserts (upserts) question objects into the specified Qdrant collection.
"""
client.upsert(
collection_name=collection_name,
points=question_objs,
)
def main() -> None:
"""
Entry point for running the database population script.
"""
NEO4J_URL = "neo4j://localhost:7687"
NEO4J_AUTH = ("neo4j", "password")
with GraphDatabase.driver(NEO4J_URL, auth=NEO4J_AUTH) as neo4j_driver:
qdrant_client = QdrantClient(url="http://localhost:6333")
populate_dbs(neo4j_driver, qdrant_client, collection_name="Jeopardy")
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,29 @@
from neo4j import GraphDatabase
from neo4j_graphrag.embeddings.sentence_transformers import (
SentenceTransformerEmbeddings,
)
from neo4j_graphrag.retrievers import QdrantNeo4jRetriever
from qdrant_client import QdrantClient
NEO4J_URL = "neo4j://localhost:7687"
NEO4J_AUTH = ("neo4j", "password")
def main() -> None:
with GraphDatabase.driver(NEO4J_URL, auth=NEO4J_AUTH) as neo4j_driver:
embedder = SentenceTransformerEmbeddings(model="all-MiniLM-L6-v2")
retriever = QdrantNeo4jRetriever(
driver=neo4j_driver,
client=QdrantClient(url="http://localhost:6333"),
collection_name="Jeopardy",
id_property_external="neo4j_id",
id_property_neo4j="id",
embedder=embedder,
)
res = retriever.search(query_text="biology", top_k=2)
print(res)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,24 @@
from embedding_biology import EMBEDDING_BIOLOGY
from neo4j import GraphDatabase
from neo4j_graphrag.retrievers import QdrantNeo4jRetriever
from qdrant_client import QdrantClient
NEO4J_URL = "neo4j://localhost:7687"
NEO4J_AUTH = ("neo4j", "password")
def main() -> None:
with GraphDatabase.driver(NEO4J_URL, auth=NEO4J_AUTH) as neo4j_driver:
retriever = QdrantNeo4jRetriever(
driver=neo4j_driver,
client=QdrantClient(url="http://localhost:6333"),
collection_name="Jeopardy",
id_property_external="neo4j_id",
id_property_neo4j="id",
)
res = retriever.search(query_vector=EMBEDDING_BIOLOGY, top_k=2)
print(res)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,38 @@
### Start services locally
This is a manual task you need to do in the terminal.
This spins up Neo4j and Weaviate containers and is configuring Weaviate to use embeddings from Hugging Face's Sentence Transformers using the "all-MiniLM-L6-v2" model, which has 384 dimensions.
```bash
docker compose -f tests/e2e/docker-compose.yml up
```
### Write data (once)
Run this from the project root to write data to both dbs.
```
uv run python -m tests/e2e/weaviate_e2e/populate_dbs.py
```
### Install Weaviate client
You need to install the `weaviate-client` package to use this retriever.
```bash
pip install weaviate-client
```
### Search
```
# search by vector
uv run python -m examples.customize.retrievers.external.weaviate.vector_search
# search by text, with embeddings generated locally (via embedder argument)
uv run python -m examples.customize.retrievers.external.weaviate.text_search_local_embedder
# search by text, with embeddings generated on the Weaviate side, via configured vectorizer
uv run python -m examples.customize.retrievers.external.weaviate.text_search_remote_embedder
```

View File

@@ -0,0 +1,40 @@
"""This example demonstrates how to use WeaviateNeo4jRetriever, ie vectors are
stored in the Weaviate database.
See the [README](./README.md) for more
information about how spin up a Weaviate and Neo4j databases if needed.
In this example, we are embeddings a text and provide a local embeder to the
WeaviateNeo4jRetriever.
"""
from neo4j import GraphDatabase
from neo4j_graphrag.embeddings.sentence_transformers import (
SentenceTransformerEmbeddings,
)
from neo4j_graphrag.retrievers import WeaviateNeo4jRetriever
from weaviate.connect.helpers import connect_to_local
NEO4J_URL = "neo4j://localhost:7687"
NEO4J_AUTH = ("neo4j", "password")
def main() -> None:
with GraphDatabase.driver(NEO4J_URL, auth=NEO4J_AUTH) as neo4j_driver:
with connect_to_local() as w_client:
embedder = SentenceTransformerEmbeddings(model="all-MiniLM-L6-v2")
retriever = WeaviateNeo4jRetriever(
driver=neo4j_driver,
client=w_client,
collection="Jeopardy",
id_property_external="neo4j_id",
id_property_neo4j="id",
embedder=embedder,
)
res = retriever.search(query_text="biology", top_k=2)
print(res)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,37 @@
"""This example demonstrates how to use WeaviateNeo4jRetriever, ie vectors are
stored in the Weaviate database.
See the [README](./README.md) for more
information about how spin up a Weaviate and Neo4j databases if needed.
In this example, we are embeddings a text and provide a remote embeder to the
WeaviateNeo4jRetriever.
"""
from neo4j import GraphDatabase
from neo4j_graphrag.retrievers import WeaviateNeo4jRetriever
from weaviate.connect.helpers import connect_to_local
NEO4J_URL = "neo4j://localhost:7687"
NEO4J_AUTH = ("neo4j", "password")
def main() -> None:
with GraphDatabase.driver(NEO4J_URL, auth=NEO4J_AUTH) as neo4j_driver:
with connect_to_local() as w_client:
retriever = WeaviateNeo4jRetriever(
driver=neo4j_driver,
client=w_client,
collection="Jeopardy",
id_property_external="neo4j_id",
id_property_neo4j="id",
)
# Weaviate is configured to embed the text on the server side
# see populate_dbs.py for the configuration
res = retriever.search(query_text="biology", top_k=2)
print(res)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,34 @@
"""This example demonstrates how to use WeaviateNeo4jRetriever, ie vectors are
stored in the Weaviate database.
See the [README](./README.md) for more
information about how spin up a Weaviate and Neo4j databases if needed.
In this example, search is performed from an already existing vector.
"""
from embedding_biology import EMBEDDING_BIOLOGY
from neo4j import GraphDatabase
from neo4j_graphrag.retrievers import WeaviateNeo4jRetriever
from weaviate.connect.helpers import connect_to_local
NEO4J_URL = "neo4j://localhost:7687"
NEO4J_AUTH = ("neo4j", "password")
def main() -> None:
with GraphDatabase.driver(NEO4J_URL, auth=NEO4J_AUTH) as neo4j_driver:
with connect_to_local() as w_client:
retriever = WeaviateNeo4jRetriever(
driver=neo4j_driver,
client=w_client,
collection="Jeopardy",
id_property_external="neo4j_id",
id_property_neo4j="id",
)
res = retriever.search(query_vector=EMBEDDING_BIOLOGY, top_k=2)
print(res)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,64 @@
from __future__ import annotations
from random import random
from neo4j import GraphDatabase
from neo4j_graphrag.embeddings.base import Embedder
from neo4j_graphrag.indexes import create_fulltext_index, create_vector_index
from neo4j_graphrag.retrievers import HybridCypherRetriever
URI = "neo4j://localhost:7687"
AUTH = ("neo4j", "password")
INDEX_NAME = "embedding-name"
FULLTEXT_INDEX_NAME = "fulltext-index-name"
DIMENSION = 1536
# Connect to Neo4j database
driver = GraphDatabase.driver(URI, auth=AUTH)
# Create Embedder object
class CustomEmbedder(Embedder):
def embed_query(self, text: str) -> list[float]:
return [random() for _ in range(DIMENSION)]
embedder = CustomEmbedder()
# Creating the index
create_vector_index(
driver,
INDEX_NAME,
label="Document",
embedding_property="vectorProperty",
dimensions=DIMENSION,
similarity_fn="euclidean",
)
create_fulltext_index(
driver, FULLTEXT_INDEX_NAME, label="Document", node_properties=["vectorProperty"]
)
# Initialize the retriever
retrieval_query = "MATCH (node)-[:AUTHORED_BY]->(author:Author)RETURN author.name"
retriever = HybridCypherRetriever(
driver, INDEX_NAME, FULLTEXT_INDEX_NAME, retrieval_query, embedder
)
# Upsert the query
vector = [random() for _ in range(DIMENSION)]
insert_query = (
"MERGE (n:Document {id: $id})"
"WITH n "
"CALL db.create.setNodeVectorProperty(n, 'vectorProperty', $vector)"
"RETURN n"
)
parameters = {
"id": 0,
"vector": vector,
}
driver.execute_query(insert_query, parameters)
# Perform the similarity search for a text query
query_text = "Find me a book about Fremen"
print(retriever.search(query_text=query_text, top_k=5))

View File

@@ -0,0 +1,61 @@
from __future__ import annotations
from random import random
from neo4j import GraphDatabase
from neo4j_graphrag.embeddings.base import Embedder
from neo4j_graphrag.indexes import create_fulltext_index, create_vector_index
from neo4j_graphrag.retrievers import HybridRetriever
URI = "neo4j://localhost:7687"
AUTH = ("neo4j", "password")
INDEX_NAME = "embedding-name"
FULLTEXT_INDEX_NAME = "fulltext-index-name"
DIMENSION = 1536
# Connect to Neo4j database
driver = GraphDatabase.driver(URI, auth=AUTH)
# Create Embedder object
class CustomEmbedder(Embedder):
def embed_query(self, text: str) -> list[float]:
return [random() for _ in range(DIMENSION)]
embedder = CustomEmbedder()
# Creating the index
create_vector_index(
driver,
INDEX_NAME,
label="Document",
embedding_property="vectorProperty",
dimensions=DIMENSION,
similarity_fn="euclidean",
)
create_fulltext_index(
driver, FULLTEXT_INDEX_NAME, label="Document", node_properties=["vectorProperty"]
)
# Initialize the retriever
retriever = HybridRetriever(driver, INDEX_NAME, FULLTEXT_INDEX_NAME, embedder)
# Upsert the query
vector = [random() for _ in range(DIMENSION)]
insert_query = (
"MERGE (n:Document {id: $id})"
"WITH n "
"CALL db.create.setNodeVectorProperty(n, 'vectorProperty', $vector)"
"RETURN n"
)
parameters = {
"id": 0,
"vector": vector,
}
driver.execute_query(insert_query, parameters)
# Perform the similarity search for a text query
query_text = "Find me a book about Fremen"
print(retriever.search(query_text=query_text, top_k=5))

View File

@@ -0,0 +1,62 @@
"""This example uses an example Movie database where movies' plots are embedded
using OpenAI embeddings. OPENAI_API_KEY needs to be set in the environment for
this example to run.
Also requires minimal Cypher knowledge to write the retrieval query.
It shows how to use a vector-cypher retriever to find context
similar to a query **text** using vector similarity + graph traversal.
"""
import neo4j
from neo4j_graphrag.embeddings.openai import OpenAIEmbeddings
from neo4j_graphrag.retrievers import VectorCypherRetriever
from neo4j_graphrag.types import RetrieverResultItem
# Define database credentials
URI = "neo4j+s://demo.neo4jlabs.com"
AUTH = ("recommendations", "recommendations")
DATABASE = "recommendations"
INDEX_NAME = "moviePlotsEmbedding"
# for each Movie node matched by the vector search, retrieve more context:
# the name of all actors starring in that movie
RETRIEVAL_QUERY = """
RETURN node.title as movieTitle,
node.plot as moviePlot,
collect { MATCH (actor:Actor)-[:ACTED_IN]->(node) RETURN actor.name } AS actors,
score as similarityScore
"""
def my_result_formatter(record: neo4j.Record) -> RetrieverResultItem:
"""The record is a row output from the RETRIEVAL_QUERY so it our case it contains
the following keys:
- movieTitle
- moviePlot
- actors
- similarityScore
"""
return RetrieverResultItem(
content=f"Movie title: {record.get('movieTitle')}, Plot: {record.get('moviePlot')}, Actors: {record.get('actors')}",
metadata={"score": record.get("similarityScore")},
)
with neo4j.GraphDatabase.driver(URI, auth=AUTH) as driver:
# Initialize the retriever
retriever = VectorCypherRetriever(
driver=driver,
index_name=INDEX_NAME,
# note: embedder is optional if you only use query_vector
embedder=OpenAIEmbeddings(),
retrieval_query=RETRIEVAL_QUERY,
result_formatter=my_result_formatter,
# optionally, set neo4j database
neo4j_database=DATABASE,
)
# Perform the similarity search for a text query
# (retrieve the top 5 most similar nodes)
query_text = "Who were the actors in Avatar?"
print(retriever.search(query_text=query_text, top_k=5))

View File

@@ -0,0 +1,152 @@
"""This example demonstrates how to customize the retriever
results format.
Retriever.get_search_result returns a RawSearchResult object that consists
in a list of neo4j.Records and an optional metadata dictionary. The
Retriever.search method returns a RetrieverResult object where each neo4j.Record
has been replaced by a RetrieverResultItem, ie a content and metadata dictionary.
The content is what will be used to augment the prompt. By default, this content
is a stringified representation of the neo4j.Record. There are multiple ways the
user can act on this format:
- Use the `return_properties` parameter
- And/or use the result_formatter parameter
Let's consider the movie database where the movies' plot have been embedded. Movie
nodes have additional properties such as title and are connected to Actor nodes that
have a name property:
(:Movie {embedding: [], title: "", plot: "", year: "", budget: 1000, ....})
<-[:ACTED_IN]-
(:Actor {name: ...})
NB: to run this example OPENAI_API_KEY needs to be in the env vars.
To use another embedder, see the corresponding examples in ../customize/embeddings
"""
import neo4j
from neo4j_graphrag.embeddings.openai import OpenAIEmbeddings
from neo4j_graphrag.retrievers import VectorRetriever
from neo4j_graphrag.types import RetrieverResultItem
URI = "neo4j+s://demo.neo4jlabs.com"
AUTH = ("recommendations", "recommendations")
DATABASE = "recommendations"
INDEX_NAME = "moviePlotsEmbedding"
# Connect to Neo4j database
driver = neo4j.GraphDatabase.driver(URI, auth=AUTH)
query_text = "Find a movie about astronauts"
top_k_results = 1
"""First, let's select the properties we want to return
with the return_properties parameter:
"""
print("=" * 50)
print("RETURN PROPERTIES")
retriever = VectorRetriever(
driver=driver,
index_name=INDEX_NAME,
embedder=OpenAIEmbeddings(),
return_properties=["title", "plot"],
neo4j_database=DATABASE,
)
print(retriever.search(query_text=query_text, top_k=top_k_results))
print()
"""
OUTPUT:
RetrieverResult(
items=[
RetrieverResultItem(
content="{'title': 'For All Mankind', 'plot': 'This movie documents the Apollo missions perhaps the most definitively of any movie under two hours. Al Reinert watched all the footage shot during the missions--over 6,000,000 feet of it, ...'}",
metadata={'score': 0.9354040622711182, 'nodeLabels': None, 'id': None}
)
]
metadata={'__retriever': 'VectorRetriever'}
)
"""
"""In a second example, we'll use the ability to format the result in detail
with a function
"""
print("=" * 50)
print("RESULT FOMATTER")
def my_result_formatter(record: neo4j.Record) -> RetrieverResultItem:
"""
If 'return_properties' are not specified, vector retrievers will return a record with keys:
- `node`: a dict representation of all node properties (except embedding if we can identify it)
- `id`: the node element ID
- `nodeLabels`: the labels attached to the node
- `score`: the score returned by the vector index search that tells us how close the node vector is from the query vector
In the case of movies, we may want to keep in the content only the title and plot
(passed to the LLM afterward) and keep in the metadata the score.
This can be achieved with this function:
"""
node = record.get("node")
return RetrieverResultItem(
content=f"{node.get('title')}: {node.get('plot')}",
metadata={"score": record.get("score")},
)
retriever = VectorRetriever(
driver=driver,
index_name=INDEX_NAME,
embedder=OpenAIEmbeddings(),
result_formatter=my_result_formatter,
)
query_text = "Find a movie about astronauts"
print(retriever.search(query_text=query_text, top_k=top_k_results))
print()
"""
OUTPUT:
RetrieverResult(
items=[
RetrieverResultItem(
content='For All Mankind: This movie documents the Apollo missions perhaps the most definitively of any movie under two hours. Al Reinert watched all the footage shot during the missions--over 6,000,000 feet of it, ...',
metadata={'score': 0.9354040622711182}
)
]
metadata={'__retriever': 'VectorRetriever'}
)
"""
"""We can mix both return_properties and result_formatter:
"""
print("=" * 50)
print("RETURN PROPERTIES + RESULT FOMATTER")
retriever = VectorRetriever(
driver=driver,
index_name=INDEX_NAME,
embedder=OpenAIEmbeddings(),
return_properties=["title", "plot"],
result_formatter=my_result_formatter,
)
query_text = "Find a movie about astronauts"
print(retriever.search(query_text=query_text, top_k=top_k_results))
print()
"""
OUTPUT:
RetrieverResult(
items=[
RetrieverResultItem(
content='For All Mankind: This movie documents the Apollo missions perhaps the most definitively of any movie under two hours. Al Reinert watched all the footage shot during the missions--over 6,000,000 feet of it, ...',
metadata={'score': 0.9354040622711182})
]
metadata={'__retriever': 'VectorRetriever'}
)
"""
driver.close()

View File

@@ -0,0 +1,76 @@
"""The example shows how to provide a custom prompt to Text2CypherRetriever.
Example using the OpenAILLM, hence the OPENAI_API_KEY needs to be set in the
environment for this example to run.
"""
import neo4j
from neo4j_graphrag.llm import OpenAILLM
from neo4j_graphrag.retrievers import Text2CypherRetriever
from neo4j_graphrag.schema import get_schema
# Define database credentials
URI = "neo4j+s://demo.neo4jlabs.com"
AUTH = ("recommendations", "recommendations")
DATABASE = "recommendations"
# Create LLM object
llm = OpenAILLM(model_name="gpt-5", model_params={"temperature": 0})
# (Optional) Specify your own Neo4j schema
# (also see get_structured_schema and get_schema functions)
neo4j_schema = """
Node properties:
User {name: STRING}
Person {name: STRING, born: INTEGER}
Movie {tagline: STRING, title: STRING, released: INTEGER}
Relationship properties:
ACTED_IN {roles: LIST}
DIRECTED {}
REVIEWED {summary: STRING, rating: INTEGER}
The relationships:
(:Person)-[:ACTED_IN]->(:Movie)
(:Person)-[:DIRECTED]->(:Movie)
(:User)-[:REVIEWED]->(:Movie)
"""
prompt = """Task: Generate a Cypher statement for querying a Neo4j graph database from a user input.
Do not use any properties or relationships not included in the schema.
Do not include triple backticks ``` or any additional text except the generated Cypher statement in your response.
Always filter movies that have not already been reviewed by the user with name: '{user_name}' using for instance:
(m:Movie)<-[:REVIEWED]-(:User {{name: <the_user_name>}})
Schema:
{schema}
Input:
{query_text}
Cypher query:
"""
with neo4j.GraphDatabase.driver(URI, auth=AUTH) as driver:
# Initialize the retriever
retriever = Text2CypherRetriever(
driver=driver,
llm=llm,
neo4j_schema=neo4j_schema,
# here we provide a custom prompt
custom_prompt=prompt,
neo4j_database=DATABASE,
)
# Generate a Cypher query using the LLM, send it to the Neo4j database, and return the results
query_text = "Which movies did Hugo Weaving star in?"
print(
retriever.search(
query_text=query_text,
prompt_params={
# you have to specify all placeholder except the {query_text} one
"schema": get_schema(driver),
"user_name": "the user asking question",
},
)
)

View File

@@ -0,0 +1,32 @@
from __future__ import annotations
import neo4j
from neo4j_graphrag.embeddings import OpenAIEmbeddings
from neo4j_graphrag.retrievers import VectorRetriever
URI = "neo4j+s://demo.neo4jlabs.com"
AUTH = ("recommendations", "recommendations")
DATABASE = "recommendations"
INDEX_NAME = "moviePlotsEmbedding"
DIMENSION = 1536
# Connect to Neo4j database
with neo4j.GraphDatabase.driver(URI, auth=AUTH) as driver:
# Initialize the retriever
retriever = VectorRetriever(driver, INDEX_NAME, embedder=OpenAIEmbeddings())
# Perform the search
query_text = "Find me a movie about love"
pre_filters = {"int_property": {"$gt": 100}}
# pre_filters = {
# "year": {
# "$nin": ["1999", "2000"]
# }
# }
retriever_result = retriever.search(
query_text=query_text,
top_k=1,
filters=pre_filters,
)
print(retriever_result)