from llama_index.core import Settings, VectorStoreIndex from llama_index.core.embeddings import MockEmbedding from llama_index.core.schema import TextNode from llama_index.core.vector_stores import ( ExactMatchFilter, FilterCondition, MetadataFilters, ) Settings.embed_model = MockEmbedding(embed_dim=8) nodes = [ TextNode(text="Refund requests require billing review.", metadata={"tenant": "atlas", "source": "billing"}), TextNode(text="Shipping claims require logistics review.", metadata={"tenant": "atlas", "source": "logistics"}), TextNode(text="Refund exports require finance approval.", metadata={"tenant": "contoso", "source": "billing"}), ] index = VectorStoreIndex(nodes) def print_matches(label, matches): print(f"{label}_count={len(matches)}") for item in sorted(matches, key=lambda match: match.node.metadata["source"]): metadata = item.node.metadata print(f"tenant={metadata['tenant']} source={metadata['source']}") tenant_filters = MetadataFilters( filters=[ExactMatchFilter(key="tenant", value="atlas")] ) tenant_retriever = index.as_retriever( similarity_top_k=3, filters=tenant_filters, ) tenant_matches = tenant_retriever.retrieve("review requests") print_matches("tenant_atlas", tenant_matches) tenant_source_filters = MetadataFilters( filters=[ ExactMatchFilter(key="tenant", value="atlas"), ExactMatchFilter(key="source", value="billing"), ], condition=FilterCondition.AND, ) tenant_source_retriever = index.as_retriever( similarity_top_k=3, filters=tenant_source_filters, ) tenant_source_matches = tenant_source_retriever.retrieve("refund review") assert {item.node.metadata["tenant"] for item in tenant_matches} == {"atlas"} assert [item.node.metadata for item in tenant_source_matches] == [ {"tenant": "atlas", "source": "billing"} ] print_matches("tenant_atlas_source_billing", tenant_source_matches)