"""Phase 0 end-to-end pipeline test. Runs the actual OntoCast workflow through the FastAPI ``/process`` endpoint with a tiny JSON input. Verifies Acceptance Gate 0 evidence: - /health, /info, /process respond OK - ontology + facts Turtle are produced - BudgetTracker reports non-zero LLM call/triple counts - Filesystem TripleStoreManager writes artifacts under working_directory Skipped unless ``LLM_API_KEY`` is set (see ``conftest.py``). """ from __future__ import annotations import importlib import json import os import sys from pathlib import Path import pytest from fastapi.testclient import TestClient REPO_ROOT = Path(__file__).resolve().parents[2] if str(REPO_ROOT) not in sys.path: sys.path.insert(0, str(REPO_ROOT)) main_module = importlib.import_module("ont_platform.api.main") deps_module = importlib.import_module("ont_platform.api.deps") platform_config = importlib.import_module("ont_platform.config") @pytest.mark.e2e @pytest.mark.asyncio async def test_full_pipeline_writes_ontology_and_facts( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: # Point the working directory at the test's tmp_path so artifacts don't # leak between runs. monkeypatch.setenv("ONTOCAST_WORKING_DIRECTORY", str(tmp_path / "work")) # Honor whatever LLM provider the operator configured. # OpenAI는 API 키 필수, Ollama는 로컬 데몬만 떠 있으면 됨. provider = os.environ.get("LLM_PROVIDER", "openai").lower() if provider == "openai": assert os.environ.get("LLM_API_KEY"), ( "LLM_API_KEY must be set for e2e with OpenAI provider" ) # Force a fresh AppContext so the new working_directory wins. deps_module.reset_app_context_for_testing() settings = platform_config.load_settings() await deps_module.initialize_app_context(settings, head_chunks=1) app = main_module.create_app() # OntoCast SemanticChunker가 HDBSCAN + UMAP을 쓰므로 최소 문장 수가 # 필요하다. 2문장짜리 toy 입력은 "k must be ≤ training points"로 실패한다. # Acceptance Gate #1은 "단일 PDF/JSON → TTL 생성"이지 문장 수와 무관하므로, # ontology-rich한 짧은 단락을 충분한 문장 수로 늘려준다. payload = { "text": ( "Alice Carter works at Acme Corporation in Berlin. " "Acme Corporation is a manufacturing company founded in 1992. " "Acme manufactures bicycles and electric scooters. " "Bob Lee is the chief engineer at Acme Corporation. " "He reports to Alice Carter, who heads the engineering division. " "Acme's main factory is located in Berlin, Germany. " "The company also operates a research center in Munich. " "Carol Schmidt leads research at the Munich center. " "She previously worked at Globex Industries in Hamburg. " "Globex Industries is a competitor in the bicycle market. " "Acme exports bicycles to France, Italy, and Spain. " "The product line includes road bikes, mountain bikes, and city bikes. " "Alice Carter graduated from the Technical University of Berlin. " "Bob Lee holds a doctorate in mechanical engineering. " "Acme employs around 450 people across its three sites. " "The company reported annual revenue of 120 million euros last year." ), "ontology_user_instruction": "Focus on person-organization-location relations.", "facts_user_instruction": "Extract employment and manufacturing facts.", } with TestClient(app) as client: # /health & /info first as smoke assert client.get("/health").status_code == 200 assert client.get("/info").status_code == 200 response = client.post( "/process", content=json.dumps(payload), headers={"Content-Type": "application/json"}, ) assert response.status_code == 200, response.text body = response.json() assert body["status"] == "success" assert body["data"]["ontology"], "ontology TTL must not be empty" assert body["data"]["facts"], "facts TTL must not be empty" budget = body["metadata"]["budget"] assert budget["calls_count"] > 0, "BudgetTracker must record at least one LLM call" assert ( budget["ontology_triples_generated"] > 0 or budget["facts_triples_generated"] > 0 ), "BudgetTracker must record triple generation" # The filesystem manager should have created artifacts somewhere under # working_directory. We don't assert exact filenames (those depend on # the document hash) — just that the directory is non-empty. work_dir = tmp_path / "work" written = list(work_dir.rglob("*")) assert any(p.suffix in {".ttl", ".rdf"} for p in written if p.is_file()), ( f"No RDF artifacts produced under {work_dir}" )