참고소스 수정본
This commit is contained in:
294
참고/ontocast-main/ontocast/agent/render_facts.py
Normal file
294
참고/ontocast-main/ontocast/agent/render_facts.py
Normal file
@@ -0,0 +1,294 @@
|
||||
"""Fact rendering agent for OntoCast.
|
||||
|
||||
This module provides functionality for rendering facts from RDF graphs into
|
||||
human-readable formats, making the extracted knowledge more accessible and
|
||||
understandable.
|
||||
"""
|
||||
|
||||
import logging
|
||||
|
||||
from langchain_core.output_parsers import PydanticOutputParser
|
||||
from langchain_core.prompts import PromptTemplate
|
||||
|
||||
from ontocast.agent.common import call_llm_with_retry, render_suggestions_prompt
|
||||
from ontocast.onto.constants import DEFAULT_IRI
|
||||
from ontocast.onto.enum import FailureStage, Status, WorkflowNode
|
||||
from ontocast.onto.model import FactsRenderReport, GraphUpdateRenderReport
|
||||
from ontocast.onto.rdfgraph import RDFGraph
|
||||
from ontocast.onto.unit_states import UnitFactsState
|
||||
from ontocast.prompt.common import (
|
||||
facts_template,
|
||||
ontology_template,
|
||||
output_instruction_empty,
|
||||
output_instruction_sparql,
|
||||
text_template,
|
||||
user_template,
|
||||
)
|
||||
from ontocast.prompt.render_facts import (
|
||||
facts_instruction_template,
|
||||
preamble,
|
||||
template_prompt,
|
||||
)
|
||||
from ontocast.tool.atomic import AtomicToolBox
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _extract_known_prefixes(state: UnitFactsState) -> dict[str, str]:
|
||||
"""Extract ontology prefixes used to patch missing declarations in LLM TTL output."""
|
||||
known_prefixes: dict[str, str] = {}
|
||||
|
||||
if state.ontology_snapshot and state.ontology_snapshot.graph:
|
||||
for prefix, namespace_uri in state.ontology_snapshot.graph.namespaces():
|
||||
if prefix: # Skip empty prefixes
|
||||
known_prefixes[prefix] = str(namespace_uri)
|
||||
|
||||
# Also add the ontology prefix explicitly if available.
|
||||
if state.ontology_snapshot.prefix and state.ontology_snapshot.namespace:
|
||||
known_prefixes[state.ontology_snapshot.prefix] = (
|
||||
state.ontology_snapshot.namespace
|
||||
)
|
||||
|
||||
return known_prefixes
|
||||
|
||||
|
||||
async def render_facts(state: UnitFactsState, tools: AtomicToolBox) -> UnitFactsState:
|
||||
"""Structured hybrid facts renderer with Turtle/SPARQL decision logic.
|
||||
|
||||
This function decides between generating bare Turtle for fresh facts
|
||||
and SPARQL operations for updates based on whether facts exist.
|
||||
|
||||
Args:
|
||||
state: The current unit facts state
|
||||
tools: The toolbox containing necessary tools
|
||||
|
||||
Returns:
|
||||
UnitFactsState: Updated state with rendered facts
|
||||
"""
|
||||
|
||||
is_fresh_facts_graph = len(state.content_unit.graph) == 0
|
||||
|
||||
progress_info = state.get_content_unit_progress_string()
|
||||
logger.info(f"Render facts for {progress_info}")
|
||||
|
||||
if is_fresh_facts_graph:
|
||||
logger.info("Generating fresh facts as Turtle")
|
||||
return await render_facts_fresh(state, tools)
|
||||
else:
|
||||
logger.info("Generating facts update")
|
||||
return await render_facts_update(state, tools)
|
||||
|
||||
|
||||
def _prepare_prompt_data(state: UnitFactsState) -> dict[str, str]:
|
||||
"""Prepare common prompt data for both fresh and update rendering.
|
||||
|
||||
Args:
|
||||
state: The current unit facts state
|
||||
|
||||
Returns:
|
||||
Dictionary containing formatted prompt components
|
||||
"""
|
||||
ontology_chapter = ontology_template.format(
|
||||
ontology_ttl=state.ontology_snapshot.graph.serialize(format="turtle")
|
||||
)
|
||||
|
||||
facts_instruction_str = facts_instruction_template.format(
|
||||
ontology_namespace=state.ontology_snapshot.namespace,
|
||||
ontology_prefix=state.ontology_snapshot.prefix,
|
||||
facts_namespace=DEFAULT_IRI,
|
||||
)
|
||||
|
||||
text_chapter = text_template.format(text=state.content_unit.text)
|
||||
|
||||
fact_chapter = ""
|
||||
|
||||
user_instruction = (
|
||||
user_template.format(user_instruction=state.facts_user_instruction)
|
||||
if state.facts_user_instruction
|
||||
else ""
|
||||
)
|
||||
|
||||
return {
|
||||
"ontology_chapter": ontology_chapter,
|
||||
"user_instruction": user_instruction,
|
||||
"facts_instruction": facts_instruction_str,
|
||||
"text_chapter": text_chapter,
|
||||
"fact_chapter": fact_chapter,
|
||||
}
|
||||
|
||||
|
||||
def _create_prompt_template() -> PromptTemplate:
|
||||
"""Create the common prompt template used by both rendering functions.
|
||||
|
||||
Returns:
|
||||
Configured PromptTemplate instance
|
||||
"""
|
||||
return PromptTemplate(
|
||||
template=template_prompt,
|
||||
input_variables=[
|
||||
"preamble",
|
||||
"facts_instruction",
|
||||
"user_instruction",
|
||||
"ontology_chapter",
|
||||
"text_chapter",
|
||||
"improvement_instruction",
|
||||
"output_instruction",
|
||||
"format_instructions",
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
def _handle_rendering_error(
|
||||
state: UnitFactsState, error: Exception, stage: FailureStage
|
||||
) -> UnitFactsState:
|
||||
"""Handle rendering errors consistently.
|
||||
|
||||
Args:
|
||||
state: The current agent state
|
||||
error: The exception that occurred
|
||||
stage: The failure stage to set
|
||||
|
||||
Returns:
|
||||
Updated state with failure information
|
||||
"""
|
||||
logger.error(f"Failed to generate triples: {str(error)}")
|
||||
state.set_failure(stage, str(error))
|
||||
state.set_node_status(WorkflowNode.TEXT_TO_FACTS, Status.FAILED)
|
||||
return state
|
||||
|
||||
|
||||
async def render_facts_fresh(
|
||||
state: UnitFactsState, tools: AtomicToolBox
|
||||
) -> UnitFactsState:
|
||||
"""Render fresh facts from the current chunk into Turtle format.
|
||||
|
||||
Args:
|
||||
state: The current unit facts state containing the chunk to render.
|
||||
tools: The toolbox instance providing utility functions.
|
||||
|
||||
Returns:
|
||||
UnitFactsState: Updated state with rendered facts.
|
||||
"""
|
||||
logger.info("Rendering fresh facts")
|
||||
llm_tool = await tools.get_llm_tool(state.budget_tracker)
|
||||
parser = PydanticOutputParser(pydantic_object=FactsRenderReport)
|
||||
|
||||
known_prefixes = _extract_known_prefixes(state)
|
||||
|
||||
prompt_data = _prepare_prompt_data(state)
|
||||
prompt_data_fresh = {
|
||||
"preamble": preamble,
|
||||
"improvement_instruction": "",
|
||||
"output_instruction": output_instruction_empty,
|
||||
}
|
||||
prompt_data.update(prompt_data_fresh)
|
||||
|
||||
prompt = _create_prompt_template()
|
||||
|
||||
try:
|
||||
# Set known prefixes in context before parsing
|
||||
RDFGraph.set_known_prefixes(known_prefixes if known_prefixes else None)
|
||||
|
||||
render_report: FactsRenderReport = await call_llm_with_retry(
|
||||
llm_tool=llm_tool,
|
||||
prompt=prompt,
|
||||
parser=parser,
|
||||
prompt_kwargs={
|
||||
"format_instructions": parser.get_format_instructions(),
|
||||
**prompt_data,
|
||||
},
|
||||
)
|
||||
state.set_external_evidence_request(
|
||||
WorkflowNode.TEXT_TO_FACTS, render_report.external_evidence_request
|
||||
)
|
||||
facts_report = render_report.facts_report
|
||||
facts_report.semantic_graph.sanitize_prefixes_namespaces()
|
||||
state.content_unit.graph = facts_report.semantic_graph
|
||||
|
||||
# Track triples in budget tracker (fresh facts)
|
||||
num_triples = len(facts_report.semantic_graph)
|
||||
logger.info(f"Fresh facts generated with {num_triples} triple(s).")
|
||||
state.budget_tracker.add_facts_update(num_operations=1, num_triples=num_triples)
|
||||
|
||||
state.clear_failure()
|
||||
state.set_node_status(WorkflowNode.TEXT_TO_FACTS, Status.SUCCESS)
|
||||
return state
|
||||
|
||||
except Exception as e:
|
||||
return _handle_rendering_error(state, e, FailureStage.GENERATE_TTL_FOR_FACTS)
|
||||
finally:
|
||||
# Clear the context after parsing
|
||||
RDFGraph.set_known_prefixes(None)
|
||||
|
||||
|
||||
async def render_facts_update(
|
||||
state: UnitFactsState, tools: AtomicToolBox
|
||||
) -> UnitFactsState:
|
||||
"""Render facts updates using SPARQL operations.
|
||||
|
||||
Args:
|
||||
state: The current unit facts state containing the chunk to render.
|
||||
tools: The toolbox instance providing utility functions.
|
||||
|
||||
Returns:
|
||||
UnitFactsState: Updated state with rendered facts.
|
||||
"""
|
||||
logger.info("Rendering updates for facts")
|
||||
llm_tool = await tools.get_llm_tool(state.budget_tracker)
|
||||
parser = PydanticOutputParser(pydantic_object=GraphUpdateRenderReport)
|
||||
|
||||
prompt_data = _prepare_prompt_data(state)
|
||||
prompt_data_update = {
|
||||
"preamble": preamble,
|
||||
"improvement_instruction": render_suggestions_prompt(
|
||||
state.suggestions, WorkflowNode.TEXT_TO_FACTS
|
||||
),
|
||||
"output_instruction": output_instruction_sparql,
|
||||
"fact_chapter": facts_template.format(
|
||||
facts_ttl=state.content_unit.graph.serialize(format="turtle")
|
||||
),
|
||||
}
|
||||
prompt_data.update(prompt_data_update)
|
||||
prompt = _create_prompt_template()
|
||||
known_prefixes = _extract_known_prefixes(state)
|
||||
|
||||
try:
|
||||
# Set known prefixes in context before parsing
|
||||
RDFGraph.set_known_prefixes(known_prefixes if known_prefixes else None)
|
||||
|
||||
render_report: GraphUpdateRenderReport = await call_llm_with_retry(
|
||||
llm_tool=llm_tool,
|
||||
prompt=prompt,
|
||||
parser=parser,
|
||||
prompt_kwargs={
|
||||
"format_instructions": parser.get_format_instructions(),
|
||||
**prompt_data,
|
||||
},
|
||||
)
|
||||
state.set_external_evidence_request(
|
||||
WorkflowNode.TEXT_TO_FACTS, render_report.external_evidence_request
|
||||
)
|
||||
graph_update = render_report.graph_update
|
||||
state.facts_updates.append(graph_update)
|
||||
state.update_facts()
|
||||
|
||||
num_operations, num_triples = graph_update.count_total_triples()
|
||||
logger.info(
|
||||
f"Facts update has {num_operations} operation(s) "
|
||||
f"with {num_triples} total triple(s)."
|
||||
)
|
||||
|
||||
# Track triples in budget tracker
|
||||
state.budget_tracker.add_facts_update(num_operations, num_triples)
|
||||
|
||||
state.set_node_status(WorkflowNode.TEXT_TO_FACTS, Status.SUCCESS)
|
||||
state.clear_failure()
|
||||
return state
|
||||
|
||||
except Exception as e:
|
||||
return _handle_rendering_error(
|
||||
state, e, FailureStage.GENERATE_SPARQL_UPDATE_FOR_FACTS
|
||||
)
|
||||
finally:
|
||||
# Clear the context after parsing
|
||||
RDFGraph.set_known_prefixes(None)
|
||||
Reference in New Issue
Block a user