From bec254fe97a354ddd3e23df63d031b81a822966b Mon Sep 17 00:00:00 2001 From: Paul Lizer Date: Tue, 4 Aug 2026 12:25:07 -0400 Subject: [PATCH 1/3] Support generated JSON and XML exports Expand the generated export artifact framework to capture valid JSON and XML outputs as downloadable chat artifacts, add XML serialization for durable generated exports, and consolidate XML document processing through the token-aware pipeline. Refs #1071. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- application/single_app/config.py | 2 +- .../single_app/functions_document_analysis.py | 126 ++++++++- application/single_app/functions_documents.py | 133 +++------ .../functions_generated_file_exports.py | 176 ++++++++++++ .../functions_tabular_generated_exports.py | 17 +- .../single_app/functions_workflow_runner.py | 100 ++++++- application/single_app/route_backend_chats.py | 259 +++++++++++++++++- .../fixes/GENERATED_JSON_XML_EXPORTS_FIX.md | 57 ++++ .../test_generated_json_xml_exports.py | 165 +++++++++++ 9 files changed, 926 insertions(+), 109 deletions(-) create mode 100644 application/single_app/functions_generated_file_exports.py create mode 100644 docs/explanation/fixes/GENERATED_JSON_XML_EXPORTS_FIX.md create mode 100644 functional_tests/test_generated_json_xml_exports.py diff --git a/application/single_app/config.py b/application/single_app/config.py index d2957115a..13c09a7e5 100644 --- a/application/single_app/config.py +++ b/application/single_app/config.py @@ -95,7 +95,7 @@ EXECUTOR_TYPE = 'thread' EXECUTOR_MAX_WORKERS = 30 SESSION_TYPE = 'filesystem' -VERSION = "0.250.112" +VERSION = "0.250.113" IS_DEVELOPMENT = is_development_env_enabled() SESSION_COOKIE_SAMESITE = os.getenv('SESSION_COOKIE_SAMESITE', 'Lax') diff --git a/application/single_app/functions_document_analysis.py b/application/single_app/functions_document_analysis.py index d82d8de04..e3308929d 100644 --- a/application/single_app/functions_document_analysis.py +++ b/application/single_app/functions_document_analysis.py @@ -363,6 +363,119 @@ def _build_window_label(document_name, window_range): return f"{document_name} - window {window_range.get('window_number')} ({range_label})" +def _prompt_requests_json_output(analysis_prompt): + prompt_text = str(analysis_prompt or '').strip().lower() + if not prompt_text: + return False + + json_markers = ( + 'json artifact', + 'json export', + 'json output', + 'json array', + 'json object', + 'json file', + 'json format', + 'valid json', + 'convert into json', + 'convert to json', + 'return json', + 'return only json', + 'respond with json', + 'format as json', + 'output as json', + 'save as json', + 'export as json', + 'download as json', + 'create json', + 'create a json', + 'make json', + 'make a json', + 'generate json', + 'generate a json', + ) + if any(marker in prompt_text for marker in json_markers): + return True + + return bool(re.search( + r'\b(convert|create|make|build|generate|produce|return|respond|format|output|save|export|download)\b[\w\s.,:;\-/]{0,80}\bjson\b', + prompt_text, + )) + + +def _prompt_requests_xml_output(analysis_prompt): + prompt_text = str(analysis_prompt or '').strip().lower() + if not prompt_text: + return False + + xml_markers = ( + 'xml artifact', + 'xml export', + 'xml output', + 'xml document', + 'xml file', + 'xml template', + 'valid xml', + 'well-formed xml', + 'convert into xml', + 'convert to xml', + 'populate xml', + 'populate the xml', + 'return xml', + 'return only xml', + 'respond with xml', + 'format as xml', + 'output as xml', + 'save as xml', + 'export as xml', + 'download as xml', + 'create xml', + 'create an xml', + 'make xml', + 'make an xml', + 'generate xml', + 'generate an xml', + ) + if any(marker in prompt_text for marker in xml_markers): + return True + + return bool(re.search( + r'\b(convert|populate|create|make|build|generate|produce|return|respond|format|output|save|export|download)\b[\w\s.,:;\-/]{0,80}\bxml\b', + prompt_text, + )) + + +def _build_requested_output_guidance(analysis_prompt, stage): + if _prompt_requests_xml_output(analysis_prompt): + if stage == 'slice': + return ( + 'The overall task requests XML output. For this slice, preserve exact XML element names, ' + 'attribute names, nesting, template placeholders, and source values needed to produce the final XML. ' + 'Do not condense repeated XML structures when they are visible in this slice. If this slice contains ' + 'everything needed to satisfy the task, return only the complete well-formed XML document.\n\n' + ) + return ( + 'The original task requests an XML file. Return only one complete well-formed XML document for the final ' + 'answer, without Markdown fences, prose, citations, or explanatory text outside the XML. Preserve the ' + 'requested template structure whenever a template is supplied.\n\n' + ) + + if _prompt_requests_json_output(analysis_prompt): + if stage == 'slice': + return ( + 'The overall task requests JSON output. For this slice, preserve exact field names, hierarchy, arrays, ' + 'template placeholders, and source values needed to produce the final JSON. Do not condense repeated ' + 'structures when they are visible in this slice. If this slice contains everything needed to satisfy ' + 'the task, return only valid JSON.\n\n' + ) + return ( + 'The original task requests a JSON file. Return only valid JSON for the final answer, without Markdown ' + 'fences, prose, citations, or explanatory text outside the JSON.\n\n' + ) + + return '' + + def _build_window_analysis_prompt(analysis_prompt, document_payload, window_payload, window_range): document_file_name = _resolve_document_file_name(document_payload) document_title = _resolve_document_title(document_payload) @@ -387,6 +500,7 @@ def _build_window_analysis_prompt(analysis_prompt, document_payload, window_payl f'Page count in slice: {window_range.get("page_count", 0)}\n\n' 'Task instructions:\n' f'{analysis_prompt}\n\n' + f'{_build_requested_output_guidance(analysis_prompt, "slice")}' 'Write a focused analysis of this slice. Preserve concrete facts, decisions, comments, action items, ' 'and open questions. Call out anything that still needs follow-up.\n\n' f'\n{_render_window_source_text(window_payload)}\n' @@ -508,11 +622,15 @@ def _prompt_requests_exhaustive_output(analysis_prompt): def _build_analysis_intent(analysis_prompt): per_source_output_requested = _prompt_requests_per_source_output(analysis_prompt) + json_output_requested = _prompt_requests_json_output(analysis_prompt) + xml_output_requested = _prompt_requests_xml_output(analysis_prompt) json_array_output_requested = _prompt_requests_json_array_output(analysis_prompt) json_code_block_requested = _prompt_requests_json_code_block(analysis_prompt) table_output_requested = _prompt_requests_table_output(analysis_prompt) exhaustive_output_requested = ( per_source_output_requested + or json_output_requested + or xml_output_requested or json_array_output_requested or table_output_requested or _prompt_requests_exhaustive_output(analysis_prompt) @@ -522,11 +640,13 @@ def _build_analysis_intent(analysis_prompt): 'exhaustive': exhaustive_output_requested, 'preserve_raw_outputs': True, 'per_source_output_requested': per_source_output_requested, + 'json_output_requested': json_output_requested, + 'xml_output_requested': xml_output_requested, 'json_array_output_requested': json_array_output_requested, 'json_code_block_requested': json_code_block_requested, 'table_output_requested': table_output_requested, - 'csv_artifact_recommended': table_output_requested or exhaustive_output_requested, - 'markdown_analysis_artifact_recommended': exhaustive_output_requested, + 'csv_artifact_recommended': table_output_requested or (exhaustive_output_requested and not json_output_requested and not xml_output_requested), + 'markdown_analysis_artifact_recommended': exhaustive_output_requested and not json_output_requested and not xml_output_requested, } @@ -627,6 +747,7 @@ def _build_reduction_prompt(analysis_prompt, items, stage_label, failed_range_la f'Task instructions:\n{analysis_prompt}\n\n' f'{failed_note}' f'{preservation_note}' + f'{_build_requested_output_guidance(analysis_prompt, "reduction")}' f'{combine_instruction}\n\n' f'\n{combined_text}\n' ) @@ -659,6 +780,7 @@ def _build_document_reduction_prompt(analysis_prompt, document_name, items, stag f'Source document: {document_name}\n' f'Task instructions:\n{analysis_prompt}\n\n' f'{failed_note}' + f'{_build_requested_output_guidance(analysis_prompt, "reduction")}' 'Combine the slice analyses below into one document-level answer.\n\n' f'\n{combined_text}\n' ) diff --git a/application/single_app/functions_documents.py b/application/single_app/functions_documents.py index b9bdb48f3..141ce5b6e 100644 --- a/application/single_app/functions_documents.py +++ b/application/single_app/functions_documents.py @@ -5159,7 +5159,7 @@ def process_txt(document_id, user_id, temp_file_path, original_filename, enable_ return total_chunks_saved, total_embedding_tokens, embedding_model_name -def process_xml(document_id, user_id, temp_file_path, original_filename, enable_enhanced_citations, update_callback, group_id=None, public_workspace_id=None): +def _process_xml_with_token_usage(document_id, user_id, temp_file_path, original_filename, enable_enhanced_citations, update_callback, group_id=None, public_workspace_id=None): """Processes XML files using RecursiveCharacterTextSplitter for structured content.""" is_group = group_id is not None is_public_workspace = public_workspace_id is not None @@ -5215,7 +5215,16 @@ def process_xml(document_id, user_id, temp_file_path, original_filename, enable_ for idx, chunk_content in enumerate(final_chunks, start=1): # Skip empty chunks if not chunk_content or not chunk_content.strip(): - print(f"Skipping empty XML chunk {idx}/{initial_chunk_count}") + log_event( + '[Documents] Skipping empty XML chunk', + { + 'document_id': document_id, + 'file_name': original_filename, + 'chunk_index': idx, + 'chunk_count': initial_chunk_count, + }, + debug_only=True, + ) continue update_callback( @@ -5247,10 +5256,29 @@ def process_xml(document_id, user_id, temp_file_path, original_filename, enable_ # Final update with actual chunks saved if total_chunks_saved != initial_chunk_count: update_callback(number_of_pages=total_chunks_saved) - print(f"Adjusted final chunk count from {initial_chunk_count} to {total_chunks_saved} after skipping empty chunks.") + log_event( + '[Documents] Adjusted XML chunk count after skipping empty chunks', + { + 'document_id': document_id, + 'file_name': original_filename, + 'initial_chunk_count': initial_chunk_count, + 'total_chunks_saved': total_chunks_saved, + }, + debug_only=True, + ) except Exception as e: - print(f"Error during XML processing for {original_filename}: {type(e).__name__}: {e}") + log_event( + '[Documents] XML processing failed', + { + 'document_id': document_id, + 'file_name': original_filename, + 'error_type': type(e).__name__, + 'error': str(e), + }, + level=logging.ERROR, + exceptionTraceback=True, + ) raise Exception(f"Failed processing XML file {original_filename}: {e}") return total_chunks_saved, total_embedding_tokens, embedding_model_name @@ -5539,92 +5567,17 @@ def process_doc(document_id, user_id, temp_file_path, original_filename, enable_ return total_chunks_saved, total_embedding_tokens, embedding_model_name def process_xml(document_id, user_id, temp_file_path, original_filename, enable_enhanced_citations, update_callback, group_id=None, public_workspace_id=None): - """Processes XML files using RecursiveCharacterTextSplitter for structured content.""" - is_group = group_id is not None - is_public_workspace = public_workspace_id is not None - - update_callback(status="Processing XML file...") - total_chunks_saved = 0 - # Character-based chunking for XML structure preservation, capped by embedding context - chunk_config = get_chunk_size_config(get_settings()) - max_chunk_size_chars = chunk_config.get('xml', {}).get('value', 4000) - - if enable_enhanced_citations: - args = { - "temp_file_path": temp_file_path, - "user_id": user_id, - "document_id": document_id, - "blob_filename": original_filename, - "update_callback": update_callback - } - - if is_group: - args["group_id"] = group_id - elif is_public_workspace: - args["public_workspace_id"] = public_workspace_id - - upload_to_blob(**args) - - try: - # Read XML content - try: - with open(temp_file_path, 'r', encoding='utf-8') as f: - xml_content = f.read() - except Exception as e: - raise Exception(f"Error reading XML file {original_filename}: {e}") - - # Use RecursiveCharacterTextSplitter with XML-aware separators - # This preserves XML structure better than simple word splitting - xml_splitter = RecursiveCharacterTextSplitter( - chunk_size=max_chunk_size_chars, - chunk_overlap=0, - length_function=len, - separators=["\n\n", "\n", ">", " ", ""], # XML-friendly separators - is_separator_regex=False - ) - - # Split the XML content - final_chunks = xml_splitter.split_text(xml_content) - - initial_chunk_count = len(final_chunks) - update_callback(number_of_pages=initial_chunk_count) - - for idx, chunk_content in enumerate(final_chunks, start=1): - # Skip empty chunks - if not chunk_content or not chunk_content.strip(): - print(f"Skipping empty XML chunk {idx}/{initial_chunk_count}") - continue - - update_callback( - current_file_chunk=idx, - status=f"Saving chunk {idx}/{initial_chunk_count}..." - ) - args = { - "page_text_content": chunk_content, - "page_number": total_chunks_saved + 1, - "file_name": original_filename, - "user_id": user_id, - "document_id": document_id - } - - if is_public_workspace: - args["public_workspace_id"] = public_workspace_id - elif is_group: - args["group_id"] = group_id - - save_chunks(**args) - total_chunks_saved += 1 - - # Final update with actual chunks saved - if total_chunks_saved != initial_chunk_count: - update_callback(number_of_pages=total_chunks_saved) - print(f"Adjusted final chunk count from {initial_chunk_count} to {total_chunks_saved} after skipping empty chunks.") - - except Exception as e: - print(f"Error during XML processing for {original_filename}: {type(e).__name__}: {e}") - raise Exception(f"Failed processing XML file {original_filename}: {e}") - - return total_chunks_saved + """Processes XML files using the consolidated token-aware XML pipeline.""" + return _process_xml_with_token_usage( + document_id, + user_id, + temp_file_path, + original_filename, + enable_enhanced_citations, + update_callback, + group_id=group_id, + public_workspace_id=public_workspace_id, + ) def process_yaml(document_id, user_id, temp_file_path, original_filename, enable_enhanced_citations, update_callback, group_id=None, public_workspace_id=None): """Processes YAML files using RecursiveCharacterTextSplitter for structured content.""" diff --git a/application/single_app/functions_generated_file_exports.py b/application/single_app/functions_generated_file_exports.py new file mode 100644 index 000000000..651976630 --- /dev/null +++ b/application/single_app/functions_generated_file_exports.py @@ -0,0 +1,176 @@ +# functions_generated_file_exports.py +"""Shared helpers for generated downloadable file exports.""" + +import json +import re +from typing import Any, Iterable +from xml.etree import ElementTree + + +SUPPORTED_GENERATED_EXPORT_FORMATS = {'csv', 'json', 'xml'} +XML_DECLARATION = '' +XML_ROOT_PATTERN = re.compile(r'<(?P[A-Za-z_][A-Za-z0-9_.:-]*)(?:\s[^<>]*)?>') + + +def normalize_generated_output_format(output_format, default='json'): + """Normalize generated artifact output formats supported by the export framework.""" + normalized_format = str(output_format or '').strip().lower().lstrip('.') + if normalized_format in SUPPORTED_GENERATED_EXPORT_FORMATS: + return normalized_format + + normalized_default = str(default or 'json').strip().lower().lstrip('.') + if normalized_default in SUPPORTED_GENERATED_EXPORT_FORMATS: + return normalized_default + return 'json' + + +def strip_markdown_code_fence(text): + """Remove a single surrounding Markdown code fence while preserving content.""" + normalized_text = str(text or '').strip() + if not normalized_text.startswith('```'): + return normalized_text + + code_fence_match = re.fullmatch( + r'```(?:[a-zA-Z0-9_-]+)?\s*(.*?)\s*```', + normalized_text, + re.DOTALL, + ) + if not code_fence_match: + return normalized_text + + return str(code_fence_match.group(1) or '').strip() + + +def _iter_xml_candidates(text) -> Iterable[str]: + normalized_text = strip_markdown_code_fence(text) + if not normalized_text: + return + + yield normalized_text + + first_xml_index = normalized_text.find(' 0: + yield normalized_text[first_xml_index:].strip() + + first_tag_index = normalized_text.find('<') + if first_tag_index > 0: + yield normalized_text[first_tag_index:].strip() + + for root_match in XML_ROOT_PATTERN.finditer(normalized_text): + root_tag = root_match.group('tag') + root_start = root_match.start() + root_open = root_match.group(0) + if root_open.rstrip().endswith('/>'): + yield normalized_text[root_start:root_match.end()].strip() + continue + + closing_tag = f'' + root_end = normalized_text.rfind(closing_tag) + if root_end <= root_start: + continue + + yield normalized_text[root_start:root_end + len(closing_tag)].strip() + + +def normalize_xml_artifact_payload(text): + """Return a complete XML document extracted from model output, or an empty string.""" + seen_candidates = set() + for candidate in _iter_xml_candidates(text): + if candidate in seen_candidates: + continue + seen_candidates.add(candidate) + try: + ElementTree.fromstring(candidate.encode('utf-8')) + except ElementTree.ParseError: + continue + return candidate + return '' + + +def normalize_json_artifact_payload(text): + """Return parsed JSON extracted from model output, or None when no JSON is present.""" + normalized_text = strip_markdown_code_fence(text) + if not normalized_text: + return None + + decoder = json.JSONDecoder() + try: + parsed_value, _ = decoder.raw_decode(normalized_text) + return parsed_value + except (TypeError, ValueError, json.JSONDecodeError): + pass + + for start_index, character in enumerate(normalized_text): + if character not in '[{': + continue + try: + parsed_value, _ = decoder.raw_decode(normalized_text[start_index:]) + return parsed_value + except (TypeError, ValueError, json.JSONDecodeError): + continue + + return None + + +def _sanitize_xml_tag_name(value, fallback_value): + normalized_name = re.sub(r'[^A-Za-z0-9_.-]+', '_', str(value or '').strip()) + normalized_name = normalized_name.strip('._-') + if not normalized_name: + normalized_name = fallback_value + if not re.match(r'^[A-Za-z_]', normalized_name): + normalized_name = f'{fallback_value}_{normalized_name}' + return normalized_name + + +def _append_xml_value(parent, value, item_name): + if isinstance(value, dict): + for key, child_value in value.items(): + child = ElementTree.SubElement( + parent, + _sanitize_xml_tag_name(key, 'Field'), + ) + _append_xml_value(child, child_value, item_name) + return + + if isinstance(value, (list, tuple)): + for item in value: + child = ElementTree.SubElement( + parent, + _sanitize_xml_tag_name(item_name, 'Item'), + ) + _append_xml_value(child, item, item_name) + return + + if value is None: + parent.text = '' + return + + if isinstance(value, bool): + parent.text = 'true' if value else 'false' + return + + parent.text = str(value) + + +def build_xml_from_value(value: Any, root_name='GeneratedOutput', item_name='Item'): + """Serialize a Python value into a deterministic XML document.""" + root = ElementTree.Element(_sanitize_xml_tag_name(root_name, 'GeneratedOutput')) + _append_xml_value(root, value, item_name) + ElementTree.indent(root, space=' ') + xml_body = ElementTree.tostring(root, encoding='unicode', short_empty_elements=True) + return f'{XML_DECLARATION}\n{xml_body}' + + +def serialize_generated_xml(value: Any, root_name='GeneratedOutput', item_name='Item'): + """Serialize generated content to XML, preserving valid XML model output when present.""" + if isinstance(value, str): + xml_payload = normalize_xml_artifact_payload(value) + if xml_payload: + return xml_payload + + return build_xml_from_value(value, root_name=root_name, item_name=item_name) + + +def serialize_generated_json(value: Any, *, indent=2): + """Serialize generated content to JSON using the export framework defaults.""" + return json.dumps(value, indent=indent, ensure_ascii=False, default=str) diff --git a/application/single_app/functions_tabular_generated_exports.py b/application/single_app/functions_tabular_generated_exports.py index 6d27ef76d..91700a347 100644 --- a/application/single_app/functions_tabular_generated_exports.py +++ b/application/single_app/functions_tabular_generated_exports.py @@ -25,6 +25,11 @@ storage_account_personal_chat_container_name, ) from functions_appinsights import log_event +from functions_generated_file_exports import ( + normalize_generated_output_format, + serialize_generated_json, + serialize_generated_xml, +) from functions_model_endpoint_runtime import build_semantic_kernel_chat_service_for_model from functions_settings import get_settings from functions_simplechat_operations import upload_generated_analysis_artifact_for_user @@ -188,7 +193,7 @@ def _sanitize_file_base_name(file_name): def _build_generated_file_name(source_file_name, output_format): timestamp_suffix = datetime.utcnow().strftime('%Y%m%d_%H%M%S') - normalized_extension = 'csv' if str(output_format or '').strip().lower() == 'csv' else 'json' + normalized_extension = normalize_generated_output_format(output_format) return f"{_sanitize_file_base_name(source_file_name)}_generated_{timestamp_suffix}.{normalized_extension}" @@ -1279,11 +1284,17 @@ def _assemble_output_entries(run): def _complete_run(run): output_entries = _assemble_output_entries(run) - output_format = str(run.get('output_format') or 'json').strip().lower() or 'json' + output_format = normalize_generated_output_format(run.get('output_format')) if output_format == 'csv': serialized_output = _build_generated_output_csv(output_entries) + elif output_format == 'xml': + serialized_output = serialize_generated_xml( + output_entries, + root_name='GeneratedRows', + item_name='Row', + ) else: - serialized_output = json.dumps(output_entries, indent=2, default=str, ensure_ascii=False) + serialized_output = serialize_generated_json(output_entries) generated_file_name = run.get('generated_file_name') or _build_generated_file_name( run.get('source_file_name'), diff --git a/application/single_app/functions_workflow_runner.py b/application/single_app/functions_workflow_runner.py index 4d8c7347f..627dcf411 100644 --- a/application/single_app/functions_workflow_runner.py +++ b/application/single_app/functions_workflow_runner.py @@ -59,6 +59,10 @@ get_collaboration_conversation, mirror_source_message_to_collaboration, ) +from functions_generated_file_exports import ( + normalize_xml_artifact_payload, + serialize_generated_json, +) from functions_document_actions import ( DOCUMENT_ACTION_ANALYSIS_MODE_PER_DOCUMENT, DOCUMENT_ACTION_CONTEXT_WORKFLOW, @@ -342,6 +346,7 @@ def _prompt_explicitly_requests_artifact(analysis_prompt): 'save it as', 'save to file', 'json file', + 'xml file', 'csv file', 'markdown file', ) @@ -361,6 +366,8 @@ def _prompt_explicitly_requests_json_artifact(analysis_prompt): 'json object', 'json format', 'valid json', + 'convert into json', + 'convert to json', 'return json', 'return only json', 'return only valid json', @@ -390,7 +397,61 @@ def _prompt_explicitly_requests_json_artifact(analysis_prompt): return True return bool(re.search( - r'\b(create|make|build|generate|produce|return|respond|format|output|save|export|download)\b[\w\s.,:;\-/]{0,60}\bjson\b', + r'\b(convert|create|make|build|generate|produce|return|respond|format|output|save|export|download)\b[\w\s.,:;\-/]{0,60}\bjson\b', + prompt_text, + )) + + +def _prompt_explicitly_requests_xml_artifact(analysis_prompt): + prompt_text = str(analysis_prompt or '').strip().lower() + if not prompt_text: + return False + + xml_markers = ( + 'xml artifact', + 'xml export', + 'xml output', + 'xml document', + 'xml file', + 'xml template', + 'valid xml', + 'well-formed xml', + 'convert into xml', + 'convert to xml', + 'populate xml', + 'populate the xml', + 'return xml', + 'return only xml', + 'return only valid xml', + 'respond with xml', + 'format as xml', + 'output as xml', + 'save as xml', + 'save it as xml', + 'export as xml', + 'download as xml', + 'create xml', + 'create an xml', + 'create a xml', + 'make xml', + 'make an xml', + 'make a xml', + 'generate xml', + 'generate an xml', + 'produce xml', + 'produce an xml', + 'save to .xml', + 'export to .xml', + 'download .xml', + 'create .xml', + 'make .xml', + 'generate .xml', + ) + if any(marker in prompt_text for marker in xml_markers): + return True + + return bool(re.search( + r'\b(convert|populate|create|make|build|generate|produce|return|respond|format|output|save|export|download)\b[\w\s.,:;\-/]{0,80}\bxml\b', prompt_text, )) @@ -808,6 +869,8 @@ def _prompt_requests_table_analysis_output(analysis_prompt): def _get_document_analysis_artifact_intent(analysis_result, analysis_prompt): analysis_result = analysis_result if isinstance(analysis_result, dict) else {} analysis_intent = analysis_result.get('analysis_intent') if isinstance(analysis_result.get('analysis_intent'), dict) else {} + json_artifact_requested = _prompt_explicitly_requests_json_artifact(analysis_prompt) + xml_artifact_requested = _prompt_explicitly_requests_xml_artifact(analysis_prompt) table_output_requested = bool( analysis_intent.get('table_output_requested') or _prompt_requests_table_analysis_output(analysis_prompt) @@ -821,14 +884,16 @@ def _get_document_analysis_artifact_intent(analysis_result, analysis_prompt): return { 'exhaustive': exhaustive_output_requested, 'table_output_requested': table_output_requested, + 'json_artifact_requested': json_artifact_requested, + 'xml_artifact_requested': xml_artifact_requested, 'csv_artifact_recommended': bool( analysis_intent.get('csv_artifact_recommended') or table_output_requested - or exhaustive_output_requested + or (exhaustive_output_requested and not json_artifact_requested and not xml_artifact_requested) ), 'markdown_analysis_artifact_recommended': bool( analysis_intent.get('markdown_analysis_artifact_recommended') - or exhaustive_output_requested + or (exhaustive_output_requested and not json_artifact_requested and not xml_artifact_requested) ), } @@ -1245,9 +1310,12 @@ def _maybe_create_document_analysis_generated_artifacts( raw_analysis_items = analysis_result.get('raw_analysis_items') if isinstance(analysis_result.get('raw_analysis_items'), list) else [] json_payload = _parse_json_artifact_payload(analysis_reply) json_artifact_requested = _prompt_explicitly_requests_json_artifact(analysis_prompt) + xml_payload = normalize_xml_artifact_payload(analysis_reply) + xml_artifact_requested = bool(artifact_intent.get('xml_artifact_requested')) create_lossless_artifacts = bool( artifact_intent.get('exhaustive') or artifact_intent.get('table_output_requested') + or xml_artifact_requested or primary_tabular_outputs ) @@ -1302,6 +1370,20 @@ def _maybe_create_document_analysis_generated_artifacts( if markdown_artifact: artifacts.append(markdown_artifact) + if xml_payload and xml_artifact_requested and not primary_tabular_outputs: + xml_file_name = _build_document_analysis_artifact_file_name(analysis_result, 'xml') + xml_summary = _build_document_analysis_artifact_summary(document_count, 'xml') + xml_artifact = _upload_document_analysis_generated_artifact( + normalized_conversation_id, + xml_file_name, + xml_payload, + 'xml', + xml_summary, + preview_lines=_build_document_analysis_preview_lines(xml_payload), + ) + if xml_artifact: + artifacts.append(xml_artifact) + if json_payload is not None and json_artifact_requested and not primary_tabular_outputs: json_file_name = _build_document_analysis_artifact_file_name(analysis_result, 'json') json_summary = _build_document_analysis_artifact_summary(document_count, 'json') @@ -1313,7 +1395,7 @@ def _maybe_create_document_analysis_generated_artifacts( json_artifact = _upload_document_analysis_generated_artifact( normalized_conversation_id, json_file_name, - json.dumps(json_payload, indent=2, ensure_ascii=False), + serialize_generated_json(json_payload), 'json', json_summary, preview_items=json_preview_items, @@ -1359,17 +1441,21 @@ def _maybe_create_document_analysis_generated_artifacts( should_generate_artifact = ( explicit_artifact_request or json_payload is not None + or bool(xml_payload) or len(analysis_reply) >= DOCUMENT_ANALYSIS_ARTIFACT_REPLY_CHAR_THRESHOLD ) if not should_generate_artifact: return {'artifacts': [], 'assistant_reply': None} - output_format = 'json' if json_payload is not None and json_artifact_requested else 'md' + output_format = 'xml' if xml_payload and xml_artifact_requested else 'json' if json_payload is not None and json_artifact_requested else 'md' preview_items = [] preview_lines = [] - if output_format == 'json': - serialized_output = json.dumps(json_payload, indent=2, ensure_ascii=False) + if output_format == 'xml': + serialized_output = xml_payload + preview_lines = _build_document_analysis_preview_lines(xml_payload) + elif output_format == 'json': + serialized_output = serialize_generated_json(json_payload) if isinstance(json_payload, list): preview_items = json_payload[:DOCUMENT_ANALYSIS_ARTIFACT_PREVIEW_ITEM_COUNT] elif isinstance(json_payload, dict): diff --git a/application/single_app/route_backend_chats.py b/application/single_app/route_backend_chats.py index 2bfc73d7b..a4adb5374 100644 --- a/application/single_app/route_backend_chats.py +++ b/application/single_app/route_backend_chats.py @@ -129,6 +129,13 @@ ) from functions_appinsights import log_event from functions_debug import debug_print +from functions_generated_file_exports import ( + normalize_json_artifact_payload, + normalize_generated_output_format, + normalize_xml_artifact_payload, + serialize_generated_json, + serialize_generated_xml, +) from functions_governance import ensure_governance_access from functions_notifications import create_chat_response_notification from functions_activity_logging import log_agent_run, log_chat_activity, log_conversation_creation, log_token_usage @@ -1530,6 +1537,173 @@ def maybe_create_assistant_table_generated_output( } +def _has_generated_file_output(existing_outputs, output_format): + normalized_output_format = normalize_generated_output_format(output_format) + for generated_output in existing_outputs or []: + if not isinstance(generated_output, dict): + continue + + raw_existing_output_format = str( + generated_output.get('output_format') or os.path.splitext(str(generated_output.get('file_name') or ''))[1], + ).strip().lower().lstrip('.') + if raw_existing_output_format not in {'csv', 'json', 'xml'}: + continue + existing_output_format = normalize_generated_output_format(raw_existing_output_format) + if existing_output_format != normalized_output_format: + continue + if ( + generated_output.get('artifact_message_id') + or generated_output.get('document_id') + or generated_output.get('export_run_id') + or generated_output.get('run_id') + ): + return True + + return False + + +def _assistant_content_disclaims_complete_file(assistant_content): + normalized_content = str(assistant_content or '').strip().lower() + if not normalized_content: + return False + + disclaimer_markers = ( + 'partial conversion', + 'partial json', + 'partial xml', + 'evidence envelope is truncated', + 'evidence_envelope_truncated', + 'not fully available', + 'only partially visible', + 'omitted rather than invented', + ) + return any(marker in normalized_content[:2000] for marker in disclaimer_markers) + + +def _build_assistant_file_export_name(output_format): + normalized_output_format = normalize_generated_output_format(output_format) + timestamp_suffix = datetime.utcnow().strftime('%Y%m%d_%H%M%S') + return f'assistant_generated_{timestamp_suffix}.{normalized_output_format}' + + +def _build_assistant_file_preview_lines(file_content, max_lines=5, max_line_length=220): + preview_lines = [] + for line in str(file_content or '').splitlines(): + normalized_line = str(line or '').strip() + if not normalized_line: + continue + if len(normalized_line) > max_line_length: + normalized_line = f'{normalized_line[:max_line_length - 1]}…' + preview_lines.append(normalized_line) + if len(preview_lines) >= max_lines: + break + return preview_lines + + +def _build_assistant_file_output_handoff(output_metadata): + output_format = str(output_metadata.get('output_format') or 'file').strip().upper() + file_name = str(output_metadata.get('file_name') or '').strip() + if file_name: + return ( + f'I created a downloadable {output_format} file and attached it to this chat as "{file_name}". ' + 'Use the download control on the artifact card for the full output.' + ) + return ( + f'I created a downloadable {output_format} file and attached it to this chat. ' + 'Use the download control on the artifact card for the full output.' + ) + + +def maybe_create_assistant_file_generated_output( + user_question, + assistant_content, + conversation_id, + existing_outputs=None, +): + """Save assistant-generated JSON/XML content as a downloadable chat artifact.""" + output_format = get_tabular_generated_output_format(user_question) + if output_format not in {'json', 'xml'}: + return None + if _has_generated_file_output(existing_outputs, output_format): + return None + if _assistant_content_disclaims_complete_file(assistant_content): + return None + + preview_items = [] + preview_lines = [] + if output_format == 'json': + json_payload = normalize_json_artifact_payload(assistant_content) + if json_payload is None: + return None + file_content = serialize_generated_json(json_payload) + if isinstance(json_payload, list): + preview_items = json_payload[:3] + elif isinstance(json_payload, dict): + preview_items = [json_payload] + else: + xml_payload = normalize_xml_artifact_payload(assistant_content) + if not xml_payload: + return None + file_content = xml_payload + preview_lines = _build_assistant_file_preview_lines(file_content) + + generated_file_name = _build_assistant_file_export_name(output_format) + summary = ( + f'Saved the generated {output_format.upper()} output in this chat as a downloadable file artifact.' + ) + try: + upload_result = upload_generated_analysis_artifact_for_current_user( + conversation_id=conversation_id, + file_name=generated_file_name, + file_content=file_content, + capability='analysis', + output_format=output_format, + summary=summary, + ) + except Exception as exc: + log_event( + '[Assistant File Export] Failed to save assistant generated file artifact', + { + 'conversation_id': conversation_id, + 'generated_file_name': generated_file_name, + 'output_format': output_format, + 'error': str(exc), + }, + debug_only=True, + ) + return None + + artifact_message_id = upload_result.get('message', {}).get('id') + if not artifact_message_id: + return None + + uploaded_file_name = upload_result.get('message', {}).get('file_name') or generated_file_name + log_event( + '[Assistant File Export] Saved assistant generated file artifact', + { + 'conversation_id': conversation_id, + 'artifact_message_id': artifact_message_id, + 'generated_file_name': uploaded_file_name, + 'output_format': output_format, + }, + debug_only=True, + ) + output_metadata = { + 'capability': 'analysis', + 'artifact_message_id': artifact_message_id, + 'conversation_id': conversation_id, + 'storage_scope': 'chat', + 'file_name': uploaded_file_name, + 'output_format': output_format, + 'summary': summary, + } + if preview_items: + output_metadata['preview_items'] = preview_items + if preview_lines: + output_metadata['preview_lines'] = preview_lines + return output_metadata + + def _safe_int(value, default=0): try: return int(value) @@ -3903,19 +4077,51 @@ def get_tabular_generated_output_format(user_question): return None json_markers = ( + 'convert into json', + 'convert to json', 'json array', 'json file', 'download json', 'save json', 'make a json', 'create a json', + 'generate json', + 'generate a json', 'return json', 'valid json', ) + xml_markers = ( + 'convert into xml', + 'convert to xml', + 'xml file', + 'download xml', + 'save xml', + 'make an xml', + 'make a xml', + 'create an xml', + 'create a xml', + 'generate xml', + 'generate an xml', + 'populate xml', + 'populate the xml', + 'return xml', + 'valid xml', + 'well-formed xml', + 'output as xml', + 'format as xml', + ) csv_markers = TABLE_EXPORT_REQUEST_MARKERS - if any(marker in normalized_question for marker in json_markers): + if any(marker in normalized_question for marker in json_markers) or re.search( + r'\b(convert|create|make|build|generate|produce|return|respond|format|output|save|export|download)\b[\w\s.,:;\-/]{0,80}\ba?\s*json\b', + normalized_question, + ): return 'json' + if any(marker in normalized_question for marker in xml_markers) or re.search( + r'\b(convert|populate|create|make|build|generate|produce|return|respond|format|output|save|export|download)\b[\w\s.,:;\-/]{0,80}\ba?\s*xml\b', + normalized_question, + ): + return 'xml' if any(marker in normalized_question for marker in csv_markers): return 'csv' return None @@ -3933,9 +4139,16 @@ def question_requests_tabular_generated_output(user_question): 'every row', 'full json', 'full csv', + 'full xml', + 'entire', + 'complete', + 'convert', 'download', 'save', 'export', + 'create', + 'generate', + 'populate', 'one object per', 'one row per', 'each object', @@ -4345,7 +4558,7 @@ def _build_tabular_generated_output_input_row(row, source_file_name=None): def _build_tabular_generated_output_file_name(source_file_name, output_format): timestamp_suffix = datetime.utcnow().strftime('%Y%m%d_%H%M%S') normalized_base_name = _sanitize_tabular_generated_output_base_name(source_file_name) - normalized_extension = 'csv' if output_format == 'csv' else 'json' + normalized_extension = normalize_generated_output_format(output_format) return f"{normalized_base_name}_generated_{timestamp_suffix}.{normalized_extension}" @@ -4569,7 +4782,7 @@ async def _generate_tabular_structured_output_entries( model_context=model_context, ) - normalized_output_format = str(output_format or 'json').strip().lower() or 'json' + normalized_output_format = normalize_generated_output_format(output_format) output_format_label = normalized_output_format.upper() batch_budget = _get_tabular_generated_output_batch_budget(settings) row_batches = _build_tabular_generated_output_row_batches(rows, settings=settings) @@ -4860,8 +5073,14 @@ async def maybe_create_tabular_generated_output( if output_format == 'csv': serialized_output = _build_tabular_generated_output_csv(output_entries) + elif output_format == 'xml': + serialized_output = serialize_generated_xml( + output_entries, + root_name='GeneratedRows', + item_name='Row', + ) else: - serialized_output = json.dumps(output_entries, indent=2, default=str, ensure_ascii=False) + serialized_output = serialize_generated_json(output_entries) generated_file_name = _build_tabular_generated_output_file_name( source_candidate.get('filename'), @@ -12205,15 +12424,25 @@ def execute_document_action_chat_request(data=None, publish_background_event=Non ) document_generated_analysis_artifacts = list(execution_result.get('generated_analysis_artifacts') or []) document_generated_tabular_outputs = list(execution_result.get('generated_tabular_outputs') or []) + document_action_reply_content = execution_result.get('reply', '') assistant_table_generated_output = maybe_create_assistant_table_generated_output( user_question=user_message, - assistant_content=execution_result.get('reply', ''), + assistant_content=document_action_reply_content, conversation_id=conversation_id, existing_outputs=document_generated_analysis_artifacts + document_generated_tabular_outputs, ) if assistant_table_generated_output: document_generated_analysis_artifacts.append(assistant_table_generated_output) document_generated_tabular_outputs.append(assistant_table_generated_output) + assistant_file_generated_output = maybe_create_assistant_file_generated_output( + user_question=user_message, + assistant_content=document_action_reply_content, + conversation_id=conversation_id, + existing_outputs=document_generated_analysis_artifacts + document_generated_tabular_outputs, + ) + if assistant_file_generated_output: + document_generated_analysis_artifacts.append(assistant_file_generated_output) + document_action_reply_content = _build_assistant_file_output_handoff(assistant_file_generated_output) generated_analysis_metadata = _build_generated_analysis_metadata( generated_analysis_artifacts=document_generated_analysis_artifacts, generated_tabular_outputs=document_generated_tabular_outputs, @@ -12232,7 +12461,7 @@ def execute_document_action_chat_request(data=None, publish_background_event=Non 'id': assistant_message_id, 'conversation_id': conversation_id, 'role': 'assistant', - 'content': execution_result.get('reply', ''), + 'content': document_action_reply_content, 'timestamp': assistant_timestamp, 'augmented': False, 'hybrid_citations': hybrid_citations_list, @@ -15863,6 +16092,15 @@ def gpt_error(e): if assistant_table_generated_output: generated_analysis_artifacts_list.append(assistant_table_generated_output) generated_tabular_outputs_list.append(assistant_table_generated_output) + assistant_file_generated_output = maybe_create_assistant_file_generated_output( + user_question=user_message, + assistant_content=ai_message, + conversation_id=conversation_id, + existing_outputs=generated_analysis_artifacts_list + generated_tabular_outputs_list, + ) + if assistant_file_generated_output: + generated_analysis_artifacts_list.append(assistant_file_generated_output) + ai_message = _build_assistant_file_output_handoff(assistant_file_generated_output) generated_analysis_metadata = _build_generated_analysis_metadata( generated_analysis_artifacts=generated_analysis_artifacts_list, generated_tabular_outputs=generated_tabular_outputs_list, @@ -19044,6 +19282,15 @@ def finalize_cancelled_agent_stream_response(): if assistant_table_generated_output: generated_analysis_artifacts_list.append(assistant_table_generated_output) generated_tabular_outputs_list.append(assistant_table_generated_output) + assistant_file_generated_output = maybe_create_assistant_file_generated_output( + user_question=user_message, + assistant_content=accumulated_content, + conversation_id=conversation_id, + existing_outputs=generated_analysis_artifacts_list + generated_tabular_outputs_list, + ) + if assistant_file_generated_output: + generated_analysis_artifacts_list.append(assistant_file_generated_output) + accumulated_content = _build_assistant_file_output_handoff(assistant_file_generated_output) generated_analysis_metadata = _build_generated_analysis_metadata( generated_analysis_artifacts=generated_analysis_artifacts_list, generated_tabular_outputs=generated_tabular_outputs_list, diff --git a/docs/explanation/fixes/GENERATED_JSON_XML_EXPORTS_FIX.md b/docs/explanation/fixes/GENERATED_JSON_XML_EXPORTS_FIX.md new file mode 100644 index 000000000..8c790f0f4 --- /dev/null +++ b/docs/explanation/fixes/GENERATED_JSON_XML_EXPORTS_FIX.md @@ -0,0 +1,57 @@ +# Generated JSON and XML Export Artifacts Fix + +Fixed/implemented in version: **0.250.113** + +## Issue Description + +JSON and XML generation requests could be returned as large inline assistant text or as a Markdown analysis artifact instead of a downloadable generated file. XML template-population workflows were especially affected: Analyze could inspect the selected XML/PDF sources, but the final handoff did not create a completed downloadable XML file. + +## Root Cause + +The generated export framework was primarily wired around CSV/tabular output. JSON support existed in portions of the tabular generated-output path, but general chat and document-analysis artifact creation did not consistently treat JSON as a file artifact. XML was not a first-class generated export format, and XML ingestion contained duplicate processing implementations. + +## Technical Details + +### Files Modified + +- `application/single_app/functions_generated_file_exports.py` +- `application/single_app/route_backend_chats.py` +- `application/single_app/functions_workflow_runner.py` +- `application/single_app/functions_document_analysis.py` +- `application/single_app/functions_documents.py` +- `application/single_app/functions_tabular_generated_exports.py` +- `application/single_app/config.py` +- `functional_tests/test_generated_json_xml_exports.py` + +### Code Changes Summary + +- Added shared generated-file helpers for JSON parsing, XML extraction, XML serialization, and output-format normalization. +- Extended chat export intent detection to recognize natural JSON and XML phrasing such as "convert into JSON" and "populate the XML". +- Added assistant-response JSON/XML artifact capture so valid generated JSON/XML content is saved as a downloadable chat artifact and the persisted assistant message becomes a concise file handoff. +- Extended document-analysis artifact creation to upload `.xml` artifacts when the final analysis reply is valid XML and the user requested XML output. +- Added document-analysis prompt guidance to preserve JSON/XML structure during windowed analysis and to return only valid final JSON/XML during reduction. +- Extended durable tabular generated exports to serialize XML output from checkpointed row batches. +- Consolidated XML document processing through one token-aware implementation and replaced directly touched XML processing `print()` diagnostics with `log_event`. + +## Testing Approach + +Added `functional_tests/test_generated_json_xml_exports.py` to verify: + +- Shared JSON/XML helper parsing and serialization. +- Chat route JSON/XML artifact hooks and no-inline handoff markers. +- Document-analysis JSON/XML intent and artifact wiring. +- XML processing consolidation and token-aware return behavior. + +## Impact + +Users who request JSON or XML file-shaped output now get the same generated artifact/download behavior used by CSV paths where valid generated content is available. XML template population and XML-to-JSON conversion have explicit artifact support instead of relying on inline responses or Markdown fallbacks. + +## Validation + +Run: + +```powershell +python functional_tests\test_generated_json_xml_exports.py +``` + +Before this fix, JSON/XML requests were not consistently recognized as generated artifact workflows and XML output had no first-class artifact path. After this fix, JSON/XML artifact intent is recognized, valid outputs are attached as downloadable files, and XML ingestion uses one consolidated processor. diff --git a/functional_tests/test_generated_json_xml_exports.py b/functional_tests/test_generated_json_xml_exports.py new file mode 100644 index 000000000..bde10af1c --- /dev/null +++ b/functional_tests/test_generated_json_xml_exports.py @@ -0,0 +1,165 @@ +#!/usr/bin/env python3 +# test_generated_json_xml_exports.py +""" +Functional test for generated JSON/XML export artifacts. +Version: 0.250.113 +Implemented in: 0.250.113 + +This test ensures JSON/XML generation requests are recognized as downloadable +artifact workflows, reuse shared serialization helpers, avoid duplicate XML +processing implementations, and preserve no-inline-output handoff behavior. +""" + +import importlib.util +import sys +from pathlib import Path + + +ROOT = Path(__file__).resolve().parents[1] +APP_ROOT = ROOT / "application" / "single_app" +CONFIG_FILE = APP_ROOT / "config.py" +GENERATED_EXPORTS_FILE = APP_ROOT / "functions_generated_file_exports.py" +CHAT_ROUTE_FILE = APP_ROOT / "route_backend_chats.py" +WORKFLOW_RUNNER_FILE = APP_ROOT / "functions_workflow_runner.py" +DOCUMENT_ANALYSIS_FILE = APP_ROOT / "functions_document_analysis.py" +DOCUMENTS_FILE = APP_ROOT / "functions_documents.py" +EXPECTED_VERSION = "0.250.113" + + +def read_text(path): + return path.read_text(encoding="utf-8") + + +def assert_contains(source_text, needle, description): + if needle not in source_text: + raise AssertionError(f"Missing {description}: {needle}") + + +def read_current_version(): + for line in read_text(CONFIG_FILE).splitlines(): + stripped_line = line.strip() + if stripped_line.startswith("VERSION = "): + return stripped_line.split('"')[1] + raise AssertionError("Expected config.py to define VERSION") + + +def load_generated_exports_module(): + spec = importlib.util.spec_from_file_location( + "functions_generated_file_exports", + GENERATED_EXPORTS_FILE, + ) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def test_shared_json_xml_export_helpers(): + print("Testing shared JSON/XML export helpers...") + module = load_generated_exports_module() + + parsed_json = module.normalize_json_artifact_payload( + "Here is the file:\n```json\n{\"name\": \"Example\", \"items\": [1, 2]}\n```" + ) + assert parsed_json == {"name": "Example", "items": [1, 2]} + + xml_payload = module.normalize_xml_artifact_payload( + "Generated XML:\n```xml\nExample\n```" + ) + assert xml_payload == "Example" + + serialized_xml = module.serialize_generated_xml( + [{"name": "A"}, {"name": "B"}], + root_name="GeneratedRows", + item_name="Row", + ) + assert serialized_xml.startswith('') + assert "" in serialized_xml + assert serialized_xml.count("") == 2 + + assert module.normalize_generated_output_format(".xml") == "xml" + assert module.normalize_generated_output_format("json") == "json" + print("Shared helper checks passed") + + +def test_chat_route_json_xml_artifact_hooks(): + print("Testing chat route JSON/XML artifact hooks...") + chat_source = read_text(CHAT_ROUTE_FILE) + + assert_contains(chat_source, "normalize_json_artifact_payload", "JSON artifact extraction import") + assert_contains(chat_source, "normalize_xml_artifact_payload", "XML artifact extraction import") + assert_contains(chat_source, "def maybe_create_assistant_file_generated_output(", "assistant JSON/XML artifact helper") + assert_contains(chat_source, "convert into json", "natural JSON conversion marker") + assert_contains(chat_source, r"\ba?\s*json\b", "natural JSON conversion regex") + assert_contains(chat_source, "populate the xml", "XML template population marker") + assert_contains(chat_source, r"\ba?\s*xml\b", "natural XML conversion regex") + assert_contains(chat_source, "_build_assistant_file_output_handoff", "no-inline assistant handoff builder") + assert chat_source.count("maybe_create_assistant_file_generated_output(") >= 4, ( + "Expected helper definition plus document-action, non-streaming, and streaming save path calls." + ) + assert_contains(chat_source, "serialize_generated_xml(", "XML serialization for generated tabular exports") + assert_contains(chat_source, "root_name='GeneratedRows'", "tabular XML root naming") + print("Chat route checks passed") + + +def test_document_analysis_xml_json_intent_and_artifacts(): + print("Testing document analysis JSON/XML intent and artifact wiring...") + analysis_source = read_text(DOCUMENT_ANALYSIS_FILE) + workflow_source = read_text(WORKFLOW_RUNNER_FILE) + + assert_contains(analysis_source, "def _prompt_requests_json_output(", "document-analysis JSON output intent") + assert_contains(analysis_source, "def _prompt_requests_xml_output(", "document-analysis XML output intent") + assert_contains(analysis_source, "Return only one complete well-formed XML document", "XML-only reduction guidance") + assert_contains(analysis_source, "Return only valid JSON for the final answer", "JSON-only reduction guidance") + assert_contains(analysis_source, "'xml_output_requested': xml_output_requested", "XML intent metadata") + + assert_contains(workflow_source, "def _prompt_explicitly_requests_xml_artifact(", "workflow XML artifact intent") + assert_contains(workflow_source, "xml_payload = normalize_xml_artifact_payload(analysis_reply)", "XML payload extraction") + assert_contains(workflow_source, "_build_document_analysis_artifact_file_name(analysis_result, 'xml')", "XML artifact filename") + assert_contains(workflow_source, "output_format = 'xml' if xml_payload and xml_artifact_requested", "XML artifact output selection") + assert_contains(workflow_source, "serialize_generated_json(json_payload)", "shared JSON serialization") + print("Document analysis checks passed") + + +def test_xml_processing_consolidated(): + print("Testing XML processing consolidation...") + documents_source = read_text(DOCUMENTS_FILE) + + assert documents_source.count("def process_xml(") == 1, "Expected exactly one public process_xml function." + assert_contains(documents_source, "def _process_xml_with_token_usage(", "token-aware XML implementation") + assert_contains(documents_source, "token_usage = save_chunks(**args)", "XML token usage accumulation") + assert_contains(documents_source, "return total_chunks_saved, total_embedding_tokens, embedding_model_name", "XML token-aware return") + assert "print(f\"Skipping empty XML chunk" not in documents_source + assert_contains(documents_source, "[Documents] XML processing failed", "XML log_event error logging") + print("XML processing checks passed") + + +def run_tests(): + current_version = read_current_version() + if current_version != EXPECTED_VERSION: + raise AssertionError(f"Expected config.py version {EXPECTED_VERSION}, got {current_version}") + + tests = [ + test_shared_json_xml_export_helpers, + test_chat_route_json_xml_artifact_hooks, + test_document_analysis_xml_json_intent_and_artifacts, + test_xml_processing_consolidated, + ] + results = [] + for test in tests: + print(f"\nRunning {test.__name__}...") + try: + test() + print("PASS") + results.append(True) + except Exception as exc: + print(f"FAIL: {exc}") + import traceback + traceback.print_exc() + results.append(False) + + print(f"\nResults: {sum(results)}/{len(results)} tests passed") + return all(results) + + +if __name__ == "__main__": + sys.exit(0 if run_tests() else 1) From 7eee379f75eaf87526f2c4c28f49c92d8cfd3c72 Mon Sep 17 00:00:00 2001 From: Paul Lizer Date: Tue, 4 Aug 2026 12:54:10 -0400 Subject: [PATCH 2/3] Update release notes for JSON XML exports Document the generated JSON and XML export artifact work under version 0.250.113. Refs #1071. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- docs/explanation/release_notes.md | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/docs/explanation/release_notes.md b/docs/explanation/release_notes.md index d6a109df5..e52696554 100644 --- a/docs/explanation/release_notes.md +++ b/docs/explanation/release_notes.md @@ -2,6 +2,16 @@ For feature-focused and fix-focused drill-downs by version, see [Features by Version](/explanation/features/) and [Fixes by Version](/explanation/fixes/). +### **(v0.250.113)** + +#### New Features + +* **Generated JSON and XML Export Artifacts** + * JSON and XML generation requests can now save valid generated output as downloadable chat artifacts instead of leaving large file-shaped content in the assistant response. + * Document Analyze and generated export flows now recognize natural JSON/XML conversion and XML template-population phrasing, with XML serialization support added to durable generated exports. + * XML document processing now uses a consolidated token-aware pipeline for more reliable analysis and export workflows. + * (Ref: #1071, `functions_generated_file_exports.py`, generated analysis artifacts, XML document processing) + ### **(v0.250.112)** #### New Features From bdf5c248d312c2726ff085be2ccc4a84df3d14ad Mon Sep 17 00:00:00 2001 From: Paul Lizer Date: Tue, 4 Aug 2026 13:06:08 -0400 Subject: [PATCH 3/3] Harden generated XML export parsing Resolve GitHub Advanced Security feedback by removing the code-fence regex and validating generated XML with defusedxml. Update versioned tests and release notes for the security fix. Refs #1071. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- application/single_app/config.py | 2 +- .../functions_generated_file_exports.py | 29 +++++++++++++------ application/single_app/requirements.txt | 1 + .../fixes/GENERATED_JSON_XML_EXPORTS_FIX.md | 2 +- docs/explanation/release_notes.md | 2 +- .../test_generated_json_xml_exports.py | 23 +++++++++++++-- 6 files changed, 44 insertions(+), 15 deletions(-) diff --git a/application/single_app/config.py b/application/single_app/config.py index 13c09a7e5..e3d2f441f 100644 --- a/application/single_app/config.py +++ b/application/single_app/config.py @@ -95,7 +95,7 @@ EXECUTOR_TYPE = 'thread' EXECUTOR_MAX_WORKERS = 30 SESSION_TYPE = 'filesystem' -VERSION = "0.250.113" +VERSION = "0.250.114" IS_DEVELOPMENT = is_development_env_enabled() SESSION_COOKIE_SAMESITE = os.getenv('SESSION_COOKIE_SAMESITE', 'Lax') diff --git a/application/single_app/functions_generated_file_exports.py b/application/single_app/functions_generated_file_exports.py index 651976630..07f2e98a7 100644 --- a/application/single_app/functions_generated_file_exports.py +++ b/application/single_app/functions_generated_file_exports.py @@ -6,6 +6,9 @@ from typing import Any, Iterable from xml.etree import ElementTree +from defusedxml import ElementTree as DefusedElementTree +from defusedxml.common import DefusedXmlException + SUPPORTED_GENERATED_EXPORT_FORMATS = {'csv', 'json', 'xml'} XML_DECLARATION = '' @@ -30,15 +33,23 @@ def strip_markdown_code_fence(text): if not normalized_text.startswith('```'): return normalized_text - code_fence_match = re.fullmatch( - r'```(?:[a-zA-Z0-9_-]+)?\s*(.*?)\s*```', - normalized_text, - re.DOTALL, - ) - if not code_fence_match: + header_end_index = normalized_text.find('\n') + if header_end_index <= 0: + return normalized_text + + header_suffix = normalized_text[3:header_end_index].strip() + if header_suffix and not all(character.isalnum() or character in {'_', '-'} for character in header_suffix): + return normalized_text + + closing_index = normalized_text.rfind('```') + if closing_index <= header_end_index: + return normalized_text + + trailing_text = normalized_text[closing_index + 3:].strip() + if trailing_text: return normalized_text - return str(code_fence_match.group(1) or '').strip() + return normalized_text[header_end_index + 1:closing_index].strip() def _iter_xml_candidates(text) -> Iterable[str]: @@ -80,8 +91,8 @@ def normalize_xml_artifact_payload(text): continue seen_candidates.add(candidate) try: - ElementTree.fromstring(candidate.encode('utf-8')) - except ElementTree.ParseError: + DefusedElementTree.fromstring(candidate.encode('utf-8')) + except (DefusedXmlException, ElementTree.ParseError): continue return candidate return '' diff --git a/application/single_app/requirements.txt b/application/single_app/requirements.txt index a9d3570c7..76408c815 100644 --- a/application/single_app/requirements.txt +++ b/application/single_app/requirements.txt @@ -12,6 +12,7 @@ docx2txt==0.8 olefile==0.47 Markdown==3.8.1 bleach==6.4.0 +defusedxml==0.7.1 azure-cosmos==4.9.0 msal==1.31.0 Flask-Session==0.8.0 diff --git a/docs/explanation/fixes/GENERATED_JSON_XML_EXPORTS_FIX.md b/docs/explanation/fixes/GENERATED_JSON_XML_EXPORTS_FIX.md index 8c790f0f4..fdaedf35a 100644 --- a/docs/explanation/fixes/GENERATED_JSON_XML_EXPORTS_FIX.md +++ b/docs/explanation/fixes/GENERATED_JSON_XML_EXPORTS_FIX.md @@ -1,6 +1,6 @@ # Generated JSON and XML Export Artifacts Fix -Fixed/implemented in version: **0.250.113** +Fixed/implemented in version: **0.250.114** ## Issue Description diff --git a/docs/explanation/release_notes.md b/docs/explanation/release_notes.md index e52696554..d7337c0c2 100644 --- a/docs/explanation/release_notes.md +++ b/docs/explanation/release_notes.md @@ -2,7 +2,7 @@ For feature-focused and fix-focused drill-downs by version, see [Features by Version](/explanation/features/) and [Fixes by Version](/explanation/fixes/). -### **(v0.250.113)** +### **(v0.250.114)** #### New Features diff --git a/functional_tests/test_generated_json_xml_exports.py b/functional_tests/test_generated_json_xml_exports.py index bde10af1c..9c2d51744 100644 --- a/functional_tests/test_generated_json_xml_exports.py +++ b/functional_tests/test_generated_json_xml_exports.py @@ -2,8 +2,8 @@ # test_generated_json_xml_exports.py """ Functional test for generated JSON/XML export artifacts. -Version: 0.250.113 -Implemented in: 0.250.113 +Version: 0.250.114 +Implemented in: 0.250.114 This test ensures JSON/XML generation requests are recognized as downloadable artifact workflows, reuse shared serialization helpers, avoid duplicate XML @@ -23,7 +23,7 @@ WORKFLOW_RUNNER_FILE = APP_ROOT / "functions_workflow_runner.py" DOCUMENT_ANALYSIS_FILE = APP_ROOT / "functions_document_analysis.py" DOCUMENTS_FILE = APP_ROOT / "functions_documents.py" -EXPECTED_VERSION = "0.250.113" +EXPECTED_VERSION = "0.250.114" def read_text(path): @@ -66,6 +66,10 @@ def test_shared_json_xml_export_helpers(): "Generated XML:\n```xml\nExample\n```" ) assert xml_payload == "Example" + assert module.normalize_xml_artifact_payload( + "]>&xxe;" + ) == "" + assert module.strip_markdown_code_fence("```json\n{\"safe\": true}\n```") == '{"safe": true}' serialized_xml = module.serialize_generated_xml( [{"name": "A"}, {"name": "B"}], @@ -133,6 +137,18 @@ def test_xml_processing_consolidated(): print("XML processing checks passed") +def test_security_review_fixes(): + print("Testing security review fix markers...") + helper_source = read_text(GENERATED_EXPORTS_FILE) + requirements_source = read_text(APP_ROOT / "requirements.txt") + + assert_contains(helper_source, "from defusedxml import ElementTree as DefusedElementTree", "defused XML parser import") + assert_contains(helper_source, "DefusedElementTree.fromstring", "hardened XML parser usage") + assert "re.fullmatch(" not in helper_source, "Generated export helper should not use regex fullmatch for code fences." + assert_contains(requirements_source, "defusedxml==0.7.1", "defusedxml dependency pin") + print("Security review fix checks passed") + + def run_tests(): current_version = read_current_version() if current_version != EXPECTED_VERSION: @@ -143,6 +159,7 @@ def run_tests(): test_chat_route_json_xml_artifact_hooks, test_document_analysis_xml_json_intent_and_artifacts, test_xml_processing_consolidated, + test_security_review_fixes, ] results = [] for test in tests: