import { describe, it, expect } from 'vitest' import { extractXMLFromString, wrapXmlBlocksInContent } from './xmlUtil' import { longTextWithAddNotes, longTextWithReadMusic, multipleXmlBlocks, nestedXmlContent, xmlWithAttributes, malformedXml, noXmlContent, emptyAndWhitespaceXml } from '../test/fixtures/xml-samples' describe('xmlUtil', () => { describe('extractXMLFromString', () => { it('should extract simple XML blocks', () => { const input = `Here is some text with content and more text.` const result = extractXMLFromString(input) expect(result).toHaveLength(1) expect(result[0]).toBe('content') }) it('should extract multiple XML blocks', () => { const input = `content1 some text content2` const result = extractXMLFromString(input) expect(result).toHaveLength(2) expect(result[0]).toBe('content1') expect(result[1]).toBe('content2') }) it('should extract XML blocks with nested elements', () => { const input = `nested content` const result = extractXMLFromString(input) expect(result).toHaveLength(1) expect(result[0]).toBe('nested content') }) it('should extract XML blocks with attributes', () => { const input = `content` const result = extractXMLFromString(input) expect(result).toHaveLength(1) expect(result[0]).toBe('content') }) it('should handle multiline XML blocks', () => { const input = ` content1 content2 ` const result = extractXMLFromString(input) expect(result).toHaveLength(1) expect(result[0]).toContain('') expect(result[0]).toContain('content1') expect(result[0]).toContain('content2') expect(result[0]).toContain('') }) it('should extract XML from complex nested content fixture', () => { const result = extractXMLFromString(nestedXmlContent) expect(result).toHaveLength(1) expect(result[0]).toContain('') expect(result[0]).toContain('') expect(result[0]).toContain('') expect(result[0]).toContain('Test Content') expect(result[0]).toContain('') }) it('should extract multiple XML blocks from fixture', () => { const result = extractXMLFromString(multipleXmlBlocks) expect(result).toHaveLength(3) expect(result[0]).toContain('') expect(result[0]).toContain('') expect(result[1]).toContain('') expect(result[1]).toContain('') expect(result[2]).toContain('') expect(result[2]).toContain('') }) it('should extract XML with attributes from fixture', () => { const result = extractXMLFromString(xmlWithAttributes) expect(result).toHaveLength(1) expect(result[0]).toContain('region_id="main"') expect(result[0]).toContain('track="melody"') expect(result[0]).toContain('id="1"') expect(result[0]).toContain('velocity="127"') }) it('should handle the long text with add_notes fixture (Case 1)', () => { const result = extractXMLFromString(longTextWithAddNotes) expect(result).toHaveLength(2) // Contains thinking tag and add_notes block // Find the add_notes block const addNotesBlock = result.find(block => block.includes('')) expect(addNotesBlock).toBeDefined() expect(addNotesBlock).toContain('') expect(addNotesBlock).toContain('') expect(addNotesBlock).toContain('C4') expect(addNotesBlock).toContain('0') expect(addNotesBlock).toContain('4') expect(addNotesBlock).toContain('') // Should contain all 12 notes const noteMatches = addNotesBlock!.match(//g) expect(noteMatches).toHaveLength(12) }) it('should handle the long text with read_music fixture (Case 2)', () => { const result = extractXMLFromString(longTextWithReadMusic) expect(result).toHaveLength(2) // First XML block (inside thinking tag) expect(result[0]).toContain('') expect(result[0]).toContain('0') expect(result[0]).toContain('32') expect(result[0]).toContain('') // Second XML block (at the end) expect(result[1]).toContain('') expect(result[1]).toContain('0') expect(result[1]).toContain('32') expect(result[1]).toContain('') }) it('should handle malformed XML gracefully', () => { const result = extractXMLFromString(malformedXml) // Should extract valid XML blocks (ignores malformed ones) expect(result.length).toBeGreaterThanOrEqual(1) // Find the definitely valid block const validBlock = result.find(block => block.includes('')) expect(validBlock).toBeDefined() expect(validBlock).toContain('This is valid') }) it('should return empty array for content with no XML', () => { const result = extractXMLFromString(noXmlContent) expect(result).toHaveLength(0) expect(result).toEqual([]) }) it('should handle empty and whitespace XML', () => { const result = extractXMLFromString(emptyAndWhitespaceXml) expect(result).toHaveLength(3) expect(result[0]).toBe('') expect(result[1]).toContain('') expect(result[1]).toContain('') expect(result[2]).toContain('') expect(result[2]).toContain('Some text with spaces') expect(result[2]).toContain('') }) it('should handle XML with underscores and hyphens in tag names', () => { const input = `content and content` const result = extractXMLFromString(input) expect(result).toHaveLength(2) expect(result[0]).toBe('content') expect(result[1]).toBe('content') }) it('should handle self-closing tags (not currently supported)', () => { const input = ` and content` const result = extractXMLFromString(input) // Current implementation doesn't support self-closing tags expect(result).toHaveLength(1) expect(result[0]).toBe('content') }) it('should trim whitespace around extracted XML', () => { const input = ` content ` const result = extractXMLFromString(input) expect(result).toHaveLength(1) expect(result[0]).toBe('content') }) }) describe('wrapXmlBlocksInContent', () => { it('should wrap single XML block in fenced code block', () => { const input = `Here is content in text.` const result = wrapXmlBlocksInContent(input) expect(result).toBe('Here is ```xml\ncontent\n``` in text.') }) it('should wrap multiple XML blocks', () => { const input = `content1 text content2` const result = wrapXmlBlocksInContent(input) expect(result).toContain('```xml\ncontent1\n```') expect(result).toContain('```xml\ncontent2\n```') }) it('should return original content when no XML blocks present', () => { const input = noXmlContent const result = wrapXmlBlocksInContent(input) expect(result).toBe(input) }) it('should handle empty input', () => { expect(wrapXmlBlocksInContent('')).toBe('') expect(wrapXmlBlocksInContent(null as any)).toBeNull() expect(wrapXmlBlocksInContent(undefined as any)).toBeUndefined() }) it('should wrap XML blocks from multipleXmlBlocks fixture', () => { const result = wrapXmlBlocksInContent(multipleXmlBlocks) expect(result).toContain('```xml\n') expect(result).toContain('\n```') expect(result).toContain('```xml\n') expect(result).toContain('\n```') expect(result).toContain('```xml\n') expect(result).toContain('\n```') // Should preserve the surrounding text expect(result).toContain('Here\'s how to add multiple musical elements:') expect(result).toContain('First, let\'s add some notes:') expect(result).toContain('Then we can read the current music:') }) it('should wrap the long add_notes XML block (Case 1)', () => { const result = wrapXmlBlocksInContent(longTextWithAddNotes) expect(result).toContain('```xml\n') expect(result).toContain('\n```') // Should preserve the thinking content and other text expect(result).toContain('') expect(result).toContain('Perfect! I can see this is a beautiful') expect(result).toContain('Let me start by adding the harmony') }) it('should wrap the long read_music XML blocks (Case 2)', () => { const result = wrapXmlBlocksInContent(longTextWithReadMusic) // Should contain two wrapped XML blocks const fencedBlocks = result.match(/```xml\n[\s\S]*?<\/read_music>\n```/g) expect(fencedBlocks).toHaveLength(2) // Should preserve the thinking content and other text expect(result).toContain('') expect(result).toContain('I\'ll help you create a pad harmony track') expect(result).toContain('Let me first read the existing music') }) it('should handle nested XML correctly', () => { const result = wrapXmlBlocksInContent(nestedXmlContent) expect(result).toContain('```xml\n') expect(result).toContain('') expect(result).toContain('') expect(result).toContain('Test Content') expect(result).toContain('\n```') }) it('should preserve XML block integrity when wrapping', () => { const input = `Text before\n\n value\n content\n\nText after` const result = wrapXmlBlocksInContent(input) expect(result).toBe(`Text before\n\`\`\`xml\n\n value\n content\n\n\`\`\`\nText after`) }) it('should handle duplicate XML blocks correctly', () => { const input = `content and then content again` const result = wrapXmlBlocksInContent(input) // Both instances should be wrapped const wrappedBlocks = result.match(/```xml\ncontent<\/same>\n```/g) expect(wrappedBlocks).toHaveLength(2) }) }) describe('edge cases and error handling', () => { it('should handle very large XML blocks', () => { const largeContent = 'x'.repeat(10000) const input = `${largeContent}` const extracted = extractXMLFromString(input) expect(extracted).toHaveLength(1) expect(extracted[0]).toContain(largeContent) const wrapped = wrapXmlBlocksInContent(input) expect(wrapped).toContain('```xml\n') expect(wrapped).toContain('\n```') }) it('should handle XML with special characters', () => { const input = `Content with < > & " '` const extracted = extractXMLFromString(input) expect(extracted).toHaveLength(1) expect(extracted[0]).toContain('< > & " '') const wrapped = wrapXmlBlocksInContent(input) expect(wrapped).toContain('```xml\nContent with < > & " '\n```') }) it('should handle XML with CDATA sections', () => { const input = ` chars]]>` const extracted = extractXMLFromString(input) expect(extracted).toHaveLength(1) expect(extracted[0]).toContain('') }) it('should handle mixed content with partial XML-like text', () => { const input = `This < is not XML and neither > is this content is valid` const extracted = extractXMLFromString(input) expect(extracted).toHaveLength(1) expect(extracted[0]).toBe('content') }) }) })