Spaces:
Sleeping
Sleeping
| """ | |
| Tests for src/analyzer/utils/text.py | |
| Tests cover: | |
| - clean(): whitespace normalization | |
| - to_number(): flexible number parsing | |
| - extract_numbers(): finding all numbers in text | |
| - safe_truncate_chars(): character truncation with ellipsis | |
| - safe_truncate_tokens(): token-aware truncation | |
| """ | |
| import pytest | |
| from src.analyzer.utils.text import ( | |
| clean, | |
| to_number, | |
| extract_numbers, | |
| safe_truncate_chars, | |
| safe_truncate_tokens, | |
| ) | |
| class TestClean: | |
| """Test clean() function for text normalization.""" | |
| def test_clean_basic(self): | |
| assert clean("hello world") == "hello world" | |
| def test_clean_multiple_spaces(self): | |
| assert clean("hello world") == "hello world" | |
| def test_clean_newlines_and_tabs(self): | |
| assert clean("hello\n\t world") == "hello world" | |
| def test_clean_leading_trailing(self): | |
| assert clean(" hello world ") == "hello world" | |
| def test_clean_none(self): | |
| assert clean(None) == "" | |
| def test_clean_empty_string(self): | |
| assert clean("") == "" | |
| def test_clean_only_whitespace(self): | |
| assert clean(" \n\t ") == "" | |
| def test_clean_numbers(self): | |
| assert clean(123) == "123" | |
| def test_clean_complex_whitespace(self): | |
| result = clean("hello \n\n world\t\tfoo bar") | |
| assert result == "hello world foo bar" | |
| class TestToNumber: | |
| """Test to_number() function for flexible number parsing.""" | |
| def test_to_number_int(self): | |
| assert to_number(42) == 42.0 | |
| def test_to_number_float(self): | |
| assert to_number(3.14) == 3.14 | |
| def test_to_number_string_int(self): | |
| assert to_number("42") == 42.0 | |
| def test_to_number_string_float(self): | |
| assert to_number("3.14") == 3.14 | |
| def test_to_number_with_commas(self): | |
| assert to_number("1,234,567") == 1234567.0 | |
| def test_to_number_with_currency(self): | |
| assert to_number("£1,234.50") == 1234.50 | |
| def test_to_number_negative(self): | |
| assert to_number("-42.5") == -42.5 | |
| def test_to_number_none(self): | |
| assert to_number(None) is None | |
| def test_to_number_empty_string(self): | |
| assert to_number("") is None | |
| def test_to_number_no_numbers(self): | |
| assert to_number("hello") is None | |
| def test_to_number_mixed_text(self): | |
| # Should extract first number from text | |
| assert to_number("Project cost: £25,000.00") == 25000.0 | |
| def test_to_number_zero(self): | |
| assert to_number(0) == 0.0 | |
| assert to_number("0") == 0.0 | |
| class TestExtractNumbers: | |
| """Test extract_numbers() for finding all numbers in text.""" | |
| def test_extract_numbers_single(self): | |
| assert extract_numbers("42") == [42.0] | |
| def test_extract_numbers_multiple(self): | |
| assert extract_numbers("10 20 30") == [10.0, 20.0, 30.0] | |
| def test_extract_numbers_mixed_text(self): | |
| result = extract_numbers("Budget: £1,234 Duration: 12 months") | |
| assert result == [1.0, 234.0, 12.0] | |
| def test_extract_numbers_floats(self): | |
| assert extract_numbers("3.14 2.71 1.41") == [3.14, 2.71, 1.41] | |
| def test_extract_numbers_negative(self): | |
| assert extract_numbers("-10 20 -30") == [-10.0, 20.0, -30.0] | |
| def test_extract_numbers_none(self): | |
| assert extract_numbers(None) == [] | |
| def test_extract_numbers_empty(self): | |
| assert extract_numbers("") == [] | |
| def test_extract_numbers_no_numbers(self): | |
| assert extract_numbers("hello world") == [] | |
| class TestSafeTruncateChars: | |
| """Test safe_truncate_chars() for character-based truncation.""" | |
| def test_truncate_chars_short(self): | |
| assert safe_truncate_chars("hello", 10) == "hello" | |
| def test_truncate_chars_exact(self): | |
| assert safe_truncate_chars("hello", 5) == "hello" | |
| def test_truncate_chars_long(self): | |
| result = safe_truncate_chars("hello world", 8) | |
| assert result == "hello w…" | |
| assert len(result) == 8 | |
| def test_truncate_chars_very_short(self): | |
| result = safe_truncate_chars("hello", 3) | |
| assert result == "he…" | |
| def test_truncate_chars_zero(self): | |
| result = safe_truncate_chars("hello", 0) | |
| assert result == "…" | |
| def test_truncate_chars_none(self): | |
| assert safe_truncate_chars(None, 10) == "" | |
| def test_truncate_chars_empty(self): | |
| assert safe_truncate_chars("", 10) == "" | |
| class TestSafeTruncateTokens: | |
| """Test safe_truncate_tokens() for token-aware truncation.""" | |
| def test_truncate_tokens_short(self): | |
| result = safe_truncate_tokens("hello world", 100) | |
| assert result == "hello world" | |
| def test_truncate_tokens_fallback(self): | |
| # Test fallback mode (when tiktoken not available or errors) | |
| # Should use char-based approximation (~4 chars per token) | |
| text = "a" * 100 | |
| result = safe_truncate_tokens(text, 10) | |
| # Should truncate to ~40 chars (10 tokens * 4 chars) | |
| assert len(result) <= 41 # 40 + ellipsis | |
| def test_truncate_tokens_none(self): | |
| assert safe_truncate_tokens(None, 10) == "" | |
| def test_truncate_tokens_empty(self): | |
| assert safe_truncate_tokens("", 10) == "" | |
| def test_truncate_tokens_exact(self): | |
| # Simple short text should not be truncated | |
| result = safe_truncate_tokens("hi", 10) | |
| assert result == "hi" | |
| class TestEdgeCases: | |
| """Test edge cases and corner scenarios.""" | |
| def test_clean_with_unicode(self): | |
| assert clean("hello 世界") == "hello 世界" | |
| def test_to_number_with_unicode_currency(self): | |
| # Test with various currency symbols | |
| assert to_number("€100") == 100.0 | |
| def test_extract_numbers_scientific_notation(self): | |
| # Should handle basic scientific notation | |
| result = extract_numbers("1e5") | |
| # Depends on implementation - might be [1.0, 5.0] or handle it | |
| assert len(result) >= 1 | |
| def test_truncate_chars_unicode(self): | |
| result = safe_truncate_chars("hello 世界", 7) | |
| assert len(result) == 7 | |
| assert result.endswith("…") | |
| def test_clean_converts_types(self): | |
| # Test that clean handles various types | |
| assert clean([1, 2, 3]) == "[1, 2, 3]" | |
| assert clean({"a": 1}) in ["{'a': 1}", "{'a':1}"] # Dict repr may vary | |
| if __name__ == "__main__": | |
| pytest.main([__file__, "-v"]) | |