annator-command-center / tests /property_tests /analytics /test_analytics_invariants.py
techprotrade's picture
Deploy ATOM FastAPI command center runtime (part 8)
aef804e verified
Raw
History Blame Contribute Delete
31.7 kB
"""
Property-Based Tests for Analytics and Reporting Invariants
Tests critical analytics and reporting business logic:
- Metrics collection and aggregation
- Report generation
- Data filtering and segmentation
- Time-based analytics
- Performance analytics
- User behavior analytics
- Analytics export
- Analytics retention
"""
import pytest
from datetime import datetime, timedelta
from hypothesis import given, settings, assume
from hypothesis import strategies as st
from typing import Dict, List, Set, Optional
import uuid
class TestMetricsCollectionInvariants:
"""Tests for metrics collection invariants"""
@given(
metric_name=st.text(min_size=1, max_size=100, alphabet='abcdefghijklmnopqrstuvwxyz_'),
metric_value=st.one_of(
st.integers(min_value=-1000000, max_value=1000000),
st.floats(min_value=-1000000.0, max_value=1000000.0, allow_nan=False, allow_infinity=False)
),
timestamp=st.datetimes(min_value=datetime(2020, 1, 1), max_value=datetime(2030, 1, 1)),
tags=st.dictionaries(
keys=st.text(min_size=1, max_size=50),
values=st.text(min_size=1, max_size=50),
min_size=0,
max_size=10
)
)
@settings(max_examples=50)
def test_metric_collection_creates_valid_metric(self, metric_name, metric_value, timestamp, tags):
"""Test that metric collection creates a valid metric"""
# Collect metric
metric = {
'id': str(uuid.uuid4()),
'name': metric_name,
'value': metric_value,
'timestamp': timestamp,
'tags': tags
}
# Verify metric
assert metric['id'] is not None, "Metric ID must be set"
assert len(metric['name']) >= 1, "Metric name must not be empty"
assert metric['timestamp'] is not None, "Timestamp must be set"
@given(
values=st.lists(st.floats(min_value=0.0, max_value=1000000.0), min_size=0, max_size=1000)
)
@settings(max_examples=50)
def test_metric_aggregation(self, values):
"""Test that metrics are aggregated correctly"""
if len(values) == 0:
assert True # No data
else:
# Calculate sum, avg, min, max, count
total = sum(values)
average = sum(values) / len(values)
minimum = min(values)
maximum = max(values)
count = len(values)
# Verify aggregations
assert total >= 0, "Non-negative sum"
# Use epsilon for floating-point comparison
epsilon = 1e-9
assert minimum - epsilon <= average <= maximum + epsilon, \
f"Average {average} should be between min {minimum} and max {maximum}"
assert count >= 1, "Count must be >= 1"
@given(
metric_name=st.text(min_size=1, max_size=100, alphabet='abcdefghijklmnopqrstuvwxyz_'),
time_window_seconds=st.integers(min_value=60, max_value=86400), # 1 minute to 24 hours
timestamp=st.datetimes(min_value=datetime(2020, 1, 1), max_value=datetime(2030, 1, 1))
)
@settings(max_examples=50)
def test_time_window_aggregation(self, metric_name, time_window_seconds, timestamp):
"""Test that time window aggregation is correct"""
# Define time window
window_start = timestamp - timedelta(seconds=time_window_seconds)
window_end = timestamp
# Verify window
assert window_end >= window_start, "Window end must be after window start"
assert (window_end - window_start).total_seconds() == time_window_seconds, "Window duration must match"
@given(
metrics=st.lists(
st.fixed_dictionaries({
'name': st.text(min_size=1, max_size=50),
'value': st.integers(min_value=0, max_value=1000),
'timestamp': st.datetimes(min_value=datetime(2020, 1, 1), max_value=datetime(2030, 1, 1))
}),
min_size=0,
max_size=100
),
group_by=st.lists(
st.text(min_size=1, max_size=50, alphabet='abcdefghijklmnopqrstuvwxyz_'),
min_size=0,
max_size=5,
unique=True
)
)
@settings(max_examples=50)
def test_metric_grouping(self, metrics, group_by):
"""Test that metrics can be grouped by tags"""
# Group metrics by tags
grouped = {}
for metric in metrics:
# Create group key from tags
group_key = tuple(metric.get(tag, '') for tag in group_by)
if group_key not in grouped:
grouped[group_key] = []
grouped[group_key].append(metric)
# Verify grouping
for group_key, group_metrics in grouped.items():
assert len(group_key) == len(group_by), "Group key must match group_by fields"
assert len(group_metrics) >= 1, "Group must have at least one metric"
class TestReportGenerationInvariants:
"""Tests for report generation invariants"""
@given(
report_id=st.uuids(),
report_type=st.sampled_from([
'usage',
'performance',
'financial',
'user_activity',
'errors',
'custom'
]),
user_id=st.uuids(),
created_at=st.datetimes(min_value=datetime(2020, 1, 1), max_value=datetime(2030, 1, 1))
)
@settings(max_examples=50)
def test_report_generation_creates_valid_report(self, report_id, report_type, user_id, created_at):
"""Test that report generation creates a valid report"""
# Generate report
report = {
'id': str(report_id),
'type': report_type,
'user_id': str(user_id),
'created_at': created_at,
'status': 'GENERATING'
}
# Verify report
assert report['id'] is not None, "Report ID must be set"
assert report['type'] in ['usage', 'performance', 'financial', 'user_activity', 'errors', 'custom'], "Valid report type"
assert report['user_id'] is not None, "User ID must be set"
assert report['created_at'] is not None, "Created at must be set"
assert report['status'] in ['GENERATING', 'COMPLETED', 'FAILED'], "Valid status"
@given(
start_date=st.datetimes(min_value=datetime(2020, 1, 1), max_value=datetime(2025, 1, 1)),
end_date=st.datetimes(min_value=datetime(2025, 1, 1), max_value=datetime(2030, 1, 1))
)
@settings(max_examples=50)
def test_report_date_range_validation(self, start_date, end_date):
"""Test that report date range is valid"""
assume(end_date >= start_date)
# Create date range
date_range = {
'start': start_date,
'end': end_date
}
# Verify date range
assert date_range['end'] >= date_range['start'], "End date must be after start date"
@given(
report_format=st.sampled_from(['pdf', 'csv', 'json', 'html', 'xlsx']),
data_row_count=st.integers(min_value=0, max_value=100000),
max_rows=st.integers(min_value=1000, max_value=100000)
)
@settings(max_examples=50)
def test_report_format_limits(self, report_format, data_row_count, max_rows):
"""Test that report format enforces row limits"""
# Check if within limits
within_limits = data_row_count <= max_rows
# Verify limits
assert data_row_count >= 0, "Row count must be non-negative"
if data_row_count <= max_rows:
assert within_limits is True, "Within row limit"
else:
assert within_limits is False, "Exceeds row limit"
@given(
report_id=st.uuids(),
status_and_size=st.one_of(
# COMPLETED status requires file_size_bytes > 0
st.tuples(
st.sampled_from(['COMPLETED']),
st.integers(min_value=1, max_value=100_000_000)
),
# GENERATING and FAILED can have file_size_bytes >= 0
st.tuples(
st.sampled_from(['GENERATING', 'FAILED']),
st.integers(min_value=0, max_value=100_000_000)
)
),
generated_at=st.datetimes(min_value=datetime(2020, 1, 1), max_value=datetime(2030, 1, 1))
)
@settings(max_examples=50)
def test_report_completion(self, report_id, status_and_size, generated_at):
"""Test that report completion is recorded correctly"""
status, file_size_bytes = status_and_size
# Complete report
report = {
'id': str(report_id),
'status': status,
'file_size_bytes': file_size_bytes,
'generated_at': generated_at
}
# Verify completion
assert report['status'] in ['GENERATING', 'COMPLETED', 'FAILED'], "Valid status"
assert report['file_size_bytes'] >= 0, "File size must be non-negative"
if report['status'] == 'COMPLETED':
assert report['generated_at'] is not None, "Generated at must be set"
assert report['file_size_bytes'] > 0, "Completed report must have file size"
class TestDataFilteringInvariants:
"""Tests for data filtering and segmentation invariants"""
@given(
data_points=st.lists(
st.fixed_dictionaries({
'user_id': st.uuids(),
'action': st.sampled_from(['view', 'click', 'purchase', 'signup']),
'timestamp': st.datetimes(min_value=datetime(2020, 1, 1), max_value=datetime(2030, 1, 1)),
'value': st.integers(min_value=0, max_value=10000)
}),
min_size=0,
max_size=100
),
filter_action=st.sampled_from(['view', 'click', 'purchase', 'signup', None])
)
@settings(max_examples=50)
def test_data_filtering(self, data_points, filter_action):
"""Test that data filtering works correctly"""
# Filter data points by action
if filter_action is not None:
filtered = [dp for dp in data_points if dp['action'] == filter_action]
else:
filtered = data_points
# Verify filtering
assert len(filtered) >= 0, "Filtered count must be non-negative"
assert len(filtered) <= len(data_points), "Filtered count must be <= total count"
if filter_action is not None:
for dp in filtered:
assert dp['action'] == filter_action, f"Filtered data must match action {filter_action}"
@given(
user_id=st.uuids(),
event_count=st.integers(min_value=0, max_value=1000),
segment_threshold=st.integers(min_value=1, max_value=100)
)
@settings(max_examples=50)
def test_user_segmentation(self, user_id, event_count, segment_threshold):
"""Test that user segmentation is correct"""
# Segment users by activity
if event_count >= segment_threshold:
segment = 'high_value'
elif event_count >= segment_threshold // 2:
segment = 'medium_value'
else:
segment = 'low_value'
# Verify segmentation
assert segment in ['high_value', 'medium_value', 'low_value'], "Valid segment"
if event_count >= segment_threshold:
assert segment == 'high_value', "High value users meet threshold"
elif event_count >= segment_threshold // 2:
assert segment == 'medium_value', "Medium value users meet half threshold"
@given(
metrics=st.lists(
st.integers(min_value=0, max_value=10000),
min_size=0,
max_size=100
),
percentile=st.integers(min_value=0, max_value=100)
)
@settings(max_examples=50)
def test_percentile_calculation(self, metrics, percentile):
"""Test that percentile calculation is correct"""
if len(metrics) == 0:
assert True # No data
else:
# Calculate percentile
sorted_metrics = sorted(metrics)
index = int(len(sorted_metrics) * percentile / 100)
index = min(index, len(sorted_metrics) - 1)
percentile_value = sorted_metrics[index]
# Verify percentile
assert 0 <= percentile <= 100, "Percentile must be between 0 and 100"
assert 0 <= percentile_value <= max(metrics), "Percentile value must be within range"
@given(
data_points=st.lists(
st.fixed_dictionaries({
'category': st.sampled_from(['A', 'B', 'C', 'D']),
'value': st.integers(min_value=0, max_value=100)
}),
min_size=0,
max_size=50
)
)
@settings(max_examples=50)
def test_categorical_aggregation(self, data_points):
"""Test that categorical aggregation is correct"""
# Aggregate by category
aggregated = {}
for dp in data_points:
category = dp['category']
if category not in aggregated:
aggregated[category] = {'count': 0, 'total': 0}
aggregated[category]['count'] += 1
aggregated[category]['total'] += dp['value']
# Verify aggregation
for category, stats in aggregated.items():
assert stats['count'] >= 1, f"Category {category} must have at least one data point"
assert stats['total'] >= 0, f"Category {category} total must be non-negative"
class TestTimeBasedAnalyticsInvariants:
"""Tests for time-based analytics invariants"""
@given(
events=st.lists(
st.fixed_dictionaries({
'timestamp': st.datetimes(min_value=datetime(2020, 1, 1), max_value=datetime(2030, 1, 1)),
'value': st.integers(min_value=0, max_value=1000)
}),
min_size=0,
max_size=100
),
interval_minutes=st.integers(min_value=1, max_value=10080) # 1 minute to 1 week
)
@settings(max_examples=50)
def test_time_series_aggregation(self, events, interval_minutes):
"""Test that time series aggregation is correct"""
if len(events) == 0:
assert True # No data
else:
# Aggregate by time interval
interval = timedelta(minutes=interval_minutes)
aggregated = {}
for event in events:
# Round timestamp to interval
timestamp = event['timestamp']
interval_start = timestamp.replace(second=0, microsecond=0)
interval_start = interval_start.replace(minute=(interval_start.minute // interval_minutes) * interval_minutes)
key = interval_start
if key not in aggregated:
aggregated[key] = {'count': 0, 'total': 0}
aggregated[key]['count'] += 1
aggregated[key]['total'] += event['value']
# Verify aggregation
for key, stats in aggregated.items():
assert stats['count'] >= 1, "Interval must have at least one event"
assert stats['total'] >= 0, "Interval total must be non-negative"
@given(
current_value=st.integers(min_value=0, max_value=1000),
previous_value=st.integers(min_value=0, max_value=1000)
)
@settings(max_examples=50)
def test_period_over_period_comparison(self, current_value, previous_value):
"""Test that period-over-period comparison is correct"""
# Calculate growth rate
if previous_value == 0:
growth_rate = None if current_value == 0 else float('inf')
else:
growth_rate = (current_value - previous_value) / previous_value
# Verify calculation
if previous_value > 0:
assert -1.0 <= growth_rate <= float('inf'), "Growth rate must be >= -100%"
@given(
values=st.lists(st.integers(min_value=0, max_value=1000), min_size=0, max_size=100),
window_size=st.integers(min_value=1, max_value=20)
)
@settings(max_examples=50)
def test_moving_average_calculation(self, values, window_size):
"""Test that moving average calculation is correct"""
if len(values) == 0:
assert True # No data
else:
# Calculate moving average
moving_averages = []
for i in range(len(values)):
window = values[max(0, i - window_size + 1):i + 1]
avg = sum(window) / len(window)
moving_averages.append(avg)
# Verify calculation
assert len(moving_averages) == len(values), "Moving averages count must match values count"
for avg in moving_averages:
assert 0 <= avg <= max(values), "Moving average must be within value range"
@given(
timestamp=st.datetimes(min_value=datetime(2020, 1, 1), max_value=datetime(2030, 1, 1)),
period_start=st.datetimes(min_value=datetime(2020, 1, 1), max_value=datetime(2025, 1, 1)),
period_end=st.datetimes(min_value=datetime(2025, 1, 1), max_value=datetime(2030, 1, 1))
)
@settings(max_examples=50)
def test_time_period_inclusion(self, timestamp, period_start, period_end):
"""Test that time period inclusion is checked correctly"""
assume(period_end >= period_start)
# Check if timestamp is in period
in_period = period_start <= timestamp <= period_end
# Verify inclusion
if period_start <= timestamp <= period_end:
assert in_period is True, "Timestamp is in period"
else:
assert in_period is False, "Timestamp is not in period"
class TestPerformanceAnalyticsInvariants:
"""Tests for performance analytics invariants"""
@given(
response_times_ms=st.lists(st.integers(min_value=0, max_value=60000), min_size=0, max_size=1000)
)
@settings(max_examples=50)
def test_response_time_percentiles(self, response_times_ms):
"""Test that response time percentiles are calculated correctly"""
if len(response_times_ms) == 0:
assert True # No data
else:
# Calculate percentiles
sorted_times = sorted(response_times_ms)
p50 = sorted_times[int(len(sorted_times) * 0.5)]
p95 = sorted_times[int(len(sorted_times) * 0.95)]
p99 = sorted_times[int(len(sorted_times) * 0.99)]
# Verify percentiles
assert p50 <= p95 <= p99 or len(sorted_times) < 100, "Percentiles must be ordered"
assert p50 >= 0, "P50 must be non-negative"
assert p99 <= max(response_times_ms), "P99 must be within max"
@given(
total_requests=st.integers(min_value=0, max_value=100000),
successful_requests=st.integers(min_value=0, max_value=100000)
)
@settings(max_examples=50)
def test_success_rate_calculation(self, total_requests, successful_requests):
"""Test that success rate is calculated correctly"""
# Ensure successful_requests <= total_requests
assume(successful_requests <= total_requests)
# Calculate success rate
if total_requests == 0:
success_rate = 1.0 # No failures if no requests
else:
success_rate = successful_requests / total_requests
# Verify calculation
assert 0.0 <= success_rate <= 1.0, "Success rate must be between 0 and 1"
@given(
uptime_percentage=st.floats(min_value=0.0, max_value=100.0, allow_nan=False, allow_infinity=False)
)
@settings(max_examples=50)
def test_uptime_percentage_validation(self, uptime_percentage):
"""Test that uptime percentage is valid"""
# Validate uptime percentage
assert 0.0 <= uptime_percentage <= 100.0, "Uptime must be between 0 and 100"
# Determine SLA compliance
sla_compliant = uptime_percentage >= 99.9
# Verify compliance
if uptime_percentage >= 99.9:
assert sla_compliant is True, "Meets SLA"
else:
assert sla_compliant is False, "Does not meet SLA"
@given(
execution_times_ms=st.lists(st.integers(min_value=0, max_value=60000), min_size=0, max_size=100)
)
@settings(max_examples=50)
def test_throughput_calculation(self, execution_times_ms):
"""Test that throughput is calculated correctly"""
if len(execution_times_ms) == 0:
assert True # No data
else:
# Calculate throughput (requests per second)
total_time_seconds = sum(execution_times_ms) / 1000
if total_time_seconds > 0:
throughput = len(execution_times_ms) / total_time_seconds
else:
throughput = 0
# Verify calculation
assert throughput >= 0, "Throughput must be non-negative"
class TestUserBehaviorAnalyticsInvariants:
"""Tests for user behavior analytics invariants"""
@given(
user_sessions=st.lists(
st.fixed_dictionaries({
'user_id': st.uuids(),
'session_start': st.datetimes(min_value=datetime(2020, 1, 1), max_value=datetime(2030, 1, 1)),
'session_end': st.datetimes(min_value=datetime(2020, 1, 1), max_value=datetime(2030, 1, 1)),
'actions_count': st.integers(min_value=0, max_value=1000)
}),
min_size=0,
max_size=50
)
)
@settings(max_examples=50)
def test_session_duration_tracking(self, user_sessions):
"""Test that session duration is tracked correctly"""
# Calculate session durations
durations = []
for session in user_sessions:
if session['session_end'] >= session['session_start']:
duration = (session['session_end'] - session['session_start']).total_seconds()
durations.append(duration)
# Verify tracking
for duration in durations:
assert duration >= 0, "Duration must be non-negative"
@given(
page_views=st.integers(min_value=0, max_value=10000),
unique_visitors=st.integers(min_value=0, max_value=1000)
)
@settings(max_examples=50)
def test_engagement_metrics(self, page_views, unique_visitors):
"""Test that engagement metrics are calculated correctly"""
# Calculate pages per visitor
if unique_visitors > 0:
pages_per_visitor = page_views / unique_visitors
else:
pages_per_visitor = 0
# Verify metrics
assert page_views >= 0, "Page views must be non-negative"
assert unique_visitors >= 0, "Unique visitors must be non-negative"
assert pages_per_visitor >= 0, "Pages per visitor must be non-negative"
@given(
events=st.lists(
st.fixed_dictionaries({
'user_id': st.uuids(),
'action': st.sampled_from(['signup', 'login', 'purchase', 'cancel']),
'timestamp': st.datetimes(min_value=datetime(2020, 1, 1), max_value=datetime(2030, 1, 1))
}),
min_size=0,
max_size=100
),
action_type=st.sampled_from(['signup', 'login', 'purchase', 'cancel'])
)
@settings(max_examples=50)
def test_funnel_analysis(self, events, action_type):
"""Test that funnel analysis is correct"""
# Filter events by action type
filtered_events = [e for e in events if e['action'] == action_type]
# Count unique users
unique_users = len(set(e['user_id'] for e in filtered_events))
# Verify analysis
assert unique_users >= 0, "Unique users must be non-negative"
assert unique_users <= len(filtered_events), "Unique users must be <= total events"
@given(
user1_events=st.integers(min_value=0, max_value=1000),
user2_events=st.integers(min_value=0, max_value=1000)
)
@settings(max_examples=50)
def test_user_cohort_analysis(self, user1_events, user2_events):
"""Test that cohort analysis groups users correctly"""
# Group users by activity level
if user1_events >= 100:
cohort1 = 'high_activity'
elif user1_events >= 10:
cohort1 = 'medium_activity'
else:
cohort1 = 'low_activity'
if user2_events >= 100:
cohort2 = 'high_activity'
elif user2_events >= 10:
cohort2 = 'medium_activity'
else:
cohort2 = 'low_activity'
# Verify cohorts
assert cohort1 in ['high_activity', 'medium_activity', 'low_activity'], "Valid cohort"
assert cohort2 in ['high_activity', 'medium_activity', 'low_activity'], "Valid cohort"
class TestAnalyticsExportInvariants:
"""Tests for analytics export invariants"""
@given(
export_id=st.uuids(),
export_format=st.sampled_from(['csv', 'json', 'xlsx', 'pdf']),
data_point_count=st.integers(min_value=0, max_value=100000),
created_at=st.datetimes(min_value=datetime(2020, 1, 1), max_value=datetime(2030, 1, 1))
)
@settings(max_examples=50)
def test_export_creation(self, export_id, export_format, data_point_count, created_at):
"""Test that export creation creates a valid export"""
# Create export
export = {
'id': str(export_id),
'format': export_format,
'data_point_count': data_point_count,
'created_at': created_at,
'status': 'PENDING'
}
# Verify export
assert export['id'] is not None, "Export ID must be set"
assert export['format'] in ['csv', 'json', 'xlsx', 'pdf'], "Valid export format"
assert export['data_point_count'] >= 0, "Data point count must be non-negative"
assert export['status'] in ['PENDING', 'COMPLETED', 'FAILED'], "Valid status"
@given(
data_point_count=st.integers(min_value=0, max_value=100000),
max_export_size_mb=st.integers(min_value=1, max_value=1000)
)
@settings(max_examples=50)
def test_export_size_limit_enforcement(self, data_point_count, max_export_size_mb):
"""Test that export size limits are enforced"""
# Estimate export size (rough estimate: 100 bytes per data point)
estimated_size_bytes = data_point_count * 100
estimated_size_mb = estimated_size_bytes / (1024 * 1024)
# Check if within limits
within_limits = estimated_size_mb <= max_export_size_mb
# Verify enforcement
if estimated_size_mb <= max_export_size_mb:
assert within_limits is True, "Within size limit"
else:
assert within_limits is False, "Exceeds size limit"
@given(
export_id=st.uuids(),
file_path=st.text(min_size=10, max_size=500),
file_size_bytes=st.integers(min_value=0, max_value=1_000_000_000), # 0 to 1 GB
completed_at=st.datetimes(min_value=datetime(2020, 1, 1), max_value=datetime(2030, 1, 1))
)
@settings(max_examples=50)
def test_export_completion(self, export_id, file_path, file_size_bytes, completed_at):
"""Test that export completion is recorded correctly"""
# Complete export
export = {
'id': str(export_id),
'file_path': file_path,
'file_size_bytes': file_size_bytes,
'completed_at': completed_at,
'status': 'COMPLETED'
}
# Verify completion
assert export['file_path'] is not None, "File path must be set"
assert export['file_size_bytes'] >= 0, "File size must be non-negative"
assert export['completed_at'] is not None, "Completed at must be set"
assert export['status'] == 'COMPLETED', "Status must be COMPLETED"
@given(
export_id=st.uuids(),
created_at=st.datetimes(min_value=datetime(2020, 1, 1), max_value=datetime(2025, 1, 1)),
retention_days=st.integers(min_value=1, max_value=365),
current_time=st.datetimes(min_value=datetime(2025, 1, 1), max_value=datetime(2030, 1, 1))
)
@settings(max_examples=50)
def test_export_retention_policy(self, export_id, created_at, retention_days, current_time):
"""Test that export retention policy is enforced"""
# Calculate export age
export_age_days = (current_time - created_at).days
# Check if should delete
should_delete = export_age_days > retention_days
# Verify retention policy
if export_age_days > retention_days:
assert should_delete is True, "Export should be deleted"
else:
assert should_delete is False, "Export should be kept"
class TestAnalyticsRetentionInvariants:
"""Tests for analytics data retention invariants"""
@given(
data_point_id=st.uuids(),
created_at=st.datetimes(min_value=datetime(2020, 1, 1), max_value=datetime(2025, 1, 1)),
retention_days=st.integers(min_value=1, max_value=3650), # 1 day to 10 years
current_time=st.datetimes(min_value=datetime(2025, 1, 1), max_value=datetime(2030, 1, 1))
)
@settings(max_examples=50)
def test_data_retention_policy(self, data_point_id, created_at, retention_days, current_time):
"""Test that data retention policy is enforced"""
# Calculate data age
data_age_days = (current_time - created_at).days
# Check if should delete
should_delete = data_age_days > retention_days
# Verify retention policy
if data_age_days > retention_days:
assert should_delete is True, "Data should be deleted"
else:
assert should_delete is False, "Data should be kept"
@given(
data_type=st.sampled_from(['raw_events', 'aggregated_metrics', 'reports', 'exports'])
)
@settings(max_examples=50)
def test_data_type_retention_policies(self, data_type):
"""Test that different data types have appropriate retention policies"""
# Define minimum retention periods by data type
min_retention_days = {
'raw_events': 30, # 30 days minimum
'aggregated_metrics': 365, # 1 year minimum
'reports': 365, # 1 year minimum
'exports': 90 # 90 days minimum
}
# Define maximum retention periods by data type
max_retention_days = {
'raw_events': 90, # 90 days maximum
'aggregated_metrics': 3650, # 10 years maximum
'reports': 3650, # 10 years maximum
'exports': 365 # 1 year maximum
}
# Verify retention policy exists for data type
assert data_type in min_retention_days, f"Minimum retention defined for {data_type}"
assert data_type in max_retention_days, f"Maximum retention defined for {data_type}"
assert min_retention_days[data_type] <= max_retention_days[data_type], f"Valid retention range for {data_type}"
@given(
data_count=st.integers(min_value=0, max_value=10000000),
deletion_batch_size=st.integers(min_value=100, max_value=10000)
)
@settings(max_examples=50)
def test_batch_deletion(self, data_count, deletion_batch_size):
"""Test that batch deletion is performed correctly"""
# Calculate number of batches
batches_needed = (data_count + deletion_batch_size - 1) // deletion_batch_size if data_count > 0 else 0
# Verify batch calculation
assert batches_needed >= 0, "Batches needed must be non-negative"
if data_count > 0:
assert batches_needed >= 1, "At least one batch needed for data"