import matplotlib.pyplot as plt import numpy as np import pandas as pd from preprocess import split_sentences, split_words def graph_sentence_lengths(files, window=20): plt.figure(figsize=(12, 6)) for file in files: with open(file.name, 'r', encoding='utf-8') as f: text = f.read() sentences = split_sentences(text) lengths = [] for sent in sentences: words = split_words(sent) lengths.append(len(words)) n = len(lengths) if n > 1: smoothed = [] half = window // 2 for i in range(n): start = max(0, i - half) end = min(n, i + half + 1) smoothed.append(np.median(lengths[start:end])) x_normalized = np.linspace(0, 1, n) plt.plot(x_normalized, smoothed, label=file.name.split('/')[-1], linewidth=1.5) plt.xlabel('Относительная позиция в тексте') plt.ylabel('Длина предложения (сглажено)') plt.title('Распределение длин предложений по тексту') plt.legend(loc='upper right', fontsize=8) plt.grid(True, alpha=0.3) return plt.gcf()