Text Generation
PEFT
Safetensors
lora
trl
grpo
gdpo
dpo
divpo
rlhf
diversity
creative-writing
mode-collapse
Instructions to use Mercity/creative-writing-llm with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use Mercity/creative-writing-llm with PEFT:
Task type is invalid.
- Notebooks
- Google Colab
- Kaggle
| """ | |
| Tests for evaluate.py's pure metric functions. These run LAST in the pipeline, | |
| so a bug here would waste every GPU-hour before it. | |
| Directionality is the thing being pinned down: self-BLEU is inverted relative to | |
| every other diversity metric (lower = more diverse), and getting that backwards | |
| in the report would invert the study's conclusion. | |
| """ | |
| import numpy as np | |
| from diversity import effective_rank | |
| from evaluate import cluster_count, distinct_n, self_bleu, topk_entropy | |
| IDENTICAL = ["The harbor clock struck twelve and the ferry did not come."] * 8 | |
| VARIED = [ | |
| "The harbor clock struck twelve and the ferry did not come.", | |
| "Marguerite sold her mother's piano to pay for the greenhouse.", | |
| "In the third week of the drought, the well began speaking Latin.", | |
| "He counted seventeen crows before admitting he was being followed.", | |
| "The recipe called for one tablespoon of regret, finely minced.", | |
| "Every letter she mailed arrived a decade before she wrote it.", | |
| "Nobody warned the astronauts that the moon would be so loud.", | |
| "My grandfather traded his shadow for a working knowledge of bees.", | |
| ] | |
| def test_distinct4_identical_is_low(): | |
| """8 copies of one sentence: only 1/8 of 4-grams are new.""" | |
| d = distinct_n(IDENTICAL, 4) | |
| assert d < 0.2, d | |
| def test_distinct4_varied_is_high(): | |
| assert distinct_n(VARIED, 4) > 0.9 | |
| def test_distinct4_ordering(): | |
| assert distinct_n(VARIED, 4) > distinct_n(IDENTICAL, 4) | |
| def test_self_bleu_is_inverted_lower_means_more_diverse(): | |
| """THE directionality check. Identical texts must score HIGH self-BLEU.""" | |
| hi = self_bleu(IDENTICAL) | |
| lo = self_bleu(VARIED) | |
| assert hi > lo, f"self-BLEU not inverted: identical={hi:.4f} varied={lo:.4f}" | |
| assert hi > 0.5, f"identical texts should have high self-BLEU, got {hi:.4f}" | |
| assert lo < 0.1, f"varied texts should have low self-BLEU, got {lo:.4f}" | |
| def test_self_bleu_bounded(): | |
| for texts in (IDENTICAL, VARIED): | |
| v = self_bleu(texts) | |
| assert 0.0 <= v <= 1.0, v | |
| def test_self_bleu_handles_short_and_single(): | |
| assert self_bleu(["hi"]) == 0.0 | |
| assert self_bleu([]) == 0.0 | |
| assert 0.0 <= self_bleu(["a b", "c d"]) <= 1.0 | |
| def test_cluster_count_two_clear_modes(): | |
| rng = np.random.default_rng(0) | |
| a = rng.standard_normal(32); a /= np.linalg.norm(a) | |
| b = rng.standard_normal(32); b /= np.linalg.norm(b) | |
| E = np.vstack([np.tile(a, (8, 1)) + 0.02 * rng.standard_normal((8, 32)), | |
| np.tile(b, (8, 1)) + 0.02 * rng.standard_normal((8, 32))]) | |
| E /= np.linalg.norm(E, axis=1, keepdims=True) | |
| assert cluster_count(E) == 2, cluster_count(E) | |
| def test_cluster_count_no_structure_is_one(): | |
| """Near-identical embeddings have no cluster structure -> a single mode. | |
| Originally FAILED at silhouette>0.05 (returned k=4 on a fully collapsed set), | |
| which is why SILHOUETTE_MIN was recalibrated to 0.50.""" | |
| rng = np.random.default_rng(1) | |
| a = rng.standard_normal(32); a /= np.linalg.norm(a) | |
| E = np.tile(a, (16, 1)) + 0.001 * rng.standard_normal((16, 32)) | |
| E /= np.linalg.norm(E, axis=1, keepdims=True) | |
| assert cluster_count(E) == 1, cluster_count(E) | |
| def test_cluster_count_small_input(): | |
| assert cluster_count(np.eye(3)) == 1 | |
| def test_topk_entropy_peaked_vs_flat(): | |
| """A near-deterministic distribution has ~0 entropy; uniform over k has log k.""" | |
| peaked = [{"a": np.log(0.999), "b": np.log(0.001)}] | |
| flat = [{c: np.log(0.25) for c in "abcd"}] | |
| assert topk_entropy(peaked) < 0.05 | |
| assert abs(topk_entropy(flat) - np.log(4)) < 1e-6 | |
| def test_topk_entropy_empty(): | |
| assert topk_entropy([]) == 0.0 | |
| assert topk_entropy([{}]) == 0.0 | |
| def test_topk_entropy_renormalizes_truncated_table(): | |
| """vLLM returns only the top-k, which does not sum to 1; we renormalize.""" | |
| partial = [{"a": np.log(0.4), "b": np.log(0.2)}] # sums to 0.6 | |
| h = topk_entropy(partial) | |
| p = np.array([2 / 3, 1 / 3]) | |
| assert abs(h - float(-(p * np.log(p)).sum())) < 1e-9 | |
| def test_eff_rank_identical_is_one(): | |
| rng = np.random.default_rng(3) | |
| a = rng.standard_normal(32); a /= np.linalg.norm(a) | |
| assert abs(effective_rank(np.tile(a, (16, 1))) - 1.0) < 1e-6 | |
| def test_eff_rank_orthogonal_is_n(): | |
| Q, _ = np.linalg.qr(np.random.default_rng(4).standard_normal((32, 8))) | |
| assert abs(effective_rank(Q.T[:8]) - 8.0) < 1e-6 | |
| def test_eff_rank_two_clusters_is_about_two(): | |
| """The case silhouette got right but only at a threshold that broke the | |
| collapsed case. Effective rank handles both without a threshold.""" | |
| rng = np.random.default_rng(5) | |
| a = rng.standard_normal(32); a /= np.linalg.norm(a) | |
| b = rng.standard_normal(32); b /= np.linalg.norm(b) | |
| E = np.vstack([np.tile(a, (8, 1)) + 0.02 * rng.standard_normal((8, 32)), | |
| np.tile(b, (8, 1)) + 0.02 * rng.standard_normal((8, 32))]) | |
| E /= np.linalg.norm(E, axis=1, keepdims=True) | |
| assert 1.8 < effective_rank(E) < 2.6, effective_rank(E) | |
| def test_eff_rank_is_monotone_in_spread(): | |
| """Monotone IN EXPECTATION. Averaged over draws because effective rank | |
| saturates around 12.6 (not 16) for 16 unit vectors in 32 dims -- random | |
| directions retain residual correlation -- so single draws at the top of the | |
| range can invert by chance. The metric is fine; the assertion has to be | |
| statistical.""" | |
| rng = np.random.default_rng(6) | |
| a = rng.standard_normal(32); a /= np.linalg.norm(a) | |
| prev = 0.0 | |
| for noise in (0.001, 0.05, 0.15, 0.5, 2.0): | |
| vals = [] | |
| for _ in range(5): | |
| E = np.tile(a, (16, 1)) + noise * rng.standard_normal((16, 32)) | |
| E /= np.linalg.norm(E, axis=1, keepdims=True) | |
| vals.append(effective_rank(E)) | |
| r = float(np.mean(vals)) | |
| assert r >= prev - 1e-6, f"non-monotone at noise={noise}: {r} < {prev}" | |
| prev = r | |
| def test_eff_rank_separates_collapsed_from_spread(): | |
| """The exact discrimination silhouette FAILED: collapsed 0.202 vs spread | |
| 0.195 were indistinguishable. Effective rank must separate them clearly.""" | |
| rng = np.random.default_rng(7) | |
| a = rng.standard_normal(32); a /= np.linalg.norm(a) | |
| collapsed = np.tile(a, (16, 1)) + 0.001 * rng.standard_normal((16, 32)) | |
| collapsed /= np.linalg.norm(collapsed, axis=1, keepdims=True) | |
| spread = rng.standard_normal((16, 32)) | |
| spread /= np.linalg.norm(spread, axis=1, keepdims=True) | |
| assert effective_rank(spread) > effective_rank(collapsed) + 8.0 | |
| if __name__ == "__main__": | |
| import sys, traceback | |
| fns = [(n, f) for n, f in sorted(globals().items()) | |
| if n.startswith("test_") and callable(f)] | |
| bad = 0 | |
| for n, f in fns: | |
| try: | |
| f(); print(f" PASS {n}") | |
| except Exception: | |
| bad += 1; print(f" FAIL {n}"); traceback.print_exc() | |
| print(f"\n{len(fns)-bad}/{len(fns)} passed") | |
| sys.exit(1 if bad else 0) | |