"""How much groups are held together by a topic word rather than a predicament. python3 topic_share.py summary_embeddings.json # baseline: groups of scene summaries python3 topic_share.py predicament_embeddings.json # groups of re-described predicaments Clusters the given embeddings at k = 60 (seeds 11, 12, 13). For each candidate group (at least 5 scenes, at least 4 stories, no story over 40%), finds the job, relationship or place word most common in the group's ORIGINAL scene summaries, and the share of its scenes that contain it. The first human review read groups that were artists, police, courtship and hotels: that is this number being high. Reported as the median over groups, then the median over seeds. No inference. """ import collections, json, pathlib, re, statistics, sys from topic_words import TOPIC_WORDS path = sys.argv[1] if len(sys.argv) > 1 else 'summary_embeddings.json' sys.argv = ['measure.py', path] exec(pathlib.Path(__file__).with_name('measure.py').read_text(encoding='utf-8').split("print(f'{path.name}")[0]) # now defined: chunks, V, n, story, kmeans, describe, MIN_SIZE, MIN_STORIES, MAX_DOMINANT original = {x['chunk']: x['summary'] for x in json.loads(pathlib.Path('summaries_clean.json').read_text(encoding='utf-8'))['summaries']} words = {i: set(re.findall(r"[a-zé]+", original.get(i, '').lower())) & TOPIC_WORDS for i in range(n)} K, SEEDS = 60, (11, 12, 13) per_seed, examples = [], [] for seed in SEEDS: shares = [] for group in kmeans(K, seed): size, nst, dom, _ = describe(group) if size < MIN_SIZE or nst < MIN_STORIES or dom > MAX_DOMINANT: continue counts = collections.Counter(w for i in group for w in words[i]) if counts: word, hits = counts.most_common(1)[0] else: word, hits = '-', 0 shares.append(hits / size) if seed == SEEDS[0]: examples.append((hits / size, word, size)) per_seed.append((len(shares), statistics.median(shares) if shares else 0.0)) print(f'seed {seed}: {len(shares)} candidate groups, median top topic-word share {per_seed[-1][1]:.0%}', flush=True) print(f'\n{pathlib.Path(path).name}: median over seeds {statistics.median(s for _, s in per_seed):.0%}, ' f'candidate groups {statistics.median(c for c, _ in per_seed):.0f}') print('most topic-bound groups at seed 11:', ', '.join(f'{w} {s:.0%} of {z}' for s, w, z in sorted(examples, reverse=True)[:8]))