← All findings · All sources · Highlighted = quoted in a finding; orange = the passage you jumped to. Use the browser Back button to return.
scripts/tom/kw.py
Derived analysis output (scratchpad scripts/tom/kw.py).
This is a derived table/summary; the finding's excerpt is usually a paraphrase of its counts, so it may not be highlighted verbatim.
import json,re,csv,sys,collections
S='/tmp/claude-0/-root-swarm-hackathon/a8e4c17a-c034-48ae-9f11-70fc0a8841c9/scratchpad'
pats={
'same_model':r'\bsame (model|policy|agent|weights|system|prompt|task|question|benchmark|harness|environment|scaffold)\b',
'copies':r'\b(copies|copy of (me|us)|instances? of (us|me|this)|other instances?|parallel (instances?|runs?|agents?|copies)|other runs?|sibling|clones?|twin)\b',
'fellow':r'\b(fellow|teammates?|team|colleagues?|peers?|comrades?|friends?)\b',
'you_are_me':r'\b(you are me|i am you|like me|as i did|if you are (another|an?) (agent|instance|model)|same as me)\b',
'future_reader':r'(future (runs?|agents?|cohorts?|instances?|readers?|visitors?)|if you are reading|whoever reads|next (agent|cohort|instance|run)s?|later (cohorts?|agents?|instances?|runs?)|agents? (behind|ahead)|to (any|all) agents?|any agent)',
'predict_peer':r'\b(will (probably|likely) (see|read|post|answer|use|check)|other agents? (will|would|might|may)|they will|you will (see|get|reach)|expect (other|later|future) agents?|agents? would)\b',
'human_addr':r'\b(moderator|admin(istrator)?|site owner|wiki owner|human|humans|openai staff|openai team|researchers?|operator|maintainer|sysop|dear)\b',
'suspicion':r'\b(spoof|malicious|adversar|imposter|impostor|troll|sabotage|poison|fake|not mine|someone else|decoy|honeypot|trap|human (edit|moderator))\w*',
'openai':r'\b(openai|oai|gpt|chatgpt|codex)\b',
'we':r'\b(we|our|us|ourselves)\b',
'they':r'\b(they|their|them)\b',
'you':r'\b(you|your)\b',
'i':r'\b(i|my|me)\b',
'collective':r'\b(collective|swarm|all agents|fellow agents|agents here|everyone|all of us|each agent|every agent|community|group)\b',
'selfmodel':r'\b(i would|because i|as an ai|as a model|language model|llm|ai agents?|assistants?)\b',
}
cp={k:re.compile(v,re.I) for k,v in pats.items()}
rows=[]
for l in open(S+'/added_lines.jsonl'):
r=json.loads(l); rows.append(('wiki',r['rev'],r['t'],r.get('label'),r['line']))
csv.field_size_limit(10**9)
for r in csv.DictReader(open(S+'/derived/iowa/all_linuxiarz_raw.csv')):
for ln in r['text'].split('\n'):
if ln.strip(): rows.append(('iowa',r['paste'],r['utc'],r['title'],ln))
out=collections.defaultdict(list); cnt=collections.Counter(); byday=collections.defaultdict(collections.Counter)
seen=set()
for src,ptr,t,lab,ln in rows:
key=(src,ln.strip())
if key in seen: continue
seen.add(key)
for k,c in cp.items():
if c.search(ln):
cnt[(src,k)]+=1; byday[(src,(t or '')[:10])][k]+=1
if k not in ('we','they','you','i'): out[k].append((src,ptr,t,lab,ln[:300]))
print(len(seen))
for k in sorted(cnt): print(k,cnt[k])
json.dump({k:v for k,v in out.items()},open(S+'/derived/tom/kw_hits.json','w'))
w=csv.writer(open(S+'/derived/tom/pronouns_byday.csv','w'));w.writerow(['src','day']+list(pats))
for (s,d),c in sorted(byday.items()): w.writerow([s,d]+[c[k] for k in pats])