Files
kefu/deploy/my-customers-20260918/audit_customer_relations.py
T
2026-09-21 10:34:06 +08:00

66 lines
4.9 KiB
Python

"""Inspect current-account customer membership using non-identifying aggregates."""
import json, sqlite3, sys, re
from pathlib import Path
from collections import Counter
ROOT=Path('C:/wechat_rpa');OUT=Path('C:/kefu/deploy/my-customers-20260918')
sys.dont_write_bytecode=True;sys.path.insert(0,str(ROOT))
import review_assistant_contacts as contacts
ctx=contacts.active_context();assert ctx.get('ready')
account=str(ctx['account_id']);paths=contacts._databases(account)
result={'selected_caches':{k:str(v.parent.parent.relative_to(ROOT)).replace('\\','/') for k,v in paths.items()},'tables':{},'aggregates':{}}
dbs={k:contacts._open(v) for k,v in paths.items()}
try:
for db in dbs.values():db.execute('BEGIN')
for filename,pattern in [('user.db',r'external|my_inner|inner_fw|colleague|wechat_workmate|half_self|out_contacts|wx_friend'),('session.db',r'customer|conversation_user_table|conversation_extra_table|vip_conversation')]:
db=dbs[filename]
for (table,) in db.execute("SELECT name FROM sqlite_master WHERE type='table' ORDER BY name"):
if not re.search(pattern,table):continue
q='"'+table.replace('"','""')+'"'
cols=[{'name':r[1],'type':r[2]} for r in db.execute('PRAGMA table_info('+q+')')]
item={'count':db.execute('SELECT count(*) FROM '+q).fetchone()[0],'columns':cols}
flags=[c['name'] for c in cols if c['name'] in {'status','flag','stranger_type','recommand_type','recommand_relation_type','customer_source_type','is_recommend','is_customer','type','customer_type'}]
for flag in flags:
item[flag+'_distribution']=[list(r) for r in db.execute(f'SELECT "{flag}",count(*) FROM {q} GROUP BY "{flag}"')]
result['tables'][filename+'/'+table]=item
u=dbs['user.db'];s=dbs['session.db'];m=dbs['message.db']
agg=result['aggregates']
agg['relation_naming']={}
for col in ['remarks','real_remarks','corp_remark','recommend_remark']:
agg['relation_naming'][col+'_present']=u.execute(f'SELECT count(*) FROM external_user_relation_v3 WHERE length(trim(coalesce("{col}",\'\')))>0').fetchone()[0]
for col in ['create_time','add_customer_time','remark_time']:
agg['relation_naming'][col+'_positive']=u.execute(f'SELECT count(*) FROM external_user_relation_v3 WHERE "{col}">0').fetchone()[0]
rows=u.execute('SELECT id,corp_id,name_status,info_level FROM user_table').fetchall()
owncorp=next((r[1] for r in rows if str(r[0])==account),None)
agg['user_table_self_present']=owncorp is not None
agg['user_corp_distribution']={'same_as_current':sum(r[1]==owncorp for r in rows),'other_nonzero':sum(bool(r[1]) and r[1]!=owncorp for r in rows),'zero_or_null':sum(not r[1] for r in rows)}
agg['user_name_status']=dict(Counter(str(r[2]) for r in rows));agg['user_info_level']=dict(Counter(str(r[3]) for r in rows))
relation={str(r[0]) for r in u.execute('SELECT user_id FROM external_user_relation_v3')}
external={str(r[0]) for r in u.execute('SELECT value FROM external_user_ids')}
users={str(r[0]) for r in rows}
agg['id_membership']={'relations':len(relation),'external_ids':len(external),'relation_in_external':len(relation&external),'relation_has_user_row':len(relation&users),'external_without_relation':len(external-relation)}
session={str(r[0]) for r in s.execute('SELECT id FROM conversation_table')}
messages={str(r[0]) for r in m.execute('SELECT DISTINCT conversation_id FROM message_table')}
def prefix_set(values):return dict(Counter(v.split(':',1)[0] if ':' in v else 'unprefixed' for v in values))
agg['session_prefix_counts']=prefix_set(session);agg['message_conversation_prefix_counts']=prefix_set(messages)
def mapping(ids,conversations):
matched={'M':set(),'S':set()};ambiguous=0;self_missing=0
for conv in conversations:
if re.fullmatch(r'M:\d+',conv):
uid=conv[2:]
if uid in ids:matched['M'].add(uid)
elif re.fullmatch(r'S:\d+_\d+',conv):
parts=conv[2:].split('_')
if account not in parts:self_missing+=1;continue
peer=[v for v in parts if v!=account]
if len(peer)!=1:ambiguous+=1;continue
if peer[0] in ids:matched['S'].add(peer[0])
return {'matched_M':len(matched['M']),'matched_S':len(matched['S']),'matched_both':len(matched['M']&matched['S']),'unmapped':len(ids-(matched['M']|matched['S'])),'S_without_current_account':self_missing,'S_ambiguous':ambiguous}
for tag,ids in [('relation',relation),('external',external)]:
agg[tag+'_mapping_session']=mapping(ids,session)
agg[tag+'_mapping_messages']=mapping(ids,messages)
agg['relation_status_name_coverage']=[list(r) for r in u.execute("SELECT r.status,r.stranger_type,(r.add_customer_time>0),count(*),sum(length(trim(coalesce(u.name,'')))>0),sum(length(trim(coalesce(r.real_remarks,'')))>0),sum(length(trim(coalesce(r.remarks,'')))>0) FROM external_user_relation_v3 r LEFT JOIN user_table u ON u.id=r.user_id GROUP BY r.status,r.stranger_type,(r.add_customer_time>0)")]
finally:
for db in dbs.values():db.close()
(OUT/'relation-aggregates.json').write_text(json.dumps(result,ensure_ascii=False,indent=2),encoding='utf-8')
print(json.dumps(result,ensure_ascii=False))