66 lines
4.9 KiB
Python
66 lines
4.9 KiB
Python
"""Inspect current-account customer membership using non-identifying aggregates."""
|
|
import json, sqlite3, sys, re
|
|
from pathlib import Path
|
|
from collections import Counter
|
|
ROOT=Path('C:/wechat_rpa');OUT=Path('C:/kefu/deploy/my-customers-20260918')
|
|
sys.dont_write_bytecode=True;sys.path.insert(0,str(ROOT))
|
|
import review_assistant_contacts as contacts
|
|
ctx=contacts.active_context();assert ctx.get('ready')
|
|
account=str(ctx['account_id']);paths=contacts._databases(account)
|
|
result={'selected_caches':{k:str(v.parent.parent.relative_to(ROOT)).replace('\\','/') for k,v in paths.items()},'tables':{},'aggregates':{}}
|
|
dbs={k:contacts._open(v) for k,v in paths.items()}
|
|
try:
|
|
for db in dbs.values():db.execute('BEGIN')
|
|
for filename,pattern in [('user.db',r'external|my_inner|inner_fw|colleague|wechat_workmate|half_self|out_contacts|wx_friend'),('session.db',r'customer|conversation_user_table|conversation_extra_table|vip_conversation')]:
|
|
db=dbs[filename]
|
|
for (table,) in db.execute("SELECT name FROM sqlite_master WHERE type='table' ORDER BY name"):
|
|
if not re.search(pattern,table):continue
|
|
q='"'+table.replace('"','""')+'"'
|
|
cols=[{'name':r[1],'type':r[2]} for r in db.execute('PRAGMA table_info('+q+')')]
|
|
item={'count':db.execute('SELECT count(*) FROM '+q).fetchone()[0],'columns':cols}
|
|
flags=[c['name'] for c in cols if c['name'] in {'status','flag','stranger_type','recommand_type','recommand_relation_type','customer_source_type','is_recommend','is_customer','type','customer_type'}]
|
|
for flag in flags:
|
|
item[flag+'_distribution']=[list(r) for r in db.execute(f'SELECT "{flag}",count(*) FROM {q} GROUP BY "{flag}"')]
|
|
result['tables'][filename+'/'+table]=item
|
|
u=dbs['user.db'];s=dbs['session.db'];m=dbs['message.db']
|
|
agg=result['aggregates']
|
|
agg['relation_naming']={}
|
|
for col in ['remarks','real_remarks','corp_remark','recommend_remark']:
|
|
agg['relation_naming'][col+'_present']=u.execute(f'SELECT count(*) FROM external_user_relation_v3 WHERE length(trim(coalesce("{col}",\'\')))>0').fetchone()[0]
|
|
for col in ['create_time','add_customer_time','remark_time']:
|
|
agg['relation_naming'][col+'_positive']=u.execute(f'SELECT count(*) FROM external_user_relation_v3 WHERE "{col}">0').fetchone()[0]
|
|
rows=u.execute('SELECT id,corp_id,name_status,info_level FROM user_table').fetchall()
|
|
owncorp=next((r[1] for r in rows if str(r[0])==account),None)
|
|
agg['user_table_self_present']=owncorp is not None
|
|
agg['user_corp_distribution']={'same_as_current':sum(r[1]==owncorp for r in rows),'other_nonzero':sum(bool(r[1]) and r[1]!=owncorp for r in rows),'zero_or_null':sum(not r[1] for r in rows)}
|
|
agg['user_name_status']=dict(Counter(str(r[2]) for r in rows));agg['user_info_level']=dict(Counter(str(r[3]) for r in rows))
|
|
relation={str(r[0]) for r in u.execute('SELECT user_id FROM external_user_relation_v3')}
|
|
external={str(r[0]) for r in u.execute('SELECT value FROM external_user_ids')}
|
|
users={str(r[0]) for r in rows}
|
|
agg['id_membership']={'relations':len(relation),'external_ids':len(external),'relation_in_external':len(relation&external),'relation_has_user_row':len(relation&users),'external_without_relation':len(external-relation)}
|
|
session={str(r[0]) for r in s.execute('SELECT id FROM conversation_table')}
|
|
messages={str(r[0]) for r in m.execute('SELECT DISTINCT conversation_id FROM message_table')}
|
|
def prefix_set(values):return dict(Counter(v.split(':',1)[0] if ':' in v else 'unprefixed' for v in values))
|
|
agg['session_prefix_counts']=prefix_set(session);agg['message_conversation_prefix_counts']=prefix_set(messages)
|
|
def mapping(ids,conversations):
|
|
matched={'M':set(),'S':set()};ambiguous=0;self_missing=0
|
|
for conv in conversations:
|
|
if re.fullmatch(r'M:\d+',conv):
|
|
uid=conv[2:]
|
|
if uid in ids:matched['M'].add(uid)
|
|
elif re.fullmatch(r'S:\d+_\d+',conv):
|
|
parts=conv[2:].split('_')
|
|
if account not in parts:self_missing+=1;continue
|
|
peer=[v for v in parts if v!=account]
|
|
if len(peer)!=1:ambiguous+=1;continue
|
|
if peer[0] in ids:matched['S'].add(peer[0])
|
|
return {'matched_M':len(matched['M']),'matched_S':len(matched['S']),'matched_both':len(matched['M']&matched['S']),'unmapped':len(ids-(matched['M']|matched['S'])),'S_without_current_account':self_missing,'S_ambiguous':ambiguous}
|
|
for tag,ids in [('relation',relation),('external',external)]:
|
|
agg[tag+'_mapping_session']=mapping(ids,session)
|
|
agg[tag+'_mapping_messages']=mapping(ids,messages)
|
|
agg['relation_status_name_coverage']=[list(r) for r in u.execute("SELECT r.status,r.stranger_type,(r.add_customer_time>0),count(*),sum(length(trim(coalesce(u.name,'')))>0),sum(length(trim(coalesce(r.real_remarks,'')))>0),sum(length(trim(coalesce(r.remarks,'')))>0) FROM external_user_relation_v3 r LEFT JOIN user_table u ON u.id=r.user_id GROUP BY r.status,r.stranger_type,(r.add_customer_time>0)")]
|
|
finally:
|
|
for db in dbs.values():db.close()
|
|
(OUT/'relation-aggregates.json').write_text(json.dumps(result,ensure_ascii=False,indent=2),encoding='utf-8')
|
|
print(json.dumps(result,ensure_ascii=False))
|