#!/usr/bin/env python3
"""Prepare separate public copies, retaining detailed internal sources unchanged."""
from pathlib import Path
from lxml import html,etree
import re,json
ROOT=Path(__file__).resolve().parents[2];OUT=Path(__file__).resolve().parent/'public';OUT.mkdir(exist_ok=True)
PAGES=['AI_AGENT_PROJECT_INTRO_FOR_BEGINNERS.html','TEAM_AI_NATIVE_COLLABORATION_GUIDE.html','AI_PRACTICE_SHARING.html','CPT_AI_ENABLEMENT_BRIEF.html']
BASE='https://img.kxunpt.cn/public/ai-marketing-platform/'
report={}
def replace(el,markup):
 new=html.fragment_fromstring(markup);el.getparent().replace(el,new)
def section(id,title,body):return f'<section class="section" id="{id}"><h2>{title}</h2>{body}</section>'
for name in PAGES:
 doc=html.document_fromstring((ROOT/name).read_text());changes=[]
 for el in doc.xpath('//*[@id="daily-evolution"]'):
  replace(el,section('daily-evolution','持续建设，留下可复查证据','<p>任务按日期留存来源、版本、验证结果和下一步，支持交接与复盘。原始任务流水保留在团队授权工作空间；外网版展示上方月度汇总。</p>'));changes.append('原始每日任务标题改为汇总说明')
 for el in doc.xpath('//*[@id="aliyun-ops-agent-v0"]'):
  replace(el,section('aliyun-ops-agent-v0','专业 Agent：受控运维实践','<p>运维 Agent 已形成资产盘点、只读诊断、变更计划与审计留痕的工作流程。</p><div class="flow"><div class="flow-step"><h3>定位问题</h3><p>整理告警时间、影响和关联资料。</p></div><div class="flow-step"><h3>只读诊断</h3><p>在已有权限内核查，保留证据。</p></div><div class="flow-step"><h3>人工审核</h3><p>确认具体范围、验收和恢复方案。</p></div><div class="flow-step"><h3>执行回读</h3><p>仅在明确授权后执行，再核对业务结果。</p></div></div><p>权限与审批各自生效。具体实例、配置和诊断记录保留在内部，不以流程建设代替实际业务恢复。</p>'));changes.append('运维实例数量/参数/精确运行状态改为流程说明')
 # Preserve the reviewed Before/After gallery; do not substitute or remove its images.
 # Remove concrete vulnerability surfaces while keeping the access-control lesson.
 for el in doc.xpath('//p|//td|//li'):
  text=el.text_content()
  if any(x in text for x in ['未掩码个人信息','卡务、账户、消费规则','粗颗粒授权']):
   el.clear();el.text='客户演示按有效期、最小权限和数据范围开放，并进行真实登录、脱敏与越权检查；未通过验收的模块不开放。';changes.append('具体安全问题细节改为控制原则')
 # Keep series navigation self-contained; label internal-only evidence links honestly.
 for a in doc.xpath('//a[@href]'):
  href=a.get('href','')
  target=href.split('/')[-1].split('#')[0].split('?')[0]
  if target in PAGES:
   fragment=('#'+href.split('#',1)[1]) if '#' in href else ''
   a.set('href',target+fragment)
  elif href.startswith('#'):pass
  elif name=='TEAM_AI_NATIVE_COLLABORATION_GUIDE.html' and a.get('data-team-guide-link')=='true' and href.startswith('https://zhctpmt.yyangpt.cn/'):
   pass
  elif 'zhctpmt.yyangpt.cn' in href or href.endswith('.md') or href.startswith(('work/','standards-stack/','control/','modules/')):
   a.tag='span';a.attrib.pop('href',None);a.set('title','详细资料在团队授权工作空间使用');changes.append('内部来源链接保留名称，不向外网展开源文档')
 for el in doc.iter():
  if el.tag in ['style','script']:continue
  for field in ['text','tail']:
   s=getattr(el,field)
   if not s:continue
   s=s.replace('哈尔滨膳质舍项目','首个试点项目').replace('Harbin','试点')
   s=s.replace('公司层面说明草案 · 待 Jack 审阅','公司层面说明 · 2026 年 9 月版')
   s=s.replace('内容草案，待审阅。','成果与指标说明。')
   setattr(el,field,s)
 # Add the same clear navigation across all public copies.
 for nav in doc.xpath('//nav[contains(@class,"topbar")]'):
  if not nav.xpath('.//a[@href="CPT_AI_ENABLEMENT_BRIEF.html"]'):
   container=nav.xpath('.//*[contains(@class,"doc-nav")]')
   if container:container[0].insert(0,html.fragment_fromstring('<a href="CPT_AI_ENABLEMENT_BRIEF.html">公司总览</a>'))
 if name=='AI_AGENT_PROJECT_INTRO_FOR_BEGINNERS.html':
  import hashlib,base64
  manifest=json.loads((OUT.parent/'before-after-image-manifest.json').read_text())
  images=doc.xpath('//section[@id="before-after"]//img[@data-before-after-id]')
  assert len(images)==4,'Restored four-image gallery must remain in the latest leadership page'
  for image,row in zip(sorted(images,key=lambda i:i.get('data-before-after-id')),manifest['images']):
   assert image.get('data-before-after-id')==row['id']
   assert hashlib.sha256(base64.b64decode(image.get('src').split(',',1)[1])).hexdigest()==row['display_sha256']
 raw='<!DOCTYPE html>\n'+etree.tostring(doc,encoding='unicode',method='html')+'\n'
 # No credentials, network endpoints or local user paths may enter the public copy.
 text=' '.join(doc.xpath('//body//text()[not(ancestor::script) and not(ancestor::style)]'))
 guards={'IPv4':r'\b(?:\d{1,3}\.){3}\d{1,3}\b','local-user-path':r'/Users/[^\s<]+','private-key':r'BEGIN (?:RSA |OPENSSH )?PRIVATE KEY','access-key':r'LTAI[0-9A-Za-z]{12,}','bearer-token':r'Bearer\s+[a-zA-Z0-9._-]{16,}'}
 found={k:len(re.findall(p,text)) for k,p in guards.items() if re.search(p,text)}
 if found:raise SystemExit(f'{name}: review needed {found}')
 assert 'CPTAI native' in raw and 'KANGBITE · AI NATIVE' not in raw
 (OUT/name).write_text(raw)
 report[name]={'edits':dict((x,changes.count(x)) for x in set(changes)),'privacy_pattern_hits':found,'public_url':BASE+name}
(OUT.parent/'public-safety-review.json').write_text(json.dumps(report,ensure_ascii=False,indent=2)+'\n')
print(json.dumps({name:v['edits'] for name,v in report.items()},ensure_ascii=False))
