Prechádzať zdrojové kódy

feat: add multi-source business knowledge ingestion

AI-Co-Authored-By: Codex
chendeben 1 mesiac pred
rodič
commit
0fae96aca5

+ 19 - 0
admin-web/src/pages/BusinessAssistant.tsx

@@ -38,6 +38,10 @@ import { useEffect, useMemo, useState } from 'react';
 
 import { Metric, PageHeader, Surface } from '@/components/Page';
 import { useBotAccess } from '@/contexts/BotAccess';
+import {
+  KnowledgeCandidatesPanel,
+  KnowledgeSourcesPanel,
+} from '@/pages/BusinessKnowledgeSources';
 import { apiRequest, jsonOptions } from '@/services/api';
 import type {
   AssistantAccountSettings,
@@ -677,6 +681,21 @@ export default function BusinessAssistantPage() {
                   </Surface>
                 ),
               },
+              {
+                key: 'knowledge-sources',
+                label: '知识来源',
+                children: <KnowledgeSourcesPanel connectionId={selectedConnection} />,
+              },
+              {
+                key: 'knowledge-candidates',
+                label: '知识候选',
+                children: (
+                  <KnowledgeCandidatesPanel
+                    connectionId={selectedConnection}
+                    onKnowledgeChanged={() => loadConnectionData(selectedConnection)}
+                  />
+                ),
+              },
               {
                 key: 'conversations',
                 label: `会话台 (${conversations.total})`,

+ 625 - 0
admin-web/src/pages/BusinessKnowledgeSources.tsx

@@ -0,0 +1,625 @@
+import {
+  Button,
+  Col,
+  Empty,
+  Form,
+  Input,
+  InputNumber,
+  Modal,
+  Row,
+  Select,
+  Space,
+  Spin,
+  Switch,
+  Table,
+  Tag,
+  Typography,
+  message,
+} from 'antd';
+import type { ColumnsType } from 'antd/es/table';
+import dayjs from 'dayjs';
+import {
+  Ban,
+  Check,
+  CirclePlus,
+  Pencil,
+  Play,
+  RefreshCw,
+  RotateCw,
+  Trash2,
+} from 'lucide-react';
+import { useEffect, useState } from 'react';
+
+import { Surface } from '@/components/Page';
+import { apiRequest, jsonOptions } from '@/services/api';
+import type {
+  AssistantKnowledgeAuthorPolicy,
+  AssistantKnowledgeCandidate,
+  AssistantKnowledgeCandidateStatus,
+  AssistantKnowledgePublicationMode,
+  AssistantKnowledgeSource,
+  AssistantKnowledgeSourceType,
+  Paged,
+} from '@/types';
+
+type SourceForm = {
+  source_type: AssistantKnowledgeSourceType;
+  chat_id: string;
+  publication_mode: AssistantKnowledgePublicationMode;
+  author_policy: AssistantKnowledgeAuthorPolicy;
+  allowed_user_ids: string[];
+  include_linked_chat: boolean;
+  backfill_days: number;
+  backfill_limit: number;
+  enabled: boolean;
+};
+
+const emptyPage = <T,>(): Paged<T> => ({ items: [], total: 0, page: 1, page_size: 20 });
+const touchButtonStyle = { minHeight: 44 };
+
+const sourceTypeLabels: Record<AssistantKnowledgeSourceType, string> = {
+  channel: '官方频道',
+  group: '群组聊天',
+  business: 'Business 人工对话',
+};
+
+const candidateStatusLabels: Record<AssistantKnowledgeCandidateStatus, string> = {
+  pending: '待审核',
+  published: '已发布',
+  rejected: '已驳回',
+  stale: '已失效',
+};
+
+const candidateStatusColors: Record<AssistantKnowledgeCandidateStatus, string> = {
+  pending: 'warning',
+  published: 'success',
+  rejected: 'error',
+  stale: 'default',
+};
+
+export function KnowledgeSourcesPanel({ connectionId }: { connectionId: string }) {
+  const [loading, setLoading] = useState(false);
+  const [page, setPage] = useState<Paged<AssistantKnowledgeSource>>(emptyPage());
+  const [open, setOpen] = useState(false);
+  const [target, setTarget] = useState<AssistantKnowledgeSource | null>(null);
+  const [form] = Form.useForm<SourceForm>();
+  const sourceType = Form.useWatch('source_type', form);
+
+  const load = async (current = 1) => {
+    setLoading(true);
+    try {
+      const params = new URLSearchParams({
+        connection_id: connectionId,
+        page: String(current),
+        page_size: '20',
+      });
+      setPage(
+        await apiRequest<Paged<AssistantKnowledgeSource>>(
+          `/business-assistant/knowledge-sources?${params}`,
+        ),
+      );
+    } catch (error) {
+      message.error(error instanceof Error ? error.message : '知识来源加载失败');
+    } finally {
+      setLoading(false);
+    }
+  };
+
+  useEffect(() => {
+    void load();
+  }, [connectionId]);
+
+  const openSource = (source?: AssistantKnowledgeSource) => {
+    setTarget(source || null);
+    form.setFieldsValue(
+      source || {
+        source_type: 'channel',
+        chat_id: '',
+        publication_mode: 'review',
+        author_policy: 'admins_or_allowlist',
+        allowed_user_ids: [],
+        include_linked_chat: true,
+        backfill_days: 30,
+        backfill_limit: 100,
+        enabled: true,
+      },
+    );
+    setOpen(true);
+  };
+
+  const saveSource = (values: SourceForm) => {
+    Modal.confirm({
+      title: target ? '保存知识来源?' : '添加知识来源?',
+      content:
+        values.publication_mode === 'auto'
+          ? '可信作者产生的候选将自动发布到知识库。'
+          : '新内容会先进入知识候选,审核后才会生效。',
+      okText: '确认保存',
+      cancelText: '取消',
+      onOk: async () => {
+        const path = target
+          ? `/business-assistant/knowledge-sources/${target.source_id}`
+          : '/business-assistant/knowledge-sources';
+        await apiRequest(
+          path,
+          jsonOptions(target ? 'PUT' : 'POST', {
+            ...values,
+            connection_id: connectionId,
+            confirm: true,
+          }),
+        );
+        setOpen(false);
+        message.success(target ? '知识来源已更新' : '知识来源已添加');
+        await load();
+      },
+    });
+  };
+
+  const sourceAction = (
+    source: AssistantKnowledgeSource,
+    action: 'sync' | 'enable' | 'disable',
+  ) => {
+    const label = action === 'sync' ? '同步历史消息' : action === 'enable' ? '启用来源' : '停用来源';
+    Modal.confirm({
+      title: `${label}?`,
+      content:
+        action === 'sync'
+          ? `系统会按最近 ${source.backfill_days} 天、最多 ${source.backfill_limit} 条消息后台提取知识。`
+          : undefined,
+      okText: '确认执行',
+      cancelText: '取消',
+      onOk: async () => {
+        await apiRequest(
+          `/business-assistant/knowledge-sources/${source.source_id}/actions`,
+          jsonOptions('POST', { action, confirm: true }),
+        );
+        message.success(action === 'sync' ? '历史同步已进入后台队列' : `${label}成功`);
+        await load(page.page);
+      },
+    });
+  };
+
+  const columns: ColumnsType<AssistantKnowledgeSource> = [
+    {
+      title: '来源',
+      key: 'source',
+      minWidth: 220,
+      render: (_, source) => (
+        <div>
+          <Typography.Text strong>{source.title}</Typography.Text>
+          <div style={{ marginTop: 4 }}>
+            <Space size={[4, 4]} wrap>
+              <Tag>{sourceTypeLabels[source.source_type]}</Tag>
+              {source.chat_id !== '0' ? <Tag>{source.chat_id}</Tag> : null}
+              {source.linked_chat_title ? <Tag>讨论群:{source.linked_chat_title}</Tag> : null}
+            </Space>
+          </div>
+        </div>
+      ),
+    },
+    {
+      title: '发布策略',
+      key: 'publication',
+      width: 150,
+      render: (_, source) => (
+        <Tag color={source.publication_mode === 'auto' ? 'processing' : 'warning'}>
+          {source.publication_mode === 'auto' ? '可信内容自动发布' : '全部人工审核'}
+        </Tag>
+      ),
+    },
+    {
+      title: '同步状态',
+      key: 'sync',
+      width: 150,
+      render: (_, source) => (
+        <div>
+          <Tag color={source.sync_status === 'failed' ? 'error' : source.sync_status === 'running' ? 'processing' : 'default'}>
+            {source.sync_status || 'idle'}
+          </Tag>
+          {typeof source.last_sync_processed === 'number' ? (
+            <div><Typography.Text type="secondary">已处理 {source.last_sync_processed} 条</Typography.Text></div>
+          ) : null}
+        </div>
+      ),
+    },
+    {
+      title: '状态',
+      dataIndex: 'enabled',
+      width: 88,
+      render: (enabled: boolean) => (
+        <Tag color={enabled ? 'success' : 'default'}>{enabled ? '启用' : '停用'}</Tag>
+      ),
+    },
+    {
+      title: '最近事件',
+      dataIndex: 'last_event_at',
+      width: 160,
+      render: (value?: string) => (value ? dayjs(value).format('YYYY-MM-DD HH:mm') : '—'),
+    },
+    {
+      title: '操作',
+      key: 'actions',
+      fixed: 'right',
+      width: 250,
+      render: (_, source) => (
+        <Space size={8} wrap>
+          <Button
+            style={touchButtonStyle}
+            aria-label={`编辑来源 ${source.title}`}
+            icon={<Pencil size={16} />}
+            onClick={() => openSource(source)}
+          >
+            编辑
+          </Button>
+          {source.source_type !== 'business' ? (
+            <Button
+              style={touchButtonStyle}
+              aria-label={`同步来源 ${source.title}`}
+              icon={<RotateCw size={16} />}
+              loading={source.sync_status === 'queued' || source.sync_status === 'running'}
+              onClick={() => sourceAction(source, 'sync')}
+            >
+              同步
+            </Button>
+          ) : null}
+          <Button
+            style={touchButtonStyle}
+            aria-label={`${source.enabled ? '停用' : '启用'}来源 ${source.title}`}
+            icon={source.enabled ? <Ban size={16} /> : <Play size={16} />}
+            onClick={() => sourceAction(source, source.enabled ? 'disable' : 'enable')}
+          >
+            {source.enabled ? '停用' : '启用'}
+          </Button>
+          <Button
+            danger
+            style={touchButtonStyle}
+            aria-label={`删除来源 ${source.title}`}
+            icon={<Trash2 size={16} />}
+            onClick={() =>
+              Modal.confirm({
+                title: `删除“${source.title}”?`,
+                content: '该来源产生的候选会失效,已关联的自动知识将立即停用。',
+                okText: '确认删除',
+                okButtonProps: { danger: true },
+                cancelText: '取消',
+                onOk: async () => {
+                  await apiRequest(
+                    `/business-assistant/knowledge-sources/${source.source_id}`,
+                    jsonOptions('DELETE', { confirm: true }),
+                  );
+                  message.success('知识来源已删除');
+                  await load();
+                },
+              })
+            }
+          >
+            删除
+          </Button>
+        </Space>
+      ),
+    },
+  ];
+
+  return (
+    <Spin spinning={loading}>
+      <Surface
+        actions={
+          <Space wrap>
+            <Button style={touchButtonStyle} icon={<RefreshCw size={16} />} onClick={() => void load(page.page)}>
+              刷新
+            </Button>
+            <Button type="primary" style={touchButtonStyle} icon={<CirclePlus size={16} />} onClick={() => openSource()}>
+              添加来源
+            </Button>
+          </Space>
+        }
+      >
+        <Typography.Paragraph type="secondary">
+          官方频道可配置自动发布;讨论群仅管理员或白名单成员可自动发布,普通成员内容进入候选审核。Business 对话只学习账号本人的人工答复。
+        </Typography.Paragraph>
+        <div className="table-wrap">
+          <Table
+            rowKey="source_id"
+            columns={columns}
+            dataSource={page.items}
+            pagination={{
+              current: page.page,
+              pageSize: page.page_size,
+              total: page.total,
+              onChange: (current) => void load(current),
+            }}
+            scroll={{ x: 1150 }}
+            locale={{ emptyText: <Empty description="尚未配置知识来源" /> }}
+          />
+        </div>
+      </Surface>
+
+      <Modal
+        title={target ? '编辑知识来源' : '添加知识来源'}
+        open={open}
+        onCancel={() => setOpen(false)}
+        onOk={() => form.submit()}
+        okText="继续"
+        cancelText="取消"
+        width={720}
+        destroyOnHidden
+      >
+        <Form form={form} layout="vertical" onFinish={saveSource} preserve={false}>
+          <Row gutter={16}>
+            <Col xs={24} md={12}>
+              <Form.Item name="source_type" label="来源类型" rules={[{ required: true }]}>
+                <Select
+                  disabled={Boolean(target)}
+                  options={Object.entries(sourceTypeLabels).map(([value, label]) => ({ value, label }))}
+                />
+              </Form.Item>
+            </Col>
+            <Col xs={24} md={12}>
+              <Form.Item
+                name="chat_id"
+                label="频道或群组 ID"
+                extra={sourceType === 'business' ? 'Business 对话无需填写。' : '例如 -1001234567890'}
+                rules={sourceType === 'business' ? [] : [{ required: true, message: '请输入聊天 ID' }]}
+              >
+                <Input disabled={Boolean(target) || sourceType === 'business'} inputMode="numeric" />
+              </Form.Item>
+            </Col>
+            <Col xs={24} md={12}>
+              <Form.Item name="publication_mode" label="发布方式">
+                <Select
+                  options={[
+                    { value: 'review', label: '全部进入候选审核' },
+                    { value: 'auto', label: '可信作者自动发布' },
+                  ]}
+                />
+              </Form.Item>
+            </Col>
+            <Col xs={24} md={12}>
+              <Form.Item name="author_policy" label="群组可信作者">
+                <Select
+                  disabled={sourceType === 'business'}
+                  options={[
+                    { value: 'admins_or_allowlist', label: '管理员或白名单' },
+                    { value: 'admins', label: '仅管理员' },
+                    { value: 'allowlist', label: '仅白名单' },
+                    { value: 'all', label: '所有成员' },
+                  ]}
+                />
+              </Form.Item>
+            </Col>
+            <Col xs={24}>
+              <Form.Item name="allowed_user_ids" label="可信成员白名单" extra="填写 Telegram 用户 ID,输入后按回车。">
+                <Select mode="tags" tokenSeparators={[',', ',', ' ']} disabled={sourceType === 'business'} />
+              </Form.Item>
+            </Col>
+            <Col xs={24} md={8}>
+              <Form.Item name="include_linked_chat" label="包含频道讨论群" valuePropName="checked">
+                <Switch disabled={sourceType !== 'channel'} />
+              </Form.Item>
+            </Col>
+            <Col xs={12} md={8}>
+              <Form.Item name="backfill_days" label="回溯天数">
+                <InputNumber min={0} max={365} style={{ width: '100%' }} />
+              </Form.Item>
+            </Col>
+            <Col xs={12} md={8}>
+              <Form.Item name="backfill_limit" label="最多消息数">
+                <InputNumber min={1} max={500} style={{ width: '100%' }} />
+              </Form.Item>
+            </Col>
+            <Col xs={24}>
+              <Form.Item name="enabled" label="启用实时采集" valuePropName="checked">
+                <Switch />
+              </Form.Item>
+            </Col>
+          </Row>
+        </Form>
+      </Modal>
+    </Spin>
+  );
+}
+
+export function KnowledgeCandidatesPanel({
+  connectionId,
+  onKnowledgeChanged,
+}: {
+  connectionId: string;
+  onKnowledgeChanged: () => void | Promise<void>;
+}) {
+  const [loading, setLoading] = useState(false);
+  const [page, setPage] = useState<Paged<AssistantKnowledgeCandidate>>(emptyPage());
+  const [status, setStatus] = useState<AssistantKnowledgeCandidateStatus | ''>('pending');
+  const [query, setQuery] = useState('');
+
+  const load = async (
+    current = 1,
+    filters?: { status?: AssistantKnowledgeCandidateStatus | ''; query?: string },
+  ) => {
+    setLoading(true);
+    try {
+      const params = new URLSearchParams({
+        connection_id: connectionId,
+        status: filters?.status ?? status,
+        query: filters?.query ?? query,
+        page: String(current),
+        page_size: '20',
+      });
+      setPage(
+        await apiRequest<Paged<AssistantKnowledgeCandidate>>(
+          `/business-assistant/knowledge-candidates?${params}`,
+        ),
+      );
+    } catch (error) {
+      message.error(error instanceof Error ? error.message : '知识候选加载失败');
+    } finally {
+      setLoading(false);
+    }
+  };
+
+  useEffect(() => {
+    void load();
+  }, [connectionId, status]);
+
+  const candidateAction = (
+    candidate: AssistantKnowledgeCandidate,
+    action: 'approve' | 'reject' | 'regenerate',
+  ) => {
+    const labels = { approve: '批准并发布', reject: '驳回候选', regenerate: '重新生成' };
+    Modal.confirm({
+      title: `${labels[action]}?`,
+      content:
+        action === 'approve'
+          ? '批准后会立即写入正式知识库并参与自动回答。'
+          : action === 'reject'
+            ? '如已发布,关联知识也会被停用。'
+            : '系统会重新读取 30 天内保留的原始事件并覆盖当前候选。',
+      okText: '确认执行',
+      okButtonProps: action === 'reject' ? { danger: true } : undefined,
+      cancelText: '取消',
+      onOk: async () => {
+        await apiRequest(
+          `/business-assistant/knowledge-candidates/${candidate.candidate_id}/actions`,
+          jsonOptions('POST', { action, confirm: true }),
+        );
+        message.success(`${labels[action]}成功`);
+        if (action === 'approve' || action === 'reject') await onKnowledgeChanged();
+        await load(page.page);
+      },
+    });
+  };
+
+  const columns: ColumnsType<AssistantKnowledgeCandidate> = [
+    {
+      title: '候选知识',
+      key: 'knowledge',
+      minWidth: 300,
+      render: (_, candidate) => (
+        <div>
+          <Typography.Text strong>{candidate.question}</Typography.Text>
+          <Typography.Paragraph style={{ margin: '6px 0 0', whiteSpace: 'pre-wrap' }} ellipsis={{ rows: 3, expandable: true }}>
+            {candidate.answer}
+          </Typography.Paragraph>
+          <Space size={[4, 4]} wrap>
+            {candidate.keywords.map((keyword) => <Tag key={keyword}>{keyword}</Tag>)}
+          </Space>
+        </div>
+      ),
+    },
+    {
+      title: '来源引用',
+      key: 'source',
+      minWidth: 210,
+      render: (_, candidate) => (
+        <div>
+          <Typography.Text>{candidate.source_snapshot.title || candidate.source_id}</Typography.Text>
+          <div><Typography.Text type="secondary">chat {candidate.source_snapshot.chat_id} · message {candidate.source_snapshot.message_id}</Typography.Text></div>
+        </div>
+      ),
+    },
+    {
+      title: '置信度',
+      dataIndex: 'confidence',
+      width: 90,
+      render: (value: number) => `${Math.round(value * 100)}%`,
+    },
+    {
+      title: '状态',
+      dataIndex: 'status',
+      width: 100,
+      render: (value: AssistantKnowledgeCandidateStatus) => (
+        <Tag color={candidateStatusColors[value]}>{candidateStatusLabels[value]}</Tag>
+      ),
+    },
+    {
+      title: '更新时间',
+      dataIndex: 'updated_at',
+      width: 160,
+      render: (value?: string) => (value ? dayjs(value).format('YYYY-MM-DD HH:mm') : '—'),
+    },
+    {
+      title: '操作',
+      key: 'actions',
+      fixed: 'right',
+      width: 250,
+      render: (_, candidate) => (
+        <Space size={8} wrap>
+          {candidate.status !== 'published' && candidate.status !== 'stale' ? (
+            <Button
+              type="primary"
+              style={touchButtonStyle}
+              icon={<Check size={16} />}
+              onClick={() => candidateAction(candidate, 'approve')}
+            >
+              批准
+            </Button>
+          ) : null}
+          {candidate.status !== 'rejected' && candidate.status !== 'stale' ? (
+            <Button
+              danger
+              style={touchButtonStyle}
+              icon={<Ban size={16} />}
+              onClick={() => candidateAction(candidate, 'reject')}
+            >
+              驳回
+            </Button>
+          ) : null}
+          <Button
+            style={touchButtonStyle}
+            icon={<RotateCw size={16} />}
+            onClick={() => candidateAction(candidate, 'regenerate')}
+          >
+            重新生成
+          </Button>
+        </Space>
+      ),
+    },
+  ];
+
+  return (
+    <Spin spinning={loading}>
+      <Surface
+        actions={
+          <Space wrap>
+            <Input.Search
+              aria-label="搜索知识候选"
+              allowClear
+              placeholder="问题、答案或关键词"
+              onSearch={(value) => {
+                setQuery(value);
+                void load(1, { query: value });
+              }}
+            />
+            <Select
+              aria-label="知识候选状态"
+              value={status || undefined}
+              allowClear
+              placeholder="全部状态"
+              style={{ width: 140, minHeight: 44 }}
+              options={Object.entries(candidateStatusLabels).map(([value, label]) => ({ value, label }))}
+              onChange={(value) => setStatus(value || '')}
+            />
+            <Button style={touchButtonStyle} icon={<RefreshCw size={16} />} onClick={() => void load(page.page)}>
+              刷新
+            </Button>
+          </Space>
+        }
+      >
+        <div className="table-wrap">
+          <Table
+            rowKey="candidate_id"
+            columns={columns}
+            dataSource={page.items}
+            pagination={{
+              current: page.page,
+              pageSize: page.page_size,
+              total: page.total,
+              onChange: (current) => void load(current),
+            }}
+            scroll={{ x: 1150 }}
+            locale={{ emptyText: <Empty description="当前筛选条件下没有知识候选" /> }}
+          />
+        </div>
+      </Surface>
+    </Spin>
+  );
+}

+ 61 - 0
admin-web/src/types.ts

@@ -130,6 +130,67 @@ export interface AssistantKnowledgeEntry {
   updated_at?: string;
 }
 
+export type AssistantKnowledgeSourceType = 'channel' | 'group' | 'business';
+export type AssistantKnowledgePublicationMode = 'review' | 'auto';
+export type AssistantKnowledgeAuthorPolicy =
+  | 'admins_or_allowlist'
+  | 'admins'
+  | 'allowlist'
+  | 'all';
+
+export interface AssistantKnowledgeSource {
+  source_id: string;
+  connection_id: string;
+  source_type: AssistantKnowledgeSourceType;
+  chat_id: string;
+  title: string;
+  publication_mode: AssistantKnowledgePublicationMode;
+  author_policy: AssistantKnowledgeAuthorPolicy;
+  allowed_user_ids: string[];
+  include_linked_chat: boolean;
+  linked_chat_id: string;
+  linked_chat_title?: string;
+  backfill_days: number;
+  backfill_limit: number;
+  enabled: boolean;
+  access_status?: string;
+  sync_status: 'idle' | 'queued' | 'running' | 'completed' | 'failed';
+  last_sync_processed?: number;
+  last_event_at?: string;
+  last_error?: string;
+  updated_at?: string;
+}
+
+export type AssistantKnowledgeCandidateStatus =
+  | 'pending'
+  | 'published'
+  | 'rejected'
+  | 'stale';
+
+export interface AssistantKnowledgeCandidate {
+  candidate_id: string;
+  connection_id: string;
+  source_id: string;
+  event_id: string;
+  question: string;
+  aliases: string[];
+  keywords: string[];
+  answer: string;
+  tags: string[];
+  confidence: number;
+  status: AssistantKnowledgeCandidateStatus;
+  knowledge_entry_id?: string;
+  stale_reason?: string;
+  source_snapshot: {
+    title: string;
+    source_type: AssistantKnowledgeSourceType | 'discussion';
+    chat_id: string;
+    message_id: string;
+    author_id: string;
+  };
+  updated_at?: string;
+}
+
 export type AssistantConversationStatus =
   | 'auto'
   | 'handoff'

+ 56 - 0
admin-web/tests/e2e/admin.spec.ts

@@ -314,6 +314,56 @@ test.beforeEach(async ({ page }) => {
         handoff_message: '请稍候,已转人工。',
         unsupported_message: '该消息需要人工处理。',
       };
+    } else if (path.endsWith('/business-assistant/knowledge-sources')) {
+      data = {
+        items: [{
+          source_id: 'source-1',
+          connection_id: 'conn-a',
+          source_type: 'channel',
+          chat_id: '-100100',
+          title: '官方公告频道',
+          publication_mode: 'auto',
+          author_policy: 'admins_or_allowlist',
+          allowed_user_ids: [],
+          include_linked_chat: true,
+          linked_chat_id: '-100101',
+          linked_chat_title: '官方讨论群',
+          backfill_days: 30,
+          backfill_limit: 100,
+          enabled: true,
+          sync_status: 'completed',
+          last_sync_processed: 12,
+        }],
+        total: 1,
+        page: 1,
+        page_size: 20,
+      };
+    } else if (path.endsWith('/business-assistant/knowledge-candidates')) {
+      data = {
+        items: [{
+          candidate_id: 'candidate-1',
+          connection_id: 'conn-a',
+          source_id: 'source-1',
+          event_id: 'event-1',
+          question: '配送范围?',
+          aliases: [],
+          keywords: ['配送'],
+          answer: '仅支持市区配送。',
+          tags: ['配送'],
+          confidence: 0.92,
+          status: 'pending',
+          source_snapshot: {
+            title: '官方公告频道',
+            source_type: 'channel',
+            chat_id: '-100100',
+            message_id: '8',
+            author_id: '900',
+          },
+        }],
+        total: 1,
+        page: 1,
+        page_size: 20,
+      };
     } else if (path.endsWith('/business-assistant/knowledge')) {
       data = { items: [], total: 0, page: 1, page_size: 20 };
     } else if (path.endsWith('/business-assistant/conversations')) {
@@ -392,6 +442,12 @@ test('登录与主要管理视图在不同视口无页面级横向滚动', async
     if (path === '/admin/business-assistant') {
       await expect(page.getByText('允许回复')).toBeVisible();
       await expect(page.getByText('允许标记已读')).toBeVisible();
+      await page.getByRole('tab', { name: '知识来源' }).click();
+      await expect(page.getByText('官方公告频道')).toBeVisible();
+      await expect(page.getByText('讨论群:官方讨论群')).toBeVisible();
+      await page.getByRole('tab', { name: '知识候选' }).click();
+      await expect(page.getByText('配送范围?')).toBeVisible();
+      await expect(page.getByRole('button', { name: '批准' })).toBeVisible();
       await page.screenshot({ path: testInfo.outputPath('business-assistant.png'), fullPage: true });
     }
     expect(await page.evaluate(() => document.documentElement.scrollWidth - window.innerWidth)).toBeLessThanOrEqual(1);

+ 71 - 0
admin-web/tests/unit/BusinessAssistant.test.tsx

@@ -92,6 +92,62 @@ beforeEach(() => {
       };
     }
     if (path.startsWith('/business-assistant/settings')) return settings;
+    if (path.startsWith('/business-assistant/knowledge-sources')) {
+      return {
+        items: [
+          {
+            source_id: 'source-1',
+            connection_id: 'conn-a',
+            source_type: 'channel',
+            chat_id: '-100100',
+            title: '官方公告',
+            publication_mode: 'auto',
+            author_policy: 'admins_or_allowlist',
+            allowed_user_ids: [],
+            include_linked_chat: true,
+            linked_chat_id: '-100101',
+            linked_chat_title: '官方讨论群',
+            backfill_days: 30,
+            backfill_limit: 100,
+            enabled: true,
+            sync_status: 'completed',
+            last_sync_processed: 12,
+          },
+        ],
+        total: 1,
+        page: 1,
+        page_size: 20,
+      };
+    }
+    if (path.startsWith('/business-assistant/knowledge-candidates')) {
+      return {
+        items: [
+          {
+            candidate_id: 'candidate-1',
+            connection_id: 'conn-a',
+            source_id: 'source-1',
+            event_id: 'event-1',
+            question: '配送范围?',
+            aliases: [],
+            keywords: ['配送'],
+            answer: '仅支持市区配送。',
+            tags: ['配送'],
+            confidence: 0.92,
+            status: 'pending',
+            source_snapshot: {
+              title: '官方公告',
+              source_type: 'channel',
+              chat_id: '-100100',
+              message_id: '8',
+              author_id: '900',
+            },
+          },
+        ],
+        total: 1,
+        page: 1,
+        page_size: 20,
+      };
+    }
     if (path.startsWith('/business-assistant/knowledge')) {
       return {
         items: [
@@ -160,3 +216,18 @@ test('展示运行状态、连接权限,并用最新搜索词查询知识库',
     ).toBe(true);
   });
 });
+
+test('展示多源知识与待审核候选', async () => {
+  render(<BusinessAssistantPage />);
+  await screen.findAllByText('店主');
+
+  fireEvent.click(screen.getByRole('tab', { name: '知识来源' }));
+  expect(await screen.findByText('官方公告')).toBeInTheDocument();
+  expect(screen.getByText('讨论群:官方讨论群')).toBeInTheDocument();
+  expect(screen.getByRole('button', { name: '同步来源 官方公告' })).toBeInTheDocument();
+
+  fireEvent.click(screen.getByRole('tab', { name: '知识候选' }));
+  expect(await screen.findByText('配送范围?')).toBeInTheDocument();
+  expect(screen.getByText('仅支持市区配送。')).toBeInTheDocument();
+  expect(screen.getByRole('button', { name: '批准' })).toBeInTheDocument();
+});

+ 18 - 0
tests/conftest.py

@@ -16,6 +16,8 @@ ROOT = Path(__file__).resolve().parents[1]
 class FakeTelegramApp:
     def __init__(self) -> None:
         self.members: dict[tuple[int, int], object] = {}
+        self.chats: dict[int, object] = {}
+        self.histories: dict[int, list[object]] = {}
         self.sent_messages: list[tuple[int, str]] = []
         self.sent_photos: list[tuple[int, object, str]] = []
 
@@ -38,6 +40,16 @@ class FakeTelegramApp:
             until_date=None,
         )
 
+    async def get_chat(self, chat_id: int):
+        chat = self.chats.get(int(chat_id))
+        if chat is None:
+            raise ValueError("chat not found")
+        return chat
+
+    async def get_chat_history(self, chat_id: int, limit: int = 100):
+        for message in self.histories.get(int(chat_id), [])[:limit]:
+            yield message
+
     async def send_message(self, chat_id: int, text: str, **_kwargs):
         self.sent_messages.append((int(chat_id), text))
         return SimpleNamespace(chat=SimpleNamespace(id=int(chat_id)), id=len(self.sent_messages))
@@ -52,6 +64,12 @@ class FakeTelegramApp:
     def on_callback_query(self, *_args, **_kwargs):
         return lambda function: function
 
+    def on_edited_message(self, *_args, **_kwargs):
+        return lambda function: function
+
+    def on_deleted_messages(self, *_args, **_kwargs):
+        return lambda function: function
+
     def on_chat_member_updated(self, *_args, **_kwargs):
         return lambda function: function
 

+ 309 - 0
tests/test_business_assistant.py

@@ -7,6 +7,7 @@ from types import SimpleNamespace
 import pytest
 from aiohttp import CookieJar
 from aiohttp.test_utils import TestClient, TestServer
+from pyrogram.enums import ChatMemberStatus
 
 
 def connection_payload(connection_id: str = "conn-a", *, can_reply: bool = True) -> dict:
@@ -48,6 +49,7 @@ class FakeProvider:
     def __init__(self, *, error: Exception | None = None) -> None:
         self.error = error
         self.calls: list[dict] = []
+        self.extraction_calls: list[dict] = []
 
     async def decide(self, **values):
         self.calls.append(values)
@@ -65,6 +67,21 @@ class FakeProvider:
     async def test_connection(self):
         return {"ok": True, "model": "test-model", "response_id": "response-1"}
 
+    async def extract_knowledge(self, **values):
+        self.extraction_calls.append(values)
+        if self.error:
+            raise self.error
+        return [
+            {
+                "question": values.get("question_context") or "如何办理?",
+                "aliases": [],
+                "keywords": ["办理"],
+                "answer": values["content"],
+                "tags": ["自动采集"],
+                "confidence": 0.92,
+            }
+        ]
+
 
 class FakeBusinessApi:
     def __init__(self) -> None:
@@ -229,6 +246,251 @@ def test_ai_contract_and_reply_window_validation(app_modules):
     )
     assert not service.business_reply_window_open({}, now=now)
 
+    extracted = service.parse_knowledge_extraction(
+        json.dumps(
+            {
+                "items": [
+                    {
+                        "question": "如何配送?",
+                        "aliases": ["配送范围"],
+                        "keywords": ["配送"],
+                        "answer": "仅支持市区配送。",
+                        "tags": ["配送"],
+                        "confidence": 0.9,
+                    }
+                ]
+            }
+        )
+    )
+    assert extracted[0]["confidence"] == 0.9
+    assert (
+        service.redact_business_knowledge_text(
+            "电话 138 0013 8000,邮箱 user@example.com,Telegram @private_user"
+        )
+        == "电话 [电话已脱敏],邮箱 [邮箱已脱敏],Telegram [用户名已脱敏]"
+    )
+
+
+async def test_source_event_candidates_publish_edit_delete_and_isolate(app_modules):
+    dbassistant = app_modules.load("wbb.utils.dbassistant")
+    await dbassistant.upsert_business_connection(connection_payload("conn-a"))
+    await dbassistant.upsert_business_connection(connection_payload("conn-b"))
+    source = await dbassistant.create_knowledge_source(
+        "conn-a",
+        {
+            "source_type": "channel",
+            "chat_id": -100100,
+            "title": "官方频道",
+            "publication_mode": "auto",
+            "author_policy": "admins_or_allowlist",
+        },
+    )
+    event, changed = await dbassistant.record_source_event(
+        source,
+        event_key="telegram:-100100:10",
+        content="每天九点营业。",
+        metadata={"chat_id": -100100, "message_id": 10, "trusted_author": True},
+    )
+    assert changed is True
+    candidates = await dbassistant.replace_event_candidates(
+        source,
+        event,
+        [
+            {
+                "question": "几点营业?",
+                "aliases": ["营业时间"],
+                "keywords": ["营业"],
+                "answer": "每天九点营业。",
+                "tags": ["营业"],
+                "confidence": 0.95,
+            }
+        ],
+        auto_publish=True,
+    )
+    candidate = candidates[0]
+    first_entry_id = candidate["knowledge_entry_id"]
+    first_entry = await app_modules.wbb.db.business_assistant_knowledge.find_one(
+        {"bot_id": "primary", "entry_id": first_entry_id}
+    )
+    assert candidate["status"] == "published"
+    assert first_entry["enabled"] is True
+
+    edited_event, changed = await dbassistant.record_source_event(
+        source,
+        event_key="telegram:-100100:10",
+        content="每天十点营业。",
+        metadata={"chat_id": -100100, "message_id": 10, "trusted_author": True},
+    )
+    assert changed is True
+    edited = await dbassistant.replace_event_candidates(
+        source,
+        edited_event,
+        [
+            {
+                "question": "几点营业?",
+                "aliases": [],
+                "keywords": ["营业"],
+                "answer": "每天十点营业。",
+                "tags": ["营业"],
+                "confidence": 0.96,
+            }
+        ],
+        auto_publish=True,
+    )
+    assert edited[0]["knowledge_entry_id"] == first_entry_id
+    updated_entry = await app_modules.wbb.db.business_assistant_knowledge.find_one(
+        {"bot_id": "primary", "entry_id": first_entry_id}
+    )
+    assert updated_entry["answer"] == "每天十点营业。"
+    assert await dbassistant.mark_source_event_deleted(
+        source["source_id"], "telegram:-100100:10"
+    )
+    stale = await dbassistant.get_knowledge_candidate(candidate["candidate_id"])
+    disabled_entry = await app_modules.wbb.db.business_assistant_knowledge.find_one(
+        {"bot_id": "primary", "entry_id": first_entry_id}
+    )
+    assert stale["status"] == "stale"
+    assert disabled_entry["enabled"] is False
+
+    sources_a, total_a = await dbassistant.list_knowledge_sources("conn-a")
+    sources_b, total_b = await dbassistant.list_knowledge_sources("conn-b")
+    assert total_a == 1 and sources_a[0]["source_id"] == source["source_id"]
+    assert total_b == 0 and sources_b == []
+    event_indexes = await app_modules.wbb.db.business_assistant_source_events.index_information()
+    assert event_indexes["expires_at_1"]["expireAfterSeconds"] == 0
+
+
+async def test_business_human_reply_learns_only_when_source_enabled(app_modules):
+    dbassistant, _service, runtime, _knowledge = await prepare_runtime(app_modules)
+    await runtime.process_update(
+        {
+            "update_id": 9,
+            "business_message": customer_message(
+                message_id=9, text="配送到哪里?", chat_id=509, sender_id=509
+            ),
+        }
+    )
+    await runtime.process_update(
+        {
+            "update_id": 10,
+            "business_message": customer_message(
+                message_id=10, text="仅支持市区配送。", chat_id=509, sender_id=900
+            ),
+        }
+    )
+    assert runtime.provider.extraction_calls == []
+
+    await dbassistant.create_knowledge_source(
+        "conn-a",
+        {
+            "source_type": "business",
+            "title": "人工接待对话",
+            "publication_mode": "auto",
+        },
+    )
+    await runtime.process_update(
+        {
+            "update_id": 15,
+            "business_message": customer_message(
+                message_id=15, text="支持哪些区域?", chat_id=515, sender_id=515
+            ),
+        }
+    )
+    await runtime.process_update(
+        {
+            "update_id": 16,
+            "business_message": customer_message(
+                message_id=16, text="仅支持市区配送。", chat_id=515, sender_id=900
+            ),
+        }
+    )
+    assert len(runtime.provider.extraction_calls) == 1
+    candidates, total = await dbassistant.list_knowledge_candidates(
+        "conn-a", status="published"
+    )
+    assert total == 1
+    assert candidates[0]["answer"] == "仅支持市区配送。"
+    entry_id = candidates[0]["knowledge_entry_id"]
+
+    await runtime.process_update(
+        {
+            "update_id": 17,
+            "edited_business_message": customer_message(
+                message_id=16, text="仅二环内支持配送。", chat_id=515, sender_id=900
+            ),
+        }
+    )
+    edited = await dbassistant.get_knowledge_candidate(candidates[0]["candidate_id"])
+    assert edited["answer"] == "仅二环内支持配送。"
+    assert edited["knowledge_entry_id"] == entry_id
+
+    await runtime.process_update(
+        {
+            "update_id": 18,
+            "deleted_business_messages": {
+                "business_connection_id": "conn-a",
+                "chat": {"id": 515},
+                "message_ids": [16],
+            },
+        }
+    )
+    stale = await dbassistant.get_knowledge_candidate(candidates[0]["candidate_id"])
+    entry = await app_modules.wbb.db.business_assistant_knowledge.find_one(
+        {"bot_id": "primary", "entry_id": entry_id}
+    )
+    assert stale["status"] == "stale"
+    assert entry["enabled"] is False
+
+
+async def test_group_collection_requires_trusted_author_for_auto_publish(app_modules):
+    dbassistant, _service, runtime, _knowledge = await prepare_runtime(app_modules)
+    module = app_modules.load("wbb.modules.business_assistant")
+    module._runtime = runtime
+    await dbassistant.create_knowledge_source(
+        "conn-a",
+        {
+            "source_type": "group",
+            "chat_id": -100300,
+            "title": "运营讨论群",
+            "publication_mode": "auto",
+            "author_policy": "admins_or_allowlist",
+        },
+    )
+
+    def group_message(message_id: int, author_id: int, text: str):
+        return SimpleNamespace(
+            id=message_id,
+            text=text,
+            caption=None,
+            outgoing=False,
+            is_automatic_forward=False,
+            date=datetime.now(UTC),
+            chat=SimpleNamespace(id=-100300),
+            from_user=SimpleNamespace(id=author_id, is_bot=False),
+        )
+
+    ordinary = group_message(1, 301, "市区支持当日配送。")
+    await module.process_knowledge_source_message(ordinary)
+    pending, pending_total = await dbassistant.list_knowledge_candidates(
+        "conn-a", status="pending"
+    )
+    assert pending_total == 1
+    assert pending[0]["status"] == "pending"
+
+    app_modules.app.members[(-100300, 302)] = SimpleNamespace(
+        status=ChatMemberStatus.ADMINISTRATOR
+    )
+    admin_message = group_message(2, 302, "每周一至周五提供配送。")
+    await module.process_knowledge_source_message(admin_message)
+    published, published_total = await dbassistant.list_knowledge_candidates(
+        "conn-a", status="published"
+    )
+    assert published_total == 1
+    assert published[0]["source_snapshot"]["author_id"] == 302
+    extraction_count = len(runtime.provider.extraction_calls)
+    await module.process_knowledge_source_message(admin_message)
+    assert len(runtime.provider.extraction_calls) == extraction_count
+
 
 async def test_business_updates_are_idempotent_and_manual_reply_pauses(app_modules):
     dbassistant, _service, runtime, knowledge = await prepare_runtime(app_modules)
@@ -598,6 +860,53 @@ async def test_business_assistant_api_permission_csrf_audit_and_clear(app_module
         )
         assert (await listed.json())["data"]["total"] == 1
 
+        created_source = await client.post(
+            "/api/admin/v1/business-assistant/knowledge-sources",
+            headers={"X-CSRF-Token": csrf},
+            json={
+                "connection_id": "conn-a",
+                "source_type": "business",
+                "title": "人工对话",
+                "publication_mode": "review",
+                "confirm": True,
+            },
+        )
+        assert created_source.status == 201
+        source = (await created_source.json())["data"]
+        event, _ = await dbassistant.record_source_event(
+            source,
+            event_key="business:701:8",
+            content="市区可配送。",
+            metadata={"chat_id": 701, "message_id": 8, "trusted_author": True},
+        )
+        candidates = await dbassistant.replace_event_candidates(
+            source,
+            event,
+            [
+                {
+                    "question": "配送范围?",
+                    "keywords": ["配送"],
+                    "answer": "市区可配送。",
+                    "confidence": 0.9,
+                }
+            ],
+            auto_publish=False,
+        )
+        candidate_page = await client.get(
+            "/api/admin/v1/business-assistant/knowledge-candidates"
+            "?connection_id=conn-a&status=pending"
+        )
+        assert candidate_page.status == 200
+        assert (await candidate_page.json())["data"]["total"] == 1
+        approved = await client.post(
+            "/api/admin/v1/business-assistant/knowledge-candidates/"
+            f"{candidates[0]['candidate_id']}/actions",
+            headers={"X-CSRF-Token": csrf},
+            json={"action": "approve", "confirm": True},
+        )
+        assert approved.status == 200
+        assert (await approved.json())["data"]["status"] == "published"
+
         cleared = await client.post(
             f"/api/admin/v1/business-assistant/conversations/{conversation['conversation_id']}/actions",
             headers={"X-CSRF-Token": csrf},

+ 192 - 0
wbb/admin/api.py

@@ -98,17 +98,26 @@ from wbb.utils.dbassistant import (
     close_conversation,
     conversation_detail,
     create_knowledge_entry,
+    create_knowledge_source,
     delete_knowledge_entry,
+    delete_knowledge_source,
     ensure_assistant_indexes,
     get_account_settings,
     get_business_connection,
+    get_knowledge_candidate,
+    get_knowledge_source,
     list_business_connections,
     list_conversations,
+    list_knowledge_candidates,
     list_knowledge_entries,
+    list_knowledge_sources,
     pause_conversation,
+    publish_knowledge_candidate,
+    reject_knowledge_candidate,
     resume_conversation,
     update_account_settings,
     update_knowledge_entry,
+    update_knowledge_source,
     usage_metrics,
 )
 from wbb.utils.dbdirectory import (
@@ -746,6 +755,34 @@ class AdminApi:
             f"{API_PREFIX}/business-assistant/knowledge/test",
             self.business_assistant_knowledge_test,
         )
+        router.add_get(
+            f"{API_PREFIX}/business-assistant/knowledge-sources",
+            self.business_assistant_knowledge_sources,
+        )
+        router.add_post(
+            f"{API_PREFIX}/business-assistant/knowledge-sources",
+            self.business_assistant_knowledge_source_create,
+        )
+        router.add_put(
+            f"{API_PREFIX}/business-assistant/knowledge-sources/{{source_id}}",
+            self.business_assistant_knowledge_source_update,
+        )
+        router.add_delete(
+            f"{API_PREFIX}/business-assistant/knowledge-sources/{{source_id}}",
+            self.business_assistant_knowledge_source_delete,
+        )
+        router.add_post(
+            f"{API_PREFIX}/business-assistant/knowledge-sources/{{source_id}}/actions",
+            self.business_assistant_knowledge_source_action,
+        )
+        router.add_get(
+            f"{API_PREFIX}/business-assistant/knowledge-candidates",
+            self.business_assistant_knowledge_candidates,
+        )
+        router.add_post(
+            f"{API_PREFIX}/business-assistant/knowledge-candidates/{{candidate_id}}/actions",
+            self.business_assistant_knowledge_candidate_action,
+        )
         router.add_get(
             f"{API_PREFIX}/business-assistant/conversations",
             self.business_assistant_conversations,
@@ -2497,6 +2534,161 @@ class AdminApi:
             raise ApiProblem("assistant_provider_error", str(exc), status=502) from exc
         return success(result)
 
+    async def business_assistant_knowledge_sources(
+        self, request: web.Request
+    ) -> web.Response:
+        connection_id = str(request.query.get("connection_id") or "").strip()
+        if not connection_id:
+            raise ApiProblem("connection_required", "请先选择一个 Business 连接。")
+        page, page_size = page_params(request)
+        items, total = await list_knowledge_sources(
+            connection_id, page=page, page_size=page_size
+        )
+        return success(
+            {"items": items, "total": total, "page": page, "page_size": page_size}
+        )
+
+    async def business_assistant_knowledge_source_create(
+        self, request: web.Request
+    ) -> web.Response:
+        body = await json_body(request)
+        require_confirmation(body)
+        connection_id = str(body.pop("connection_id", "")).strip()
+        body.pop("confirm", None)
+        if not connection_id:
+            raise ApiProblem("connection_required", "请先选择一个 Business 连接。")
+        from wbb.modules.business_assistant import validate_knowledge_source
+
+        validated = await validate_knowledge_source(body)
+        source = await create_knowledge_source(connection_id, validated)
+        set_audit(
+            request,
+            "business_assistant.knowledge_source.create",
+            target_id=source["source_id"],
+            summary=str(source["title"]),
+            metadata={"connection_id": connection_id, "source_type": source["source_type"]},
+        )
+        return success(source, status=201)
+
+    async def business_assistant_knowledge_source_update(
+        self, request: web.Request
+    ) -> web.Response:
+        body = await json_body(request)
+        require_confirmation(body)
+        body.pop("confirm", None)
+        source_id = request.match_info["source_id"]
+        if body.get("enabled") is False:
+            from wbb.modules.business_assistant import cancel_knowledge_source_sync
+
+            await cancel_knowledge_source_sync(source_id)
+        source = await update_knowledge_source(source_id, body)
+        set_audit(
+            request,
+            "business_assistant.knowledge_source.update",
+            target_id=source_id,
+            summary=str(source.get("title") or "更新知识来源"),
+        )
+        return success(source)
+
+    async def business_assistant_knowledge_source_delete(
+        self, request: web.Request
+    ) -> web.Response:
+        body = await json_body(request)
+        require_confirmation(body)
+        source_id = request.match_info["source_id"]
+        source = await get_knowledge_source(source_id)
+        if not source:
+            raise AssistantDataError("source_not_found", "未找到知识来源。")
+        set_audit(
+            request,
+            "business_assistant.knowledge_source.delete",
+            target_id=source_id,
+            summary=str(source.get("title") or "删除知识来源"),
+        )
+        from wbb.modules.business_assistant import cancel_knowledge_source_sync
+
+        await cancel_knowledge_source_sync(source_id)
+        await delete_knowledge_source(source_id)
+        return success({"deleted": True})
+
+    async def business_assistant_knowledge_source_action(
+        self, request: web.Request
+    ) -> web.Response:
+        body = await json_body(request)
+        require_confirmation(body)
+        source_id = request.match_info["source_id"]
+        action = str(body.get("action") or "")
+        if action in {"enable", "disable"}:
+            if action == "disable":
+                from wbb.modules.business_assistant import cancel_knowledge_source_sync
+
+                await cancel_knowledge_source_sync(source_id)
+            result = await update_knowledge_source(
+                source_id, {"enabled": action == "enable"}
+            )
+        elif action == "sync":
+            from wbb.modules.business_assistant import schedule_knowledge_source_sync
+
+            result = await schedule_knowledge_source_sync(source_id)
+        else:
+            raise ApiProblem("invalid_action", "不支持的知识来源操作。")
+        set_audit(
+            request,
+            f"business_assistant.knowledge_source.{action}",
+            target_id=source_id,
+            summary=f"知识来源操作:{action}",
+        )
+        return success(result)
+
+    async def business_assistant_knowledge_candidates(
+        self, request: web.Request
+    ) -> web.Response:
+        connection_id = str(request.query.get("connection_id") or "").strip()
+        if not connection_id:
+            raise ApiProblem("connection_required", "请先选择一个 Business 连接。")
+        page, page_size = page_params(request)
+        items, total = await list_knowledge_candidates(
+            connection_id,
+            status=str(request.query.get("status") or ""),
+            source_id=str(request.query.get("source_id") or ""),
+            query=str(request.query.get("query") or ""),
+            page=page,
+            page_size=page_size,
+        )
+        return success(
+            {"items": items, "total": total, "page": page, "page_size": page_size}
+        )
+
+    async def business_assistant_knowledge_candidate_action(
+        self, request: web.Request
+    ) -> web.Response:
+        body = await json_body(request)
+        require_confirmation(body)
+        candidate_id = request.match_info["candidate_id"]
+        action = str(body.get("action") or "")
+        if not await get_knowledge_candidate(candidate_id):
+            raise AssistantDataError("candidate_not_found", "未找到知识候选。")
+        if action == "approve":
+            result: Any = await publish_knowledge_candidate(candidate_id)
+        elif action == "reject":
+            result = await reject_knowledge_candidate(candidate_id)
+        elif action == "regenerate":
+            from wbb.modules.business_assistant import regenerate_knowledge_candidate
+
+            try:
+                result = await regenerate_knowledge_candidate(candidate_id)
+            except AssistantProviderError as exc:
+                raise ApiProblem("assistant_provider_error", str(exc), status=502) from exc
+        else:
+            raise ApiProblem("invalid_action", "不支持的知识候选操作。")
+        set_audit(
+            request,
+            f"business_assistant.knowledge_candidate.{action}",
+            target_id=candidate_id,
+            summary=f"知识候选操作:{action}",
+        )
+        return success(result)
+
     async def business_assistant_conversations(
         self, request: web.Request
     ) -> web.Response:

+ 260 - 2
wbb/modules/business_assistant.py

@@ -1,24 +1,35 @@
 from __future__ import annotations
 
+import asyncio
 from contextlib import suppress
+from datetime import UTC, datetime, timedelta
+from typing import Any
 
 from pyrogram import filters
-from pyrogram.enums import ChatMemberStatus
+from pyrogram.enums import ChatMemberStatus, ChatType
 
 import wbb
 from wbb import SUDOERS, app
-from wbb.services.business_assistant import BusinessAssistantRuntime
+from wbb.services.business_assistant import BusinessAssistantRuntime, classify_handoff
 from wbb.utils.dbassistant import (
+    AssistantDataError,
     get_account_settings,
     get_business_connection,
     get_conversation,
+    get_knowledge_source,
+    list_enabled_sources_for_chat,
+    list_sources_for_chat,
+    mark_source_event_deleted,
     resume_conversation,
+    touch_knowledge_source,
+    update_knowledge_source_sync,
 )
 
 __MODULE__ = "智能接待"
 __HELP__ = "Telegram Business 智能接待由 Web 管理后台配置。"
 
 _runtime: BusinessAssistantRuntime | None = None
+_source_sync_tasks: dict[str, asyncio.Task[None]] = {}
 
 
 async def start_business_assistant_runtime() -> None:
@@ -40,6 +51,12 @@ async def start_business_assistant_runtime() -> None:
 
 async def stop_business_assistant_runtime() -> None:
     global _runtime
+    tasks = [task for task in _source_sync_tasks.values() if not task.done()]
+    for task in tasks:
+        task.cancel()
+    if tasks:
+        await asyncio.gather(*tasks, return_exceptions=True)
+    _source_sync_tasks.clear()
     if _runtime is None:
         return
     await _runtime.stop()
@@ -50,6 +67,247 @@ def get_business_assistant_runtime() -> BusinessAssistantRuntime | None:
     return _runtime
 
 
+async def validate_knowledge_source(values: dict[str, Any]) -> dict[str, Any]:
+    source_type = str(values.get("source_type") or "")
+    if source_type == "business":
+        return {
+            **values,
+            "chat_id": 0,
+            "title": str(values.get("title") or "Business 日常对话"),
+            "linked_chat_id": 0,
+            "linked_chat_title": "",
+            "access_status": "business_connected",
+        }
+    try:
+        chat_id = int(values.get("chat_id") or 0)
+    except (TypeError, ValueError) as exc:
+        raise AssistantDataError("invalid_source", "来源聊天 ID 必须是整数。") from exc
+    if not chat_id:
+        raise AssistantDataError("invalid_source", "请填写频道或群组聊天 ID。")
+    try:
+        chat = await app.get_chat(chat_id)
+        member = await app.get_chat_member(chat_id, int(wbb.BOT_ID))
+    except Exception as exc:
+        raise AssistantDataError(
+            "source_access_denied", "机器人无法访问该频道或群组,请先将机器人加入。"
+        ) from exc
+    chat_type = getattr(chat, "type", None)
+    if source_type == "channel" and chat_type != ChatType.CHANNEL:
+        raise AssistantDataError("source_type_mismatch", "该聊天不是频道。")
+    if source_type == "group" and chat_type not in {ChatType.GROUP, ChatType.SUPERGROUP}:
+        raise AssistantDataError("source_type_mismatch", "该聊天不是群组。")
+    if source_type not in {"channel", "group"}:
+        raise AssistantDataError("invalid_source", "知识来源类型无效。")
+    if source_type == "channel" and member.status not in {
+        ChatMemberStatus.OWNER,
+        ChatMemberStatus.ADMINISTRATOR,
+    }:
+        raise AssistantDataError(
+            "source_access_denied", "机器人必须是频道管理员才能持续采集频道消息。"
+        )
+    linked_chat = getattr(chat, "linked_chat", None)
+    return {
+        **values,
+        "chat_id": chat_id,
+        "title": str(getattr(chat, "title", None) or values.get("title") or chat_id),
+        "linked_chat_id": int(getattr(linked_chat, "id", 0) or 0),
+        "linked_chat_title": str(getattr(linked_chat, "title", "") or ""),
+        "access_status": str(getattr(member.status, "value", member.status)),
+    }
+
+
+def _message_text(message: Any) -> str:
+    return str(getattr(message, "text", None) or getattr(message, "caption", None) or "").strip()
+
+
+async def _trusted_source_author(
+    source: dict[str, Any], message: Any, *, effective_source_type: str
+) -> bool:
+    if effective_source_type == "channel":
+        return True
+    author = getattr(message, "from_user", None)
+    author_id = int(getattr(author, "id", 0) or 0)
+    policy = str(source.get("author_policy") or "admins_or_allowlist")
+    allowlisted = author_id in {int(item) for item in source.get("allowed_user_ids") or []}
+    if policy == "all":
+        return True
+    if policy == "allowlist":
+        return allowlisted
+    sender_chat_id = int(getattr(getattr(message, "sender_chat", None), "id", 0) or 0)
+    is_admin = sender_chat_id == int(message.chat.id)
+    if author_id:
+        with suppress(Exception):
+            member = await app.get_chat_member(int(message.chat.id), author_id)
+            is_admin = member.status in {
+                ChatMemberStatus.OWNER,
+                ChatMemberStatus.ADMINISTRATOR,
+            }
+    if policy == "admins":
+        return is_admin
+    return allowlisted or is_admin
+
+
+async def _ingest_source_message(source: dict[str, Any], message: Any) -> bool:
+    runtime = get_business_assistant_runtime()
+    if runtime is None:
+        return False
+    text = _message_text(message)
+    author = getattr(message, "from_user", None)
+    if (
+        not text
+        or bool(getattr(message, "outgoing", False))
+        or bool(getattr(author, "is_bot", False))
+        or classify_handoff(text) == "sensitive_request"
+    ):
+        return False
+    chat_id = int(message.chat.id)
+    linked_discussion = bool(
+        source.get("include_linked_chat")
+        and int(source.get("linked_chat_id") or 0) == chat_id
+    )
+    if linked_discussion and bool(getattr(message, "is_automatic_forward", False)):
+        return False
+    effective_source_type = "discussion" if linked_discussion else str(source["source_type"])
+    trusted = await _trusted_source_author(
+        source, message, effective_source_type=effective_source_type
+    )
+    author_id = int(getattr(author, "id", 0) or 0)
+    message_id = int(getattr(message, "id", 0) or 0)
+    if not message_id:
+        return False
+    try:
+        await runtime.ingestion.ingest(
+            source,
+            event_key=f"telegram:{chat_id}:{message_id}",
+            content=text,
+            metadata={
+                "effective_source_type": effective_source_type,
+                "chat_id": chat_id,
+                "message_id": message_id,
+                "author_id": author_id,
+                "trusted_author": trusted,
+                "message_date": getattr(message, "date", None),
+            },
+            auto_publish=bool(
+                source.get("publication_mode") == "auto" and trusted
+            ),
+        )
+        await touch_knowledge_source(str(source["source_id"]))
+        return True
+    except Exception as exc:
+        await touch_knowledge_source(str(source["source_id"]), error=str(exc))
+        wbb.log.error(f"知识来源采集失败:{source['source_id']} / {exc}")
+        return False
+
+
+async def process_knowledge_source_message(message: Any) -> None:
+    chat_id = int(getattr(getattr(message, "chat", None), "id", 0) or 0)
+    if not chat_id:
+        return
+    for source in await list_enabled_sources_for_chat(chat_id):
+        await _ingest_source_message(source, message)
+
+
+async def _run_source_sync(source_id: str) -> None:
+    processed = 0
+    try:
+        source = await get_knowledge_source(source_id)
+        if not source:
+            return
+        await update_knowledge_source_sync(source_id, status="running", processed=0)
+        cutoff = datetime.now(UTC) - timedelta(days=int(source.get("backfill_days") or 0))
+        chat_ids = [int(source.get("chat_id") or 0)]
+        if source.get("include_linked_chat") and source.get("linked_chat_id"):
+            chat_ids.append(int(source["linked_chat_id"]))
+        limit = int(source.get("backfill_limit") or 100)
+        for chat_id in dict.fromkeys(item for item in chat_ids if item):
+            remaining = limit - processed
+            if remaining <= 0:
+                break
+            async for message in app.get_chat_history(chat_id, limit=remaining):
+                message_date = getattr(message, "date", None)
+                if isinstance(message_date, datetime):
+                    aware_date = (
+                        message_date.replace(tzinfo=UTC)
+                        if message_date.tzinfo is None
+                        else message_date.astimezone(UTC)
+                    )
+                    if int(source.get("backfill_days") or 0) and aware_date < cutoff:
+                        break
+                if await _ingest_source_message(source, message):
+                    processed += 1
+        await update_knowledge_source_sync(
+            source_id, status="completed", processed=processed
+        )
+    except asyncio.CancelledError:
+        raise
+    except Exception as exc:
+        with suppress(Exception):
+            await update_knowledge_source_sync(
+                source_id, status="failed", error=str(exc), processed=processed
+            )
+        wbb.log.error(f"知识来源历史同步失败:{source_id} / {exc}")
+
+
+async def schedule_knowledge_source_sync(source_id: str) -> dict[str, Any]:
+    if get_business_assistant_runtime() is None:
+        raise AssistantDataError(
+            "assistant_runtime_unavailable", "智能接待运行时尚未启动。"
+        )
+    source = await get_knowledge_source(source_id)
+    if not source:
+        raise AssistantDataError("source_not_found", "未找到知识来源。")
+    current = _source_sync_tasks.get(source_id)
+    if current and not current.done():
+        return source
+    queued = await update_knowledge_source_sync(source_id, status="queued", processed=0)
+    task = asyncio.create_task(
+        _run_source_sync(source_id), name=f"business-knowledge-sync-{source_id}"
+    )
+    _source_sync_tasks[source_id] = task
+    task.add_done_callback(lambda _task: _source_sync_tasks.pop(source_id, None))
+    return queued
+
+
+async def cancel_knowledge_source_sync(source_id: str) -> None:
+    task = _source_sync_tasks.pop(source_id, None)
+    if task and not task.done():
+        task.cancel()
+        await asyncio.gather(task, return_exceptions=True)
+
+
+async def regenerate_knowledge_candidate(candidate_id: str) -> list[dict[str, Any]]:
+    runtime = get_business_assistant_runtime()
+    if runtime is None:
+        raise AssistantDataError(
+            "assistant_runtime_unavailable", "智能接待运行时尚未启动。"
+        )
+    return await runtime.ingestion.regenerate_candidate(candidate_id)
+
+
+@app.on_message(filters.channel | filters.group, group=-30)
+async def collect_knowledge_source_message(_, message):
+    await process_knowledge_source_message(message)
+
+
+@app.on_edited_message(filters.channel | filters.group, group=-30)
+async def collect_edited_knowledge_source_message(_, message):
+    await process_knowledge_source_message(message)
+
+
+@app.on_deleted_messages(filters.channel | filters.group, group=-30)
+async def collect_deleted_knowledge_source_messages(_, messages):
+    for message in messages:
+        chat_id = int(getattr(getattr(message, "chat", None), "id", 0) or 0)
+        message_id = int(getattr(message, "id", 0) or 0)
+        if not chat_id or not message_id:
+            continue
+        for source in await list_sources_for_chat(chat_id, enabled_only=False):
+            await mark_source_event_deleted(
+                str(source["source_id"]), f"telegram:{chat_id}:{message_id}"
+            )
+
+
 async def _ops_group_admin_allowed(query, settings: dict) -> bool:
     ops_group_id = int(settings.get("ops_group_id") or 0)
     if not ops_group_id or not query.message or int(query.message.chat.id) != ops_group_id:

+ 330 - 6
wbb/services/business_assistant.py

@@ -20,15 +20,25 @@ from wbb.utils.dbassistant import (
     get_business_connection,
     get_conversation,
     get_conversation_by_chat,
+    get_knowledge_candidate,
+    get_knowledge_source,
     get_or_create_conversation,
+    get_source_event,
+    get_source_event_by_key,
+    invalidate_source_event_candidates,
     list_business_connections,
+    list_business_sources,
+    list_enabled_business_sources,
     load_update_offset,
     mark_digest_sent,
+    mark_source_event_deleted,
     mark_update_done,
     mark_update_failed,
     match_knowledge,
     pause_conversation_for_human,
     recent_conversation_messages,
+    record_source_event,
+    replace_event_candidates,
     reserve_ai_usage,
     resume_conversation,
     runtime_status,
@@ -37,6 +47,7 @@ from wbb.utils.dbassistant import (
     touch_business_connection,
     update_conversation_summary,
     update_runtime_status,
+    update_source_event_extraction,
     upsert_business_connection,
     usage_metrics,
     utc_now,
@@ -225,6 +236,63 @@ def parse_ai_decision(value: str, *, allowed_entry_ids: set[str]) -> dict[str, A
     }
 
 
+def _short_string_list(value: Any, *, items: int, length: int) -> list[str]:
+    raw_items = value if isinstance(value, list) else []
+    result: list[str] = []
+    for item in raw_items:
+        text = " ".join(str(item or "").strip().split())[:length]
+        if text and text.casefold() not in {existing.casefold() for existing in result}:
+            result.append(text)
+    return result[:items]
+
+
+def parse_knowledge_extraction(value: str) -> list[dict[str, Any]]:
+    try:
+        payload = json.loads(_strip_json_fence(value))
+    except (json.JSONDecodeError, TypeError) as exc:
+        raise AssistantProviderError("模型没有返回有效的知识提取 JSON。") from exc
+    if not isinstance(payload, dict) or not isinstance(payload.get("items"), list):
+        raise AssistantProviderError("知识提取结果必须包含 items 数组。")
+    result: list[dict[str, Any]] = []
+    for raw_item in payload["items"][:5]:
+        if not isinstance(raw_item, dict):
+            raise AssistantProviderError("知识候选必须是 JSON 对象。")
+        question = " ".join(str(raw_item.get("question") or "").strip().split())[:300]
+        answer = str(raw_item.get("answer") or "").strip()[:4000]
+        if not question or not answer:
+            raise AssistantProviderError("知识候选缺少标准问题或答案。")
+        try:
+            confidence = float(raw_item.get("confidence") or 0)
+        except (TypeError, ValueError):
+            confidence = 0
+        result.append(
+            {
+                "question": question,
+                "aliases": _short_string_list(
+                    raw_item.get("aliases"), items=20, length=200
+                ),
+                "keywords": _short_string_list(
+                    raw_item.get("keywords"), items=30, length=50
+                ),
+                "answer": answer,
+                "tags": _short_string_list(raw_item.get("tags"), items=20, length=30),
+                "confidence": max(0.0, min(confidence, 1.0)),
+            }
+        )
+    return result
+
+
+def redact_business_knowledge_text(value: str) -> str:
+    text = re.sub(
+        r"(?<![\w.])[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}(?!\w)",
+        "[邮箱已脱敏]",
+        str(value),
+        flags=re.IGNORECASE,
+    )
+    text = re.sub(r"(?<!\d)(?:\+?\d[\d\s-]{7,}\d)(?!\d)", "[电话已脱敏]", text)
+    return re.sub(r"(?<!\w)@[A-Za-z][A-Za-z0-9_]{4,31}", "[用户名已脱敏]", text)
+
+
 class OpenAICompatibleAssistant:
     def __init__(
         self,
@@ -324,6 +392,82 @@ class OpenAICompatibleAssistant:
             allowed_entry_ids={str(item["entry_id"]) for item in knowledge},
         )
 
+    async def extract_knowledge(
+        self,
+        *,
+        source_type: str,
+        source_title: str,
+        content: str,
+        question_context: str = "",
+    ) -> list[dict[str, Any]]:
+        if not self.configured:
+            raise AssistantProviderError("OpenAI 兼容模型尚未配置。")
+        if source_type == "business":
+            safe_content = redact_business_knowledge_text(content)
+            safe_question = redact_business_knowledge_text(question_context)
+            source_rule = (
+                "这是 Business 人工接待对话。customer_question 只能作为标准问题参考,"
+                "答案事实只能来自 human_reply。不得把客户自述、联系方式、用户名、"
+                "投诉、退款要求或未经确认的承诺写入知识。"
+            )
+            source_payload = {
+                "customer_question": safe_question,
+                "human_reply": safe_content,
+            }
+        else:
+            source_rule = (
+                "这是频道或群组内容。只提取明确、稳定、可复用的业务事实;"
+                "忽略寒暄、广告口号、个人观点、临时活动、投诉、退款争议和未经确认的承诺。"
+                "没有稳定事实时返回空 items。"
+            )
+            source_payload = {"content": content}
+        system_prompt = (
+            "你是知识库整理器,只提取输入中明确存在的事实,不得推测或补全。"
+            f"{source_rule}"
+            "只输出一个 JSON 对象,不使用 Markdown,格式为 "
+            '{"items":[{"question":"","aliases":[],"keywords":[],"answer":"",'
+            '"tags":[],"confidence":0.0}]}。items 最多 5 条。'
+        )
+        request_body = {
+            "model": self.model,
+            "temperature": 0.1,
+            "max_tokens": self.max_output_tokens,
+            "messages": [
+                {"role": "system", "content": system_prompt},
+                {
+                    "role": "user",
+                    "content": json.dumps(
+                        {
+                            "source_type": source_type,
+                            "source_title": source_title,
+                            **source_payload,
+                        },
+                        ensure_ascii=False,
+                    ),
+                },
+            ],
+        }
+        try:
+            async with asyncio.timeout(self.timeout_seconds):
+                async with self._session.post(
+                    f"{self.base_url}/chat/completions",
+                    headers={
+                        "Authorization": f"Bearer {self.api_key}",
+                        "Content-Type": "application/json",
+                    },
+                    json=request_body,
+                ) as response:
+                    payload = await response.json(content_type=None)
+        except (ClientError, TimeoutError, ValueError) as exc:
+            raise AssistantProviderError("知识提取模型暂时不可用。") from exc
+        if response.status != 200:
+            raise AssistantProviderError(f"知识提取模型返回 HTTP {response.status}。")
+        try:
+            result = payload["choices"][0]["message"]["content"]
+        except (KeyError, IndexError, TypeError) as exc:
+            raise AssistantProviderError("知识提取模型响应格式无效。") from exc
+        return parse_knowledge_extraction(str(result))
+
     async def test_connection(self) -> dict[str, Any]:
         if not self.configured:
             raise AssistantProviderError("OpenAI 兼容模型尚未配置。")
@@ -366,6 +510,75 @@ def provider_from_wbb(session: ClientSession) -> OpenAICompatibleAssistant:
     )
 
 
+class KnowledgeIngestionService:
+    def __init__(self, provider: OpenAICompatibleAssistant) -> None:
+        self.provider = provider
+
+    async def ingest(
+        self,
+        source: dict[str, Any],
+        *,
+        event_key: str,
+        content: str,
+        metadata: dict[str, Any],
+        question_context: str = "",
+        auto_publish: bool = False,
+        force: bool = False,
+    ) -> list[dict[str, Any]]:
+        event, changed = await record_source_event(
+            source,
+            event_key=event_key,
+            content=content,
+            metadata={
+                **metadata,
+                "question_context": question_context,
+                "trusted_author": bool(metadata.get("trusted_author")),
+            },
+        )
+        if not changed and not force:
+            return []
+        try:
+            items = await self.provider.extract_knowledge(
+                source_type=str(metadata.get("effective_source_type") or source["source_type"]),
+                source_title=str(source.get("title") or ""),
+                content=content,
+                question_context=question_context,
+            )
+            return await replace_event_candidates(
+                source, event, items, auto_publish=auto_publish
+            )
+        except Exception as exc:
+            await invalidate_source_event_candidates(
+                str(event["event_id"]), reason="extraction_failed"
+            )
+            await update_source_event_extraction(
+                str(event["event_id"]), status="failed", error=str(exc)
+            )
+            raise
+
+    async def regenerate_candidate(self, candidate_id: str) -> list[dict[str, Any]]:
+        candidate = await get_knowledge_candidate(candidate_id)
+        if not candidate:
+            raise AssistantProviderError("未找到知识候选。")
+        source = await get_knowledge_source(str(candidate["source_id"]))
+        event = await get_source_event(str(candidate["event_id"]))
+        if not source or not event or event.get("deleted"):
+            raise AssistantProviderError("知识来源或原始事件已失效。")
+        metadata = event.get("metadata") if isinstance(event.get("metadata"), dict) else {}
+        auto_publish = bool(
+            source.get("publication_mode") == "auto" and metadata.get("trusted_author")
+        )
+        return await self.ingest(
+            source,
+            event_key=str(event["event_key"]),
+            content=str(event.get("content") or ""),
+            metadata=metadata,
+            question_context=str(metadata.get("question_context") or ""),
+            auto_publish=auto_publish,
+            force=True,
+        )
+
+
 class BusinessAssistantRuntime:
     def __init__(
         self,
@@ -376,6 +589,7 @@ class BusinessAssistantRuntime:
     ) -> None:
         self.api = TelegramBusinessApi(token, session)
         self.provider = provider or provider_from_wbb(session)
+        self.ingestion = KnowledgeIngestionService(self.provider)
         self._poll_task: asyncio.Task[None] | None = None
         self._digest_task: asyncio.Task[None] | None = None
         self._stopping = asyncio.Event()
@@ -558,17 +772,27 @@ class BusinessAssistantRuntime:
         )
         if owner_reply:
             settings = await get_account_settings(connection_id)
+            human_text = str(message.get("text") or message.get("caption") or "").strip()
             await append_conversation_message(
                 conversation["conversation_id"],
                 direction="human",
                 telegram_message_id=message_id,
-                text=str(message.get("text") or message.get("caption") or ""),
+                text=human_text,
                 sender_id=owner_id,
             )
             await pause_conversation_for_human(
                 conversation["conversation_id"],
                 hours=int(settings.get("human_pause_hours") or 24),
             )
+            if human_text:
+                await self._learn_from_human_reply(
+                    connection_id=connection_id,
+                    conversation_id=str(conversation["conversation_id"]),
+                    chat_id=chat_id,
+                    message_id=message_id,
+                    owner_id=owner_id,
+                    human_text=human_text,
+                )
             return
         await self._handle_customer_message(
             connection=connection,
@@ -577,6 +801,72 @@ class BusinessAssistantRuntime:
             sender=sender,
         )
 
+    async def _learn_from_human_reply(
+        self,
+        *,
+        connection_id: str,
+        conversation_id: str,
+        chat_id: int,
+        message_id: int,
+        owner_id: int,
+        human_text: str,
+    ) -> None:
+        sources = await list_enabled_business_sources(connection_id)
+        if not sources:
+            return
+        messages = await recent_conversation_messages(conversation_id, limit=20)
+        customer_question = next(
+            (
+                str(item.get("text") or "")
+                for item in reversed(messages)
+                if item.get("direction") == "incoming"
+                and str(item.get("text") or "") != "[非文本消息]"
+            ),
+            "",
+        )
+        if not customer_question:
+            return
+        for source in sources:
+            event_key = f"business:{chat_id}:{message_id}"
+            existing_event = await get_source_event_by_key(
+                str(source["source_id"]), event_key
+            )
+            existing_metadata = (
+                existing_event.get("metadata")
+                if existing_event and isinstance(existing_event.get("metadata"), dict)
+                else {}
+            )
+            source_question = str(
+                existing_metadata.get("question_context") or customer_question
+            )
+            if any(
+                term in f"{source_question}\n{human_text}".casefold()
+                for term in ("承诺", "保证", "投诉", "退款", "退钱", "赔偿", "律师", "起诉")
+            ):
+                continue
+            try:
+                await self.ingestion.ingest(
+                    source,
+                    event_key=event_key,
+                    content=human_text,
+                    question_context=source_question,
+                    metadata={
+                        "effective_source_type": "business",
+                        "chat_id": int(chat_id),
+                        "message_id": int(message_id),
+                        "author_id": int(owner_id),
+                        "trusted_author": True,
+                    },
+                    auto_publish=source.get("publication_mode") == "auto",
+                )
+            except Exception as exc:
+                await update_runtime_status(
+                    {
+                        "knowledge_ingestion_last_error": str(exc)[:1000],
+                        "knowledge_ingestion_last_error_at": utc_now(),
+                    }
+                )
+
     async def _handle_customer_message(
         self,
         *,
@@ -792,19 +1082,48 @@ class BusinessAssistantRuntime:
     async def _handle_edited_message(self, message: dict[str, Any]) -> None:
         connection_id = str(message.get("business_connection_id") or "")
         chat_id = int((message.get("chat") or {}).get("id") or 0)
-        if not connection_id or not chat_id:
+        if not connection_id or not chat_id or message.get("sender_business_bot"):
             return
         await touch_business_connection(connection_id)
         conversation = await get_conversation_by_chat(connection_id, chat_id)
         if not conversation:
             return
+        connection = await self._resolve_connection(connection_id)
+        sender_id = int((message.get("from") or {}).get("id") or 0)
+        owner_id = int((connection.get("user") or {}).get("id") or 0)
+        message_id = int(message.get("message_id") or 0)
+        text = str(message.get("text") or message.get("caption") or "").strip()
+        if sender_id == owner_id:
+            settings = await get_account_settings(connection_id)
+            await append_conversation_message(
+                conversation["conversation_id"],
+                direction="human",
+                telegram_message_id=-message_id,
+                text=text or "[人工消息已编辑]",
+                sender_id=owner_id,
+                metadata={"edited_message_id": message_id},
+            )
+            await pause_conversation_for_human(
+                conversation["conversation_id"],
+                hours=int(settings.get("human_pause_hours") or 24),
+            )
+            if text:
+                await self._learn_from_human_reply(
+                    connection_id=connection_id,
+                    conversation_id=str(conversation["conversation_id"]),
+                    chat_id=chat_id,
+                    message_id=message_id,
+                    owner_id=owner_id,
+                    human_text=text,
+                )
+            return
         await append_conversation_message(
             conversation["conversation_id"],
             direction="incoming",
-            telegram_message_id=-int(message.get("message_id") or 0),
-            text=str(message.get("text") or "[消息已编辑]"),
-            sender_id=int((message.get("from") or {}).get("id") or 0),
-            metadata={"edited_message_id": int(message.get("message_id") or 0)},
+            telegram_message_id=-message_id,
+            text=text or "[消息已编辑]",
+            sender_id=sender_id,
+            metadata={"edited_message_id": message_id},
         )
 
     async def _handle_deleted_messages(self, payload: dict[str, Any]) -> None:
@@ -815,6 +1134,11 @@ class BusinessAssistantRuntime:
         conversation = await get_conversation_by_chat(connection_id, chat_id)
         if not conversation:
             return
+        for source in await list_business_sources(connection_id):
+            for message_id in payload.get("message_ids") or []:
+                await mark_source_event_deleted(
+                    str(source["source_id"]), f"business:{chat_id}:{int(message_id)}"
+                )
         await append_conversation_message(
             conversation["conversation_id"],
             direction="incoming",

+ 656 - 0
wbb/utils/dbassistant.py

@@ -2,6 +2,7 @@ from __future__ import annotations
 
 import asyncio
 import hashlib
+import json
 import re
 from datetime import UTC, datetime, timedelta
 from typing import Any
@@ -16,6 +17,9 @@ from wbb import BOT_PROFILE_ID, db
 connectionsdb = db.business_assistant_connections
 settingsdb = db.business_assistant_settings
 knowledgedb = db.business_assistant_knowledge
+knowledge_sourcesdb = db.business_assistant_knowledge_sources
+source_eventsdb = db.business_assistant_source_events
+knowledge_candidatesdb = db.business_assistant_knowledge_candidates
 conversationsdb = db.business_assistant_conversations
 messagesdb = db.business_assistant_messages
 usagedb = db.business_assistant_usage
@@ -44,6 +48,10 @@ DEFAULT_ACCOUNT_SETTINGS: dict[str, Any] = {
 
 VALID_NOTIFICATION_DESTINATIONS = {"owner", "ops", "both"}
 VALID_TONES = {"professional", "friendly", "concise"}
+VALID_SOURCE_TYPES = {"channel", "group", "business"}
+VALID_PUBLICATION_MODES = {"review", "auto"}
+VALID_AUTHOR_POLICIES = {"admins_or_allowlist", "admins", "allowlist", "all"}
+VALID_CANDIDATE_STATUSES = {"pending", "published", "rejected", "stale"}
 HANDOFF_TERMS = (
     "人工",
     "真人",
@@ -142,6 +150,23 @@ def _normalize_string_list(value: Any, *, max_items: int, max_length: int) -> li
     return normalized
 
 
+def _normalize_int_list(value: Any, *, max_items: int = 100) -> list[int]:
+    values = value if isinstance(value, (list, tuple, set)) else str(value or "").split(",")
+    normalized: list[int] = []
+    for item in values:
+        if item in {None, ""}:
+            continue
+        try:
+            parsed = int(item)
+        except (TypeError, ValueError) as exc:
+            raise AssistantDataError("invalid_source", "白名单用户 ID 必须是整数。") from exc
+        if parsed and parsed not in normalized:
+            normalized.append(parsed)
+    if len(normalized) > max_items:
+        raise AssistantDataError("too_many_items", f"最多允许 {max_items} 项。")
+    return normalized
+
+
 async def ensure_assistant_indexes() -> None:
     global _indexes_ready
     if _indexes_ready:
@@ -161,6 +186,51 @@ async def ensure_assistant_indexes() -> None:
         await knowledgedb.create_index(
             [("bot_id", ASCENDING), ("entry_id", ASCENDING)], unique=True
         )
+        await knowledge_sourcesdb.create_index(
+            [("bot_id", ASCENDING), ("source_id", ASCENDING)], unique=True
+        )
+        await knowledge_sourcesdb.create_index(
+            [("bot_id", ASCENDING), ("source_key", ASCENDING)], unique=True
+        )
+        await knowledge_sourcesdb.create_index(
+            [
+                ("bot_id", ASCENDING),
+                ("connection_id", ASCENDING),
+                ("enabled", ASCENDING),
+                ("updated_at", DESCENDING),
+            ]
+        )
+        await source_eventsdb.create_index(
+            [("bot_id", ASCENDING), ("event_id", ASCENDING)], unique=True
+        )
+        await source_eventsdb.create_index(
+            [
+                ("bot_id", ASCENDING),
+                ("source_id", ASCENDING),
+                ("event_key", ASCENDING),
+            ],
+            unique=True,
+        )
+        await source_eventsdb.create_index("expires_at", expireAfterSeconds=0)
+        await knowledge_candidatesdb.create_index(
+            [("bot_id", ASCENDING), ("candidate_id", ASCENDING)], unique=True
+        )
+        await knowledge_candidatesdb.create_index(
+            [
+                ("bot_id", ASCENDING),
+                ("event_id", ASCENDING),
+                ("item_index", ASCENDING),
+            ],
+            unique=True,
+        )
+        await knowledge_candidatesdb.create_index(
+            [
+                ("bot_id", ASCENDING),
+                ("connection_id", ASCENDING),
+                ("status", ASCENDING),
+                ("updated_at", DESCENDING),
+            ]
+        )
         await knowledgedb.create_index(
             [
                 ("bot_id", ASCENDING),
@@ -432,6 +502,12 @@ async def create_knowledge_entry(
             values.get("priority", 0), "知识优先级", minimum=-1000, maximum=1000
         ),
         "enabled": bool(values.get("enabled", True)),
+        "source_candidate_id": clean_text(
+            values.get("source_candidate_id"), max_length=64
+        ),
+        "source_references": values.get("source_references")
+        if isinstance(values.get("source_references"), list)
+        else [],
         "created_at": now,
         "updated_at": now,
     }
@@ -467,6 +543,15 @@ async def update_knowledge_entry(entry_id: str, values: dict[str, Any]) -> dict[
         )
     if "enabled" in values:
         update["enabled"] = bool(values.get("enabled"))
+    if "source_candidate_id" in values:
+        update["source_candidate_id"] = clean_text(
+            values.get("source_candidate_id"), max_length=64
+        )
+    if "source_references" in values:
+        references = values.get("source_references")
+        if not isinstance(references, list):
+            raise AssistantDataError("invalid_knowledge", "知识来源引用必须是数组。")
+        update["source_references"] = references[:20]
     if not update:
         raise AssistantDataError("unchanged", "没有可保存的知识条目字段。")
     update["updated_at"] = utc_now()
@@ -510,6 +595,577 @@ async def list_knowledge_entries(
     return [item async for item in cursor], total
 
 
+def _knowledge_source_key(connection_id: str, source_type: str, chat_id: int) -> str:
+    return f"{connection_id}:{source_type}:{int(chat_id)}"
+
+
+def normalize_knowledge_source(
+    values: dict[str, Any], *, previous: dict[str, Any] | None = None
+) -> dict[str, Any]:
+    current = {**(previous or {}), **values}
+    source_type = str(current.get("source_type") or "")
+    if source_type not in VALID_SOURCE_TYPES:
+        raise AssistantDataError("invalid_source", "知识来源类型无效。")
+    try:
+        chat_id = int(current.get("chat_id") or 0)
+    except (TypeError, ValueError) as exc:
+        raise AssistantDataError("invalid_source", "来源聊天 ID 必须是整数。") from exc
+    if source_type == "business":
+        chat_id = 0
+    elif not chat_id:
+        raise AssistantDataError("invalid_source", "频道或群组来源必须填写聊天 ID。")
+    publication_mode = str(current.get("publication_mode") or "review")
+    if publication_mode not in VALID_PUBLICATION_MODES:
+        raise AssistantDataError("invalid_source", "知识发布方式无效。")
+    author_policy = str(current.get("author_policy") or "admins_or_allowlist")
+    if author_policy not in VALID_AUTHOR_POLICIES:
+        raise AssistantDataError("invalid_source", "来源作者策略无效。")
+    linked_chat_id = 0
+    try:
+        linked_chat_id = int(current.get("linked_chat_id") or 0)
+    except (TypeError, ValueError) as exc:
+        raise AssistantDataError("invalid_source", "关联讨论群 ID 必须是整数。") from exc
+    return {
+        "source_type": source_type,
+        "chat_id": chat_id,
+        "title": clean_text(
+            current.get("title") or ("Business 日常对话" if source_type == "business" else ""),
+            max_length=200,
+            required=True,
+        ),
+        "publication_mode": publication_mode,
+        "author_policy": author_policy,
+        "allowed_user_ids": _normalize_int_list(current.get("allowed_user_ids")),
+        "include_linked_chat": bool(current.get("include_linked_chat", False))
+        if source_type == "channel"
+        else False,
+        "linked_chat_id": linked_chat_id if source_type == "channel" else 0,
+        "linked_chat_title": clean_text(
+            current.get("linked_chat_title"), max_length=200
+        )
+        if source_type == "channel"
+        else "",
+        "backfill_days": _bounded_int(
+            current.get("backfill_days", 30), "历史回溯天数", minimum=0, maximum=365
+        ),
+        "backfill_limit": _bounded_int(
+            current.get("backfill_limit", 100), "历史回溯条数", minimum=1, maximum=500
+        ),
+        "enabled": bool(current.get("enabled", True)),
+        "access_status": clean_text(current.get("access_status"), max_length=64),
+    }
+
+
+async def create_knowledge_source(
+    connection_id: str, values: dict[str, Any]
+) -> dict[str, Any]:
+    await ensure_assistant_indexes()
+    if not await get_business_connection(connection_id):
+        raise AssistantDataError("connection_not_found", "未找到 Business 连接。")
+    normalized = normalize_knowledge_source(values)
+    now = utc_now()
+    source = {
+        "bot_id": BOT_PROFILE_ID,
+        "source_id": uuid4().hex,
+        "source_key": _knowledge_source_key(
+            str(connection_id), normalized["source_type"], normalized["chat_id"]
+        ),
+        "connection_id": str(connection_id),
+        **normalized,
+        "sync_status": "idle",
+        "last_error": "",
+        "created_at": now,
+        "updated_at": now,
+    }
+    try:
+        await knowledge_sourcesdb.insert_one(source)
+    except DuplicateKeyError as exc:
+        raise AssistantDataError("source_exists", "该连接已配置相同的知识来源。") from exc
+    return source
+
+
+async def get_knowledge_source(source_id: str) -> dict[str, Any] | None:
+    await ensure_assistant_indexes()
+    return await knowledge_sourcesdb.find_one(_scope(source_id=str(source_id)))
+
+
+async def update_knowledge_source(
+    source_id: str, values: dict[str, Any]
+) -> dict[str, Any]:
+    current = await get_knowledge_source(source_id)
+    if not current:
+        raise AssistantDataError("source_not_found", "未找到知识来源。")
+    for field in ("connection_id", "source_type", "chat_id"):
+        if field in values and str(values[field]) != str(current[field]):
+            raise AssistantDataError("immutable_source", "来源连接、类型和聊天 ID 不可修改。")
+    normalized = normalize_knowledge_source(values, previous=current)
+    await knowledge_sourcesdb.update_one(
+        _scope(source_id=str(source_id)),
+        {"$set": {**normalized, "updated_at": utc_now()}},
+    )
+    return await get_knowledge_source(source_id) or current
+
+
+async def list_knowledge_sources(
+    connection_id: str,
+    *,
+    page: int = 1,
+    page_size: int = 20,
+) -> tuple[list[dict[str, Any]], int]:
+    await ensure_assistant_indexes()
+    filters = _scope(connection_id=str(connection_id))
+    page = max(1, int(page))
+    page_size = max(1, min(int(page_size), 100))
+    total = await knowledge_sourcesdb.count_documents(filters)
+    cursor = (
+        knowledge_sourcesdb.find(filters)
+        .sort("updated_at", DESCENDING)
+        .skip((page - 1) * page_size)
+        .limit(page_size)
+    )
+    return [item async for item in cursor], total
+
+
+async def list_sources_for_chat(
+    chat_id: int, *, enabled_only: bool = True
+) -> list[dict[str, Any]]:
+    await ensure_assistant_indexes()
+    values: dict[str, Any] = {
+        "$or": [
+            {"chat_id": int(chat_id)},
+            {"include_linked_chat": True, "linked_chat_id": int(chat_id)},
+        ]
+    }
+    if enabled_only:
+        values["enabled"] = True
+    filters = _scope(values)
+    return [item async for item in knowledge_sourcesdb.find(filters)]
+
+
+async def list_enabled_sources_for_chat(chat_id: int) -> list[dict[str, Any]]:
+    return await list_sources_for_chat(chat_id, enabled_only=True)
+
+
+async def list_enabled_business_sources(connection_id: str) -> list[dict[str, Any]]:
+    await ensure_assistant_indexes()
+    return [
+        item
+        async for item in knowledge_sourcesdb.find(
+            _scope(
+                connection_id=str(connection_id),
+                source_type="business",
+                enabled=True,
+            )
+        )
+    ]
+
+
+async def list_business_sources(connection_id: str) -> list[dict[str, Any]]:
+    await ensure_assistant_indexes()
+    return [
+        item
+        async for item in knowledge_sourcesdb.find(
+            _scope(connection_id=str(connection_id), source_type="business")
+        )
+    ]
+
+
+async def update_knowledge_source_sync(
+    source_id: str,
+    *,
+    status: str,
+    error: str = "",
+    processed: int | None = None,
+) -> dict[str, Any]:
+    values: dict[str, Any] = {
+        "sync_status": clean_text(status, max_length=32, required=True),
+        "last_error": clean_text(error, max_length=1000),
+        "updated_at": utc_now(),
+    }
+    if status == "running":
+        values["last_sync_started_at"] = utc_now()
+    if status in {"completed", "failed"}:
+        values["last_sync_finished_at"] = utc_now()
+    if processed is not None:
+        values["last_sync_processed"] = max(0, int(processed))
+    result = await knowledge_sourcesdb.update_one(
+        _scope(source_id=str(source_id)), {"$set": values}
+    )
+    if not result.matched_count:
+        raise AssistantDataError("source_not_found", "未找到知识来源。")
+    return await get_knowledge_source(source_id) or {}
+
+
+async def touch_knowledge_source(source_id: str, *, error: str = "") -> None:
+    await knowledge_sourcesdb.update_one(
+        _scope(source_id=str(source_id)),
+        {
+            "$set": {
+                "last_event_at": utc_now(),
+                "last_error": clean_text(error, max_length=1000),
+                "updated_at": utc_now(),
+            }
+        },
+    )
+
+
+async def _disable_candidate_entries(filters: dict[str, Any]) -> None:
+    entry_ids = [
+        str(item.get("knowledge_entry_id") or "")
+        async for item in knowledge_candidatesdb.find(filters)
+        if item.get("knowledge_entry_id")
+    ]
+    if entry_ids:
+        await knowledgedb.update_many(
+            _scope({"entry_id": {"$in": entry_ids}}),
+            {"$set": {"enabled": False, "updated_at": utc_now()}},
+        )
+
+
+async def delete_knowledge_source(source_id: str) -> None:
+    source = await get_knowledge_source(source_id)
+    if not source:
+        raise AssistantDataError("source_not_found", "未找到知识来源。")
+    candidate_filters = _scope(source_id=str(source_id))
+    await _disable_candidate_entries(candidate_filters)
+    await knowledge_candidatesdb.update_many(
+        candidate_filters,
+        {"$set": {"status": "stale", "stale_reason": "source_deleted", "updated_at": utc_now()}},
+    )
+    await source_eventsdb.update_many(
+        _scope(source_id=str(source_id)),
+        {"$set": {"deleted": True, "delete_reason": "source_deleted", "updated_at": utc_now()}},
+    )
+    await knowledge_sourcesdb.delete_one(_scope(source_id=str(source_id)))
+
+
+def source_event_id_for(source_id: str, event_key: str) -> str:
+    digest = hashlib.sha256(
+        f"{BOT_PROFILE_ID}:{source_id}:{event_key}".encode()
+    ).hexdigest()
+    return digest[:32]
+
+
+async def record_source_event(
+    source: dict[str, Any],
+    *,
+    event_key: str,
+    content: str,
+    metadata: dict[str, Any] | None = None,
+) -> tuple[dict[str, Any], bool]:
+    await ensure_assistant_indexes()
+    cleaned_content = clean_text(
+        content, max_length=20000, required=True, preserve_lines=True
+    )
+    cleaned_key = clean_text(event_key, max_length=300, required=True)
+    event_metadata = metadata if isinstance(metadata, dict) else {}
+    content_hash = hashlib.sha256(
+        json.dumps(
+            {"content": cleaned_content, "metadata": event_metadata},
+            ensure_ascii=False,
+            sort_keys=True,
+            default=str,
+        ).encode()
+    ).hexdigest()
+    filters = _scope(source_id=str(source["source_id"]), event_key=cleaned_key)
+    current = await source_eventsdb.find_one(filters)
+    if current and current.get("content_hash") == content_hash and not current.get("deleted"):
+        return current, False
+    now = utc_now()
+    event_id = str(current.get("event_id")) if current else source_event_id_for(
+        str(source["source_id"]), cleaned_key
+    )
+    version = int(current.get("version") or 0) + 1 if current else 1
+    document = {
+        "bot_id": BOT_PROFILE_ID,
+        "event_id": event_id,
+        "source_id": str(source["source_id"]),
+        "connection_id": str(source["connection_id"]),
+        "event_key": cleaned_key,
+        "content": cleaned_content,
+        "content_hash": content_hash,
+        "metadata": event_metadata,
+        "version": version,
+        "deleted": False,
+        "extraction_status": "pending",
+        "extraction_error": "",
+        "updated_at": now,
+        "expires_at": now + timedelta(days=30),
+    }
+    await source_eventsdb.update_one(
+        filters,
+        {"$set": document, "$setOnInsert": {"created_at": now}},
+        upsert=True,
+    )
+    return await source_eventsdb.find_one(filters) or document, True
+
+
+async def get_source_event(event_id: str) -> dict[str, Any] | None:
+    await ensure_assistant_indexes()
+    return await source_eventsdb.find_one(_scope(event_id=str(event_id)))
+
+
+async def get_source_event_by_key(
+    source_id: str, event_key: str
+) -> dict[str, Any] | None:
+    await ensure_assistant_indexes()
+    return await source_eventsdb.find_one(
+        _scope(source_id=str(source_id), event_key=str(event_key))
+    )
+
+
+async def update_source_event_extraction(
+    event_id: str, *, status: str, error: str = "", item_count: int = 0
+) -> None:
+    await source_eventsdb.update_one(
+        _scope(event_id=str(event_id)),
+        {
+            "$set": {
+                "extraction_status": clean_text(status, max_length=32, required=True),
+                "extraction_error": clean_text(error, max_length=1000),
+                "extracted_item_count": max(0, int(item_count)),
+                "extracted_at": utc_now(),
+                "updated_at": utc_now(),
+            }
+        },
+    )
+
+
+async def invalidate_source_event_candidates(
+    event_id: str, *, reason: str
+) -> None:
+    filters = _scope(event_id=str(event_id))
+    await _disable_candidate_entries(filters)
+    await knowledge_candidatesdb.update_many(
+        filters,
+        {
+            "$set": {
+                "status": "stale",
+                "stale_reason": clean_text(reason, max_length=100),
+                "updated_at": utc_now(),
+            }
+        },
+    )
+
+
+async def mark_source_event_deleted(source_id: str, event_key: str) -> bool:
+    event = await source_eventsdb.find_one(
+        _scope(source_id=str(source_id), event_key=str(event_key))
+    )
+    if not event:
+        return False
+    await source_eventsdb.update_one(
+        _scope(event_id=str(event["event_id"])),
+        {"$set": {"deleted": True, "updated_at": utc_now()}},
+    )
+    await invalidate_source_event_candidates(
+        str(event["event_id"]), reason="source_message_deleted"
+    )
+    return True
+
+
+def _normalize_candidate_item(item: dict[str, Any]) -> dict[str, Any]:
+    try:
+        confidence = float(item.get("confidence") or 0)
+    except (TypeError, ValueError):
+        confidence = 0
+    return {
+        "question": clean_text(item.get("question"), max_length=300, required=True),
+        "aliases": _normalize_string_list(item.get("aliases"), max_items=20, max_length=200),
+        "keywords": _normalize_string_list(item.get("keywords"), max_items=30, max_length=50),
+        "answer": clean_text(
+            item.get("answer"), max_length=4000, required=True, preserve_lines=True
+        ),
+        "tags": _normalize_string_list(item.get("tags"), max_items=20, max_length=30),
+        "confidence": max(0.0, min(confidence, 1.0)),
+    }
+
+
+async def replace_event_candidates(
+    source: dict[str, Any],
+    event: dict[str, Any],
+    items: list[dict[str, Any]],
+    *,
+    auto_publish: bool,
+) -> list[dict[str, Any]]:
+    await invalidate_source_event_candidates(
+        str(event["event_id"]), reason="source_message_changed"
+    )
+    existing = {
+        int(item.get("item_index") or 0): item
+        async for item in knowledge_candidatesdb.find(
+            _scope(event_id=str(event["event_id"]))
+        )
+    }
+    results: list[dict[str, Any]] = []
+    metadata = event.get("metadata") if isinstance(event.get("metadata"), dict) else {}
+    for index, raw_item in enumerate(items[:5]):
+        normalized = _normalize_candidate_item(raw_item)
+        current = existing.get(index) or {}
+        now = utc_now()
+        candidate_id = str(current.get("candidate_id") or uuid4().hex)
+        candidate = {
+            "bot_id": BOT_PROFILE_ID,
+            "candidate_id": candidate_id,
+            "connection_id": str(source["connection_id"]),
+            "source_id": str(source["source_id"]),
+            "event_id": str(event["event_id"]),
+            "item_index": index,
+            **normalized,
+            "status": "pending",
+            "stale_reason": "",
+            "knowledge_entry_id": str(current.get("knowledge_entry_id") or ""),
+            "source_snapshot": {
+                "title": str(source.get("title") or ""),
+                "source_type": str(source.get("source_type") or ""),
+                "chat_id": int(metadata.get("chat_id") or source.get("chat_id") or 0),
+                "message_id": int(metadata.get("message_id") or 0),
+                "author_id": int(metadata.get("author_id") or 0),
+            },
+            "updated_at": now,
+        }
+        await knowledge_candidatesdb.update_one(
+            _scope(event_id=str(event["event_id"]), item_index=index),
+            {"$set": candidate, "$setOnInsert": {"created_at": now}},
+            upsert=True,
+        )
+        if auto_publish:
+            candidate = await publish_knowledge_candidate(candidate_id)
+        else:
+            candidate = await get_knowledge_candidate(candidate_id) or candidate
+        results.append(candidate)
+    await update_source_event_extraction(
+        str(event["event_id"]), status="completed", item_count=len(results)
+    )
+    return results
+
+
+async def get_knowledge_candidate(candidate_id: str) -> dict[str, Any] | None:
+    await ensure_assistant_indexes()
+    return await knowledge_candidatesdb.find_one(
+        _scope(candidate_id=str(candidate_id))
+    )
+
+
+async def publish_knowledge_candidate(candidate_id: str) -> dict[str, Any]:
+    candidate = await get_knowledge_candidate(candidate_id)
+    if not candidate:
+        raise AssistantDataError("candidate_not_found", "未找到知识候选。")
+    if candidate.get("status") == "stale":
+        raise AssistantDataError("candidate_stale", "来源已变更或删除,不能发布该候选。")
+    if not await get_knowledge_source(str(candidate.get("source_id") or "")):
+        await knowledge_candidatesdb.update_one(
+            _scope(candidate_id=str(candidate_id)),
+            {
+                "$set": {
+                    "status": "stale",
+                    "stale_reason": "source_deleted",
+                    "updated_at": utc_now(),
+                }
+            },
+        )
+        raise AssistantDataError("source_not_found", "知识来源已删除,不能发布该候选。")
+    reference = {
+        **(candidate.get("source_snapshot") or {}),
+        "source_id": str(candidate.get("source_id") or ""),
+        "event_id": str(candidate.get("event_id") or ""),
+    }
+    values = {
+        key: candidate.get(key)
+        for key in ("question", "aliases", "keywords", "answer", "tags")
+    }
+    values.update(
+        {
+            "priority": round(float(candidate.get("confidence") or 0) * 100),
+            "enabled": True,
+            "source_candidate_id": str(candidate_id),
+            "source_references": [reference],
+        }
+    )
+    entry_id = str(candidate.get("knowledge_entry_id") or "")
+    if entry_id:
+        try:
+            entry = await update_knowledge_entry(entry_id, values)
+        except AssistantDataError as exc:
+            if exc.code != "knowledge_not_found":
+                raise
+            entry = await create_knowledge_entry(str(candidate["connection_id"]), values)
+    else:
+        entry = await create_knowledge_entry(str(candidate["connection_id"]), values)
+    await knowledge_candidatesdb.update_one(
+        _scope(candidate_id=str(candidate_id)),
+        {
+            "$set": {
+                "status": "published",
+                "knowledge_entry_id": str(entry["entry_id"]),
+                "published_at": utc_now(),
+                "updated_at": utc_now(),
+            }
+        },
+    )
+    return await get_knowledge_candidate(candidate_id) or candidate
+
+
+async def reject_knowledge_candidate(candidate_id: str) -> dict[str, Any]:
+    candidate = await get_knowledge_candidate(candidate_id)
+    if not candidate:
+        raise AssistantDataError("candidate_not_found", "未找到知识候选。")
+    entry_id = str(candidate.get("knowledge_entry_id") or "")
+    if entry_id:
+        await knowledgedb.update_one(
+            _scope(entry_id=entry_id),
+            {"$set": {"enabled": False, "updated_at": utc_now()}},
+        )
+    await knowledge_candidatesdb.update_one(
+        _scope(candidate_id=str(candidate_id)),
+        {
+            "$set": {
+                "status": "rejected",
+                "rejected_at": utc_now(),
+                "updated_at": utc_now(),
+            }
+        },
+    )
+    return await get_knowledge_candidate(candidate_id) or candidate
+
+
+async def list_knowledge_candidates(
+    connection_id: str,
+    *,
+    status: str = "",
+    source_id: str = "",
+    query: str = "",
+    page: int = 1,
+    page_size: int = 20,
+) -> tuple[list[dict[str, Any]], int]:
+    await ensure_assistant_indexes()
+    filters: dict[str, Any] = _scope(connection_id=str(connection_id))
+    if status:
+        if status not in VALID_CANDIDATE_STATUSES:
+            raise AssistantDataError("invalid_status", "知识候选状态无效。")
+        filters["status"] = status
+    if source_id:
+        filters["source_id"] = str(source_id)
+    normalized_query = clean_text(query, max_length=100)
+    if normalized_query:
+        pattern = re.escape(normalized_query)
+        filters["$or"] = [
+            {"question": {"$regex": pattern, "$options": "i"}},
+            {"answer": {"$regex": pattern, "$options": "i"}},
+            {"keywords": {"$regex": pattern, "$options": "i"}},
+        ]
+    page = max(1, int(page))
+    page_size = max(1, min(int(page_size), 100))
+    total = await knowledge_candidatesdb.count_documents(filters)
+    cursor = (
+        knowledge_candidatesdb.find(filters)
+        .sort("updated_at", DESCENDING)
+        .skip((page - 1) * page_size)
+        .limit(page_size)
+    )
+    return [item async for item in cursor], total
+
+
 def _search_text(value: str) -> str:
     return re.sub(r"[^\w\u3400-\u9fff]+", "", value.casefold())