feat(knowledge): support editable chunks and metadata

This commit is contained in:
wizardchen
2026-07-28 20:23:48 +08:00
committed by lyingbug
parent 002434bbb3
commit ac3eccddf5
30 changed files with 1506 additions and 55 deletions
+40
View File
@@ -333,6 +333,35 @@ export function getKnowledgeDetailsCon(id: string, page: number) {
return get(`/api/v1/chunks/${id}?page=${page}&page_size=25`);
}
export interface ChunkEditPayload {
content?: string;
is_enabled?: boolean;
expected_revision?: number;
}
export function updateDocumentChunk(knowledgeId: string, chunkId: string, data: ChunkEditPayload) {
return put(`/api/v1/chunks/${knowledgeId}/${chunkId}`, data);
}
export function listChunkRevisions(knowledgeId: string, chunkId: string) {
return get(`/api/v1/chunks/${knowledgeId}/${chunkId}/revisions`);
}
export function revertDocumentChunk(knowledgeId: string, chunkId: string, revision: number, expectedRevision: number) {
return post(`/api/v1/chunks/${knowledgeId}/${chunkId}/revert`, {
revision,
expected_revision: expectedRevision,
});
}
export function updateKnowledgeMetadata(knowledgeId: string, customMetadata: Record<string, unknown>) {
return put(`/api/v1/knowledge/${knowledgeId}`, { custom_metadata: customMetadata });
}
export function regenerateKnowledgeSummary(knowledgeId: string) {
return post(`/api/v1/knowledge/${knowledgeId}/regenerate-summary`, {});
}
// Get chunk by chunk_id only (new endpoint - to be added to backend)
export function getChunkByIdOnly(chunkId: string) {
return get(`/api/v1/chunks/by-id/${chunkId}`);
@@ -343,6 +372,17 @@ export function deleteGeneratedQuestion(chunkId: string, questionId: string) {
return del(`/api/v1/chunks/by-id/${chunkId}/questions`, { question_id: questionId });
}
export function upsertGeneratedQuestion(chunkId: string, question: string, questionId?: string) {
return put(`/api/v1/chunks/by-id/${chunkId}/questions`, {
question_id: questionId || '',
question,
});
}
export function regenerateGeneratedQuestions(chunkId: string) {
return post(`/api/v1/chunks/by-id/${chunkId}/questions/regenerate`, {});
}
export function listKnowledgeTags(
kbId: string,
params?: { page?: number; page_size?: number; keyword?: string },
+270 -3
View File
@@ -8,7 +8,11 @@ import hljs from "highlight.js";
import "highlight.js/styles/github.css";
import mermaid from "mermaid";
import { onMounted, ref, nextTick, onUnmounted, watch, computed } from "vue";
import { downKnowledgeDetails, deleteGeneratedQuestion, getChunkByIdOnly, previewKnowledgeFile } from "@/api/knowledge-base/index";
import {
downKnowledgeDetails, deleteGeneratedQuestion, getChunkByIdOnly, previewKnowledgeFile,
updateDocumentChunk, listChunkRevisions, revertDocumentChunk, updateKnowledgeMetadata,
regenerateKnowledgeSummary, upsertGeneratedQuestion, regenerateGeneratedQuestions,
} from "@/api/knowledge-base/index";
import { MessagePlugin, DialogPlugin } from "tdesign-vue-next";
import { sanitizeHTML, safeMarkdownToHTML, createSafeImage, isValidImageURL, hydrateProtectedFileImages, isValidURL } from '@/utils/security';
import { normalizeSpuriousTablePrefixes } from '@/utils/markdownTableNormalize';
@@ -30,6 +34,49 @@ const canDeleteGeneratedQuestion = computed(() => {
if (props.canEditKB === true) return true;
return authStore.hasRole('admin');
});
const canEditContent = canDeleteGeneratedQuestion;
const metadataEditing = ref(false);
const metadataDraft = ref('{}');
const metadataSaving = ref(false);
const summaryRefreshing = ref(false);
const syncMetadataDraft = () => {
metadataDraft.value = JSON.stringify(props.details?.custom_metadata || {}, null, 2);
};
const saveMetadata = async () => {
try {
const value = JSON.parse(metadataDraft.value || '{}');
if (!value || Array.isArray(value) || typeof value !== 'object') throw new Error(t('knowledgeBase.metadataObjectRequired'));
metadataSaving.value = true;
await updateKnowledgeMetadata(props.details.id, value);
props.details.custom_metadata = value;
metadataEditing.value = false;
props.details.summary_status = props.details.description ? 'pending' : props.details.summary_status;
MessagePlugin.success(t('common.saveSuccess'));
} catch (error: any) {
MessagePlugin.error(error?.message || t('common.saveFailed'));
} finally {
metadataSaving.value = false;
}
};
const refreshSummary = async () => {
summaryRefreshing.value = true;
try {
const result: any = await regenerateKnowledgeSummary(props.details.id);
if (result?.data) {
props.details.description = result.data.description || '';
props.details.summary_status = result.data.summary_status || 'completed';
}
MessagePlugin.success(t('knowledgeBase.summaryRefreshed'));
} catch (error: any) {
MessagePlugin.error(error?.message || t('common.error'));
} finally {
summaryRefreshing.value = false;
}
};
const detailTags = computed(() => {
const tags = props.details?.tags;
@@ -86,6 +133,7 @@ mermaid.initialize({
});
const props = defineProps(["visible", "details", "knowledgeType", "sourceInfo", "canEditKB", "canDownloadKB", "parse_status", "kbId"]);
const emit = defineEmits(["closeDoc", "getDoc", "questionDeleted"]);
watch(() => props.details?.id, syncMetadataDraft, { immediate: true });
const hasTimelineSpans = ref(false);
const timelineDrawerVisible = ref(false);
@@ -892,6 +940,149 @@ const toggleQuestions = (index: number) => {
const isExpanded = (index: number) => expandedChunks.value.has(index);
const editingChunkId = ref('');
const chunkDraft = ref('');
const savingChunkId = ref('');
const startChunkEdit = (item: any) => {
editingChunkId.value = item.id;
chunkDraft.value = item.content || '';
};
const saveChunkEdit = async (item: any) => {
if (!chunkDraft.value.trim()) {
MessagePlugin.warning(t('knowledgeBase.chunkContentRequired'));
return;
}
savingChunkId.value = item.id;
try {
const result: any = await updateDocumentChunk(props.details.id, item.id, {
content: chunkDraft.value,
expected_revision: item.content_revision || 0,
});
Object.assign(item, result.data);
editingChunkId.value = '';
props.details.summary_status = props.details.description ? 'pending' : props.details.summary_status;
MessagePlugin.success(t('common.saveSuccess'));
} catch (error: any) {
MessagePlugin.error(error?.message || t('common.saveFailed'));
} finally {
savingChunkId.value = '';
}
};
const toggleChunkEnabled = async (item: any) => {
try {
const result: any = await updateDocumentChunk(props.details.id, item.id, {
is_enabled: !item.is_enabled,
expected_revision: item.content_revision || 0,
});
Object.assign(item, result.data);
} catch (error: any) {
MessagePlugin.error(error?.message || t('common.error'));
}
};
const retryChunkIndex = async (item: any) => {
try {
const result: any = await updateDocumentChunk(props.details.id, item.id, {
expected_revision: item.content_revision || 0,
});
Object.assign(item, result.data);
if (item.index_status === 'failed') throw new Error(t('knowledgeBase.indexFailed'));
MessagePlugin.success(t('knowledgeBase.indexRetrySuccess'));
} catch (error: any) {
MessagePlugin.error(error?.message || t('common.error'));
}
};
const showChunkHistory = async (item: any) => {
try {
const result: any = await listChunkRevisions(props.details.id, item.id);
const revisions = result?.data || [];
const body = revisions.length
? revisions.map((r: any) => `v${r.revision} · ${new Date(r.edited_at).toLocaleString()}\n${r.content.slice(0, 240)}`).join('\n\n')
: t('knowledgeBase.noChunkHistory');
if (!revisions.length) {
DialogPlugin.alert({ header: t('knowledgeBase.chunkHistory'), body });
return;
}
const dialog = DialogPlugin.confirm({
header: t('knowledgeBase.chunkHistory'), body,
confirmBtn: t('knowledgeBase.revertRevision'), cancelBtn: t('common.close'),
onConfirm: async () => {
const value = window.prompt(t('knowledgeBase.enterRevision'), String(revisions[0].revision));
if (value !== null && Number.isInteger(Number(value))) await revertChunk(item, Number(value));
dialog.hide();
},
onClose: () => dialog.hide(),
});
} catch (error: any) {
MessagePlugin.error(error?.message || t('common.error'));
}
};
const revertChunk = async (item: any, revision: number) => {
try {
const result: any = await revertDocumentChunk(props.details.id, item.id, revision, item.content_revision || 0);
Object.assign(item, result.data);
MessagePlugin.success(t('knowledgeBase.chunkReverted'));
} catch (error: any) {
MessagePlugin.error(error?.message || t('common.error'));
}
};
const questionDrafts = ref<Record<string, string>>({});
const savingQuestionChunk = ref('');
const regeneratingQuestionChunk = ref('');
const addQuestion = async (item: any) => {
const question = (questionDrafts.value[item.id] || '').trim();
if (!question) return;
savingQuestionChunk.value = item.id;
try {
const result: any = await upsertGeneratedQuestion(item.id, question);
questionDrafts.value[item.id] = '';
const metadata = typeof item.metadata === 'string' ? JSON.parse(item.metadata || '{}') : (item.metadata || {});
metadata.generated_questions = [...(metadata.generated_questions || []), result.data];
metadata.generated_questions_revision = item.content_revision || 0;
item.metadata = metadata;
MessagePlugin.success(t('common.saveSuccess'));
} catch (error: any) {
MessagePlugin.error(error?.message || t('common.error'));
} finally {
savingQuestionChunk.value = '';
}
};
const editQuestion = async (item: any, question: GeneratedQuestion) => {
const value = window.prompt(t('knowledgeBase.editGeneratedQuestion'), question.question);
if (!value || value.trim() === question.question) return;
try {
await upsertGeneratedQuestion(item.id, value.trim(), question.id);
question.question = value.trim();
MessagePlugin.success(t('common.saveSuccess'));
} catch (error: any) {
MessagePlugin.error(error?.message || t('common.error'));
}
};
const regenerateQuestions = async (item: any) => {
regeneratingQuestionChunk.value = item.id;
try {
const result: any = await regenerateGeneratedQuestions(item.id);
const metadata = typeof item.metadata === 'string' ? JSON.parse(item.metadata || '{}') : (item.metadata || {});
metadata.generated_questions = result?.data || [];
metadata.generated_questions_revision = item.content_revision || 0;
item.metadata = metadata;
MessagePlugin.success(t('knowledgeBase.questionsRegenerated'));
} catch (error: any) {
MessagePlugin.error(error?.message || t('common.error'));
} finally {
regeneratingQuestionChunk.value = '';
}
};
// 删除中的状态
const deletingQuestion = ref<{ chunkIndex: number; questionId: string } | null>(null);
@@ -1174,6 +1365,27 @@ const handleDetailsScroll = () => {
</t-tag>
</span>
</div>
<div class="doc-detail-row doc-metadata-row">
<span class="doc-detail-label">{{ $t('knowledgeBase.customMetadata') }}</span>
<span class="doc-detail-value metadata-value">
<template v-if="!metadataEditing">
<span v-if="Object.keys(details.custom_metadata || {}).length" class="metadata-preview">
{{ Object.entries(details.custom_metadata || {}).map(([key, value]) => `${key}: ${value}`).join(' · ') }}
</span>
<span v-else class="metadata-empty">{{ $t('knowledgeBase.noCustomMetadata') }}</span>
<t-button v-if="canEditContent" size="small" variant="text" @click="metadataEditing = true; syncMetadataDraft()">
{{ $t('common.edit') }}
</t-button>
</template>
<template v-else>
<t-textarea v-model="metadataDraft" :autosize="{ minRows: 3, maxRows: 8 }" placeholder='{"department":"R&D"}' />
<div class="metadata-actions">
<t-button size="small" theme="primary" :loading="metadataSaving" @click="saveMetadata">{{ $t('common.save') }}</t-button>
<t-button size="small" variant="text" @click="metadataEditing = false">{{ $t('common.cancel') }}</t-button>
</div>
</template>
</span>
</div>
</div>
</section>
@@ -1190,7 +1402,12 @@ const handleDetailsScroll = () => {
</section>
<section v-if="showSummarySection" class="setting-drawer__section">
<h4 class="setting-drawer__section-title">{{ $t('knowledgeBase.documentSummary') }}</h4>
<div class="section-title-actions">
<h4 class="setting-drawer__section-title">{{ $t('knowledgeBase.documentSummary') }}</h4>
<t-button v-if="canEditContent" size="small" variant="text" :loading="summaryRefreshing" @click="refreshSummary">
{{ $t('knowledgeBase.regenerateSummary') }}
</t-button>
</div>
<div v-if="details.description" class="summary_wrapper"
:class="{ 'summary_clickable': summaryOverflow || summaryExpanded }"
@click="(summaryOverflow || summaryExpanded) && (summaryExpanded = !summaryExpanded)">
@@ -1267,9 +1484,24 @@ const handleDetailsScroll = () => {
{{ $t('knowledgeBase.questions') }} {{ chunk.questions.length }}
</t-tag>
<span class="chunk-meta">{{ chunk.meta }}</span>
<t-button v-if="chunk.original.index_status === 'failed' && canEditContent" size="small" theme="danger" variant="text" @click="retryChunkIndex(chunk.original)">{{ $t('knowledgeBase.retryIndex') }}</t-button>
<template v-if="canEditContent">
<t-button size="small" variant="text" @click="startChunkEdit(chunk.original)">{{ $t('common.edit') }}</t-button>
<t-button size="small" variant="text" @click="showChunkHistory(chunk.original)">{{ $t('knowledgeBase.history') }}</t-button>
<t-button size="small" variant="text" @click="toggleChunkEnabled(chunk.original)">
{{ chunk.original.is_enabled ? $t('knowledgeBase.disableChunk') : $t('knowledgeBase.enableChunk') }}
</t-button>
</template>
</div>
</div>
<div class="md-content" v-html="chunk.processedContent"></div>
<div v-if="editingChunkId === chunk.original.id" class="chunk-editor">
<t-textarea v-model="chunkDraft" :autosize="{ minRows: 6, maxRows: 20 }" />
<div class="chunk-editor-actions">
<t-button size="small" theme="primary" :loading="savingChunkId === chunk.original.id" @click="saveChunkEdit(chunk.original)">{{ $t('common.save') }}</t-button>
<t-button size="small" variant="text" @click="editingChunkId = ''">{{ $t('common.cancel') }}</t-button>
</div>
</div>
<div v-else class="md-content" :class="{ 'chunk-disabled': !chunk.original.is_enabled }" v-html="chunk.processedContent"></div>
<!-- 父 Chunk 上下文展开 -->
<div v-if="chunk.hasParent" class="parent-context-section">
@@ -1294,6 +1526,10 @@ const handleDetailsScroll = () => {
<div v-for="question in chunk.questions" :key="question.id" class="question-item">
<t-icon name="help-circle" size="14px" class="question-icon" />
<span class="question-text">{{ question.question }}</span>
<t-button v-if="canEditContent && !question.id.startsWith('legacy-')" theme="default" variant="text" size="small"
@click.stop="editQuestion(chunk.original, question)">
<template #icon><t-icon name="edit" size="14px" /></template>
</t-button>
<t-button v-if="canDeleteGeneratedQuestion" theme="default" variant="text" size="small"
class="delete-question-btn" :loading="isDeleting(index, question.id)"
@click.stop="handleDeleteQuestion(chunk.original, index, question)">
@@ -1302,6 +1538,18 @@ const handleDetailsScroll = () => {
</template>
</t-button>
</div>
<div v-if="canEditContent" class="question-add-row">
<t-input v-model="questionDrafts[chunk.original.id]" :placeholder="$t('knowledgeBase.addGeneratedQuestion')" @enter="addQuestion(chunk.original)" />
<t-button size="small" theme="primary" :loading="savingQuestionChunk === chunk.original.id" @click="addQuestion(chunk.original)">{{ $t('common.add') }}</t-button>
<t-button size="small" variant="outline" :loading="regeneratingQuestionChunk === chunk.original.id" @click="regenerateQuestions(chunk.original)">{{ $t('knowledgeBase.regenerateQuestions') }}</t-button>
</div>
</div>
</div>
<div v-else-if="canEditContent" class="questions-section empty-questions">
<div class="question-add-row">
<t-input v-model="questionDrafts[chunk.original.id]" :placeholder="$t('knowledgeBase.addGeneratedQuestion')" @enter="addQuestion(chunk.original)" />
<t-button size="small" theme="primary" :loading="savingQuestionChunk === chunk.original.id" @click="addQuestion(chunk.original)">{{ $t('common.add') }}</t-button>
<t-button size="small" variant="outline" :loading="regeneratingQuestionChunk === chunk.original.id" @click="regenerateQuestions(chunk.original)">{{ $t('knowledgeBase.regenerateQuestions') }}</t-button>
</div>
</div>
</div>
@@ -1322,6 +1570,25 @@ const handleDetailsScroll = () => {
<style scoped lang="less">
@import "./css/markdown.less";
.section-title-actions,
.metadata-actions,
.chunk-editor-actions,
.question-add-row {
display: flex;
align-items: center;
gap: 8px;
}
.section-title-actions { justify-content: space-between; }
.metadata-value { flex: 1; min-width: 0; }
.metadata-preview { color: var(--td-text-color-secondary); word-break: break-word; }
.metadata-empty { color: var(--td-text-color-placeholder); }
.metadata-actions, .chunk-editor-actions { margin-top: 8px; justify-content: flex-end; }
.chunk-editor { margin: 8px 0; }
.chunk-disabled { opacity: .5; }
.question-add-row { margin-top: 8px; }
.question-add-row :deep(.t-input__wrap) { flex: 1; }
/* Drawer widths are now driven by the `:size` prop on each <t-drawer>
(see mainDrawerSize / timelineDrawerSize in <script>). CSS rules with
!important were removed because they fought each other across the
+3
View File
@@ -34,6 +34,7 @@ export default function (knowledgeBaseId?: string) {
summary_status: "",
parse_status: "",
error_message: "",
custom_metadata: {} as Record<string, unknown>,
chunkLoading: false,
chunkLoadError: "",
tags: [] as Array<{ id: string; name: string; color?: string }>,
@@ -189,6 +190,7 @@ export default function (knowledgeBaseId?: string) {
summary_status: "",
parse_status: "",
error_message: "",
custom_metadata: {},
chunkLoadError: "",
tags: item?.tags ? [...item.tags] : [],
});
@@ -208,6 +210,7 @@ export default function (knowledgeBaseId?: string) {
summary_status: data.summary_status || '',
parse_status: data.parse_status || '',
error_message: data.error_message || '',
custom_metadata: data.custom_metadata || {},
tags: data.tags?.length ? data.tags : (item?.tags || []),
});
}
+22
View File
@@ -529,6 +529,27 @@ export default {
parentContextLoadFailed: 'Failed to load parent context',
confirmDeleteQuestion: 'Are you sure you want to delete this question? The corresponding vector index will also be removed.',
legacyQuestionCannotDelete: 'Legacy format questions cannot be deleted. Please regenerate questions.',
customMetadata: 'Custom metadata',
noCustomMetadata: 'No custom metadata',
metadataObjectRequired: 'Metadata must be a JSON object',
regenerateSummary: 'Regenerate summary',
summaryRefreshed: 'Summary refreshed',
indexFailed: 'Index sync failed',
retryIndex: 'Retry index',
indexRetrySuccess: 'Index synchronized',
history: 'History',
chunkHistory: 'Chunk edit history',
noChunkHistory: 'No edit history',
revertRevision: 'Revert revision',
enterRevision: 'Enter the revision to restore',
chunkReverted: 'Chunk reverted',
chunkContentRequired: 'Chunk content is required',
disableChunk: 'Disable',
enableChunk: 'Enable',
addGeneratedQuestion: 'Add retrieval question',
editGeneratedQuestion: 'Edit retrieval question',
regenerateQuestions: 'Regenerate questions',
questionsRegenerated: 'Retrieval questions refreshed',
notInitialized: 'Knowledge base is not initialized. Please configure models in settings before uploading files',
missingStorageEngine: 'This knowledge base has no storage engine selected. Please configure a storage engine in settings before uploading content.',
missingStorageEngineUpload: 'Please configure a storage engine before uploading content',
@@ -2094,6 +2115,7 @@ export default {
}
},
common: {
add: 'Add',
me: 'Me',
confirm: 'Confirm',
cancel: 'Cancel',
+22
View File
@@ -539,6 +539,27 @@ export default {
confirmDeleteQuestion:
"이 질문을 삭제하시겠습니까? 삭제 시 해당 벡터 인덱스도 함께 제거됩니다.",
legacyQuestionCannotDelete: "이전 형식의 질문은 삭제할 수 없습니다. 질문을 다시 생성하세요",
customMetadata: "사용자 정의 메타데이터",
noCustomMetadata: "사용자 정의 메타데이터 없음",
metadataObjectRequired: "메타데이터는 JSON 객체여야 합니다",
regenerateSummary: "요약 다시 생성",
summaryRefreshed: "요약이 업데이트되었습니다",
indexFailed: "인덱스 동기화 실패",
retryIndex: "인덱스 재시도",
indexRetrySuccess: "인덱스가 동기화되었습니다",
history: "기록",
chunkHistory: "청크 편집 기록",
noChunkHistory: "편집 기록 없음",
revertRevision: "버전 되돌리기",
enterRevision: "복원할 버전을 입력하세요",
chunkReverted: "청크가 복원되었습니다",
chunkContentRequired: "청크 내용은 필수입니다",
disableChunk: "비활성화",
enableChunk: "활성화",
addGeneratedQuestion: "검색 보조 질문 추가",
editGeneratedQuestion: "검색 보조 질문 편집",
regenerateQuestions: "질문 다시 생성",
questionsRegenerated: "검색 질문이 업데이트되었습니다",
docActionUnsupported: "현재 지식베이스 유형은 이 작업을 지원하지 않습니다",
notInitialized:
"이 지식베이스는 아직 초기화되지 않았습니다. 설정 페이지에서 모델 정보를 먼저 구성한 후 파일을 업로드하세요",
@@ -1953,6 +1974,7 @@ export default {
},
},
common: {
add: "추가",
me: "나",
confirm: "확인",
cancel: "취소",
+22
View File
@@ -464,6 +464,27 @@ export default {
parentContextLoadFailed: 'Не удалось загрузить родительский контекст',
confirmDeleteQuestion: 'Вы уверены, что хотите удалить этот вопрос? Соответствующий векторный индекс также будет удален.',
legacyQuestionCannotDelete: 'Вопросы в устаревшем формате нельзя удалить. Пожалуйста, сгенерируйте вопросы заново.',
customMetadata: 'Пользовательские метаданные',
noCustomMetadata: 'Нет пользовательских метаданных',
metadataObjectRequired: 'Метаданные должны быть JSON-объектом',
regenerateSummary: 'Пересоздать сводку',
summaryRefreshed: 'Сводка обновлена',
indexFailed: 'Ошибка синхронизации индекса',
retryIndex: 'Повторить индексацию',
indexRetrySuccess: 'Индекс синхронизирован',
history: 'История',
chunkHistory: 'История изменений чанка',
noChunkHistory: 'История изменений пуста',
revertRevision: 'Откатить версию',
enterRevision: 'Введите номер версии',
chunkReverted: 'Чанк восстановлен',
chunkContentRequired: 'Содержимое чанка обязательно',
disableChunk: 'Отключить',
enableChunk: 'Включить',
addGeneratedQuestion: 'Добавить вопрос для поиска',
editGeneratedQuestion: 'Изменить вопрос для поиска',
regenerateQuestions: 'Пересоздать вопросы',
questionsRegenerated: 'Вопросы для поиска обновлены',
notInitialized: 'База знаний не инициализирована. Пожалуйста, настройте модели в разделе настроек перед загрузкой файлов',
missingStorageEngine: 'Для этой базы знаний не выбрано хранилище. Пожалуйста, настройте хранилище в параметрах перед загрузкой содержимого.',
missingStorageEngineUpload: 'Пожалуйста, настройте хранилище перед загрузкой содержимого',
@@ -1964,6 +1985,7 @@ export default {
}
},
common: {
add: 'Добавить',
confirm: 'Подтвердить',
cancel: 'Отмена',
save: 'Сохранить',
+22
View File
@@ -528,6 +528,27 @@ export default {
parentContextLoadFailed: "加载父上下文失败",
confirmDeleteQuestion: "确定要删除这个问题吗?删除后将同时移除对应的向量索引。",
legacyQuestionCannotDelete: "旧格式问题无法删除,请重新生成问题",
customMetadata: "自定义元数据",
noCustomMetadata: "暂无自定义元数据",
metadataObjectRequired: "元数据必须是 JSON 对象",
regenerateSummary: "重新生成摘要",
summaryRefreshed: "摘要已更新",
indexFailed: "索引同步失败",
retryIndex: "重试索引",
indexRetrySuccess: "索引已同步",
history: "历史",
chunkHistory: "分块编辑历史",
noChunkHistory: "暂无编辑历史",
revertRevision: "回滚版本",
enterRevision: "输入要回滚的版本号",
chunkReverted: "分块已回滚",
chunkContentRequired: "分块内容不能为空",
disableChunk: "停用",
enableChunk: "启用",
addGeneratedQuestion: "添加辅助召回问题",
editGeneratedQuestion: "编辑辅助召回问题",
regenerateQuestions: "重新生成问题",
questionsRegenerated: "辅助召回问题已更新",
docActionUnsupported: "当前知识库类型不支持该操作",
notInitialized:
"该知识库尚未完成初始化配置,请先前往设置页面配置模型信息后再上传文件",
@@ -1959,6 +1980,7 @@ export default {
},
},
common: {
add: "添加",
me: "我",
confirm: "确认",
cancel: "取消",
+4
View File
@@ -1274,6 +1274,9 @@ func (t *KnowledgeSearchTool) formatOutput(
if snippet != "" {
ob.WriteString(fmt.Sprintf("<match_snippet>%s</match_snippet>\n", xmlEscape(snippet)))
}
if result.KnowledgeCustomMetadata != "" {
ob.WriteString(fmt.Sprintf("<metadata>%s</metadata>\n", xmlEscape(result.KnowledgeCustomMetadata)))
}
ob.WriteString(fmt.Sprintf("<content>%s</content>\n", result.Content))
if result.ImageInfo != "" {
@@ -1302,6 +1305,7 @@ func (t *KnowledgeSearchTool) formatOutput(
"knowledge_id": result.KnowledgeID,
"knowledge_base_id": result.KnowledgeBaseID,
"knowledge_title": result.KnowledgeTitle,
"knowledge_metadata": result.KnowledgeCustomMetadata,
"match_type": result.MatchType,
"source_query": result.SourceQuery,
"query_type": result.QueryType,
+58
View File
@@ -14,6 +14,8 @@ import (
"gorm.io/gorm"
)
var ErrChunkRevisionConflict = errors.New("chunk revision conflict")
// ErrChunkNotFound is returned when a chunk lookup finds no row. A typed
// sentinel (matching the ErrXNotFound convention used by the other repos)
// so callers can errors.Is it safely through wrapping — replacing the
@@ -40,6 +42,12 @@ func NewChunkRepository(db *gorm.DB) interfaces.ChunkRepository {
func (r *chunkRepository) CreateChunks(ctx context.Context, chunks []*types.Chunk) error {
for _, chunk := range chunks {
chunk.Content = common.CleanInvalidUTF8(chunk.Content)
if chunk.SourceContent == "" {
chunk.SourceContent = chunk.Content
}
if chunk.IndexStatus == "" {
chunk.IndexStatus = "ready"
}
}
db := r.db.WithContext(ctx)
@@ -298,6 +306,56 @@ func (r *chunkRepository) UpdateChunk(ctx context.Context, chunk *types.Chunk) e
return r.db.WithContext(ctx).Omit("SeqID").Save(chunk).Error
}
func (r *chunkRepository) CreateChunkRevision(ctx context.Context, revision *types.ChunkRevision) error {
return r.db.WithContext(ctx).Create(revision).Error
}
func (r *chunkRepository) SaveChunkRevision(
ctx context.Context, chunk *types.Chunk, revision *types.ChunkRevision, expectedRevision int,
) error {
return r.db.WithContext(ctx).Transaction(func(tx *gorm.DB) error {
result := tx.Model(&types.Chunk{}).
Where("id = ? AND tenant_id = ? AND content_revision = ?", chunk.ID, chunk.TenantID, expectedRevision).
Updates(map[string]interface{}{
"content": common.CleanInvalidUTF8(chunk.Content),
"source_content": common.CleanInvalidUTF8(chunk.SourceContent),
"content_revision": chunk.ContentRevision,
"is_enabled": chunk.IsEnabled,
"metadata": chunk.Metadata,
"index_status": chunk.IndexStatus,
"last_editor_id": chunk.LastEditorID,
"updated_at": chunk.UpdatedAt,
})
if result.Error != nil {
return result.Error
}
if result.RowsAffected != 1 {
return ErrChunkRevisionConflict
}
return tx.Create(revision).Error
})
}
func (r *chunkRepository) ListChunkRevisions(
ctx context.Context, tenantID uint64, chunkID string,
) ([]*types.ChunkRevision, error) {
var revisions []*types.ChunkRevision
err := r.db.WithContext(ctx).
Where("tenant_id = ? AND chunk_id = ?", tenantID, chunkID).
Order("revision DESC").Find(&revisions).Error
return revisions, err
}
func (r *chunkRepository) GetChunkRevision(
ctx context.Context, tenantID uint64, chunkID string, revision int,
) (*types.ChunkRevision, error) {
var item types.ChunkRevision
err := r.db.WithContext(ctx).
Where("tenant_id = ? AND chunk_id = ? AND revision = ?", tenantID, chunkID, revision).
First(&item).Error
return &item, err
}
// SaveChunks persists full chunk objects in a single transaction using GORM Save (UPDATE).
func (r *chunkRepository) SaveChunks(ctx context.Context, chunks []*types.Chunk) error {
if len(chunks) == 0 {
@@ -0,0 +1,62 @@
package repository
import (
"context"
"errors"
"testing"
"time"
"github.com/Tencent/WeKnora/internal/types"
"github.com/google/uuid"
"github.com/stretchr/testify/require"
"gorm.io/driver/sqlite"
"gorm.io/gorm"
)
func TestSaveChunkRevisionIsAtomicAndOptimistic(t *testing.T) {
db, err := gorm.Open(sqlite.Open("file:"+uuid.NewString()+"?mode=memory&cache=shared"), &gorm.Config{})
require.NoError(t, err)
require.NoError(t, db.AutoMigrate(&types.Chunk{}, &types.ChunkRevision{}))
repo := NewChunkRepository(db)
ctx := context.Background()
now := time.Now()
chunk := &types.Chunk{
ID: uuid.NewString(), TenantID: 1, KnowledgeBaseID: "kb", KnowledgeID: "knowledge",
Content: "before", SourceContent: "before", ChunkType: types.ChunkTypeText,
IsEnabled: true, IndexStatus: "ready", CreatedAt: now, UpdatedAt: now,
}
require.NoError(t, repo.CreateChunks(ctx, []*types.Chunk{chunk}))
snapshot := &types.ChunkRevision{
ID: uuid.NewString(), TenantID: 1, KnowledgeBaseID: "kb", KnowledgeID: "knowledge",
ChunkID: chunk.ID, Revision: 0, Content: "before", IsEnabled: true,
EditSource: "user", EditedAt: now, CreatedAt: now,
}
chunk.Content = "after"
chunk.ContentRevision = 1
require.NoError(t, repo.SaveChunkRevision(ctx, chunk, snapshot, 0))
stored, err := repo.GetChunkByID(ctx, 1, chunk.ID)
require.NoError(t, err)
require.Equal(t, "after", stored.Content)
require.Equal(t, 1, stored.ContentRevision)
revisions, err := repo.ListChunkRevisions(ctx, 1, chunk.ID)
require.NoError(t, err)
require.Len(t, revisions, 1)
require.Equal(t, "before", revisions[0].Content)
stale := *chunk
stale.Content = "stale write"
stale.ContentRevision = 1
staleSnapshot := *snapshot
staleSnapshot.ID = uuid.NewString()
require.ErrorIs(t, repo.SaveChunkRevision(ctx, &stale, &staleSnapshot, 0), ErrChunkRevisionConflict)
stored, err = repo.GetChunkByID(ctx, 1, chunk.ID)
require.NoError(t, err)
require.Equal(t, "after", stored.Content)
count := int64(0)
require.NoError(t, db.Model(&types.ChunkRevision{}).Count(&count).Error)
require.Equal(t, int64(1), count)
require.False(t, errors.Is(gorm.ErrRecordNotFound, ErrChunkRevisionConflict))
}
+7 -1
View File
@@ -188,7 +188,13 @@ func (r *knowledgeRepository) ListPagedKnowledgeByKnowledgeBaseID(
// UpdateKnowledge updates knowledge
func (r *knowledgeRepository) UpdateKnowledge(ctx context.Context, knowledge *types.Knowledge) error {
err := r.db.WithContext(ctx).Omit(omitFieldsOnUpdate...).Save(knowledge).Error
omit := omitFieldsOnUpdate
// Legacy/unit-test schemas created before custom_metadata should continue
// to support unrelated updates when the caller did not provide the field.
if knowledge.CustomMetadata == nil {
omit = append(append([]string{}, omitFieldsOnUpdate...), "custom_metadata")
}
err := r.db.WithContext(ctx).Omit(omit...).Save(knowledge).Error
return err
}
@@ -240,6 +240,7 @@ func buildDocumentHeader(results []*types.SearchResult) string {
type docMeta struct {
title string
description string
metadata string
}
seen := make(map[string]struct{})
@@ -265,6 +266,7 @@ func buildDocumentHeader(results []*types.SearchResult) string {
docs = append(docs, docMeta{
title: title,
description: r.KnowledgeDescription,
metadata: r.KnowledgeCustomMetadata,
})
}
@@ -280,6 +282,9 @@ func buildDocumentHeader(results []*types.SearchResult) string {
if d.description != "" {
b.WriteString(fmt.Sprintf("<description>%s</description>\n", d.description))
}
if d.metadata != "" {
b.WriteString(fmt.Sprintf("<metadata>%s</metadata>\n", d.metadata))
}
b.WriteString("</document>\n")
}
b.WriteString("</documents>")
+309
View File
@@ -7,18 +7,26 @@ import (
"context"
"errors"
"fmt"
"sort"
"strings"
"time"
"github.com/Tencent/WeKnora/internal/application/repository"
"github.com/Tencent/WeKnora/internal/application/service/retriever"
"github.com/Tencent/WeKnora/internal/logger"
"github.com/Tencent/WeKnora/internal/types"
"github.com/Tencent/WeKnora/internal/types/interfaces"
"github.com/google/uuid"
)
var ErrChunkRevisionConflict = repository.ErrChunkRevisionConflict
// chunkService implements the ChunkService interface
// It provides operations for managing document chunks in the knowledge base
// Chunks are segments of documents that have been processed and prepared for indexing
type chunkService struct {
chunkRepository interfaces.ChunkRepository // Repository for chunk data persistence
knowledgeRepo interfaces.KnowledgeRepository
kbRepository interfaces.KnowledgeBaseRepository
modelService interfaces.ModelService
retrieveEngine interfaces.RetrieveEngineRegistry
@@ -34,6 +42,7 @@ type chunkService struct {
// - interfaces.ChunkService: Initialized chunk service implementation
func NewChunkService(
chunkRepository interfaces.ChunkRepository,
knowledgeRepo interfaces.KnowledgeRepository,
kbRepository interfaces.KnowledgeBaseRepository,
modelService interfaces.ModelService,
retrieveEngine interfaces.RetrieveEngineRegistry,
@@ -41,6 +50,7 @@ func NewChunkService(
) interfaces.ChunkService {
return &chunkService{
chunkRepository: chunkRepository,
knowledgeRepo: knowledgeRepo,
kbRepository: kbRepository,
modelService: modelService,
retrieveEngine: retrieveEngine,
@@ -48,6 +58,8 @@ func NewChunkService(
}
}
const maxEditableChunkLength = 200000
// GetRepository gets the chunk repository
// Parameters:
// - ctx: Context with authentication and request information
@@ -354,6 +366,303 @@ func (s *chunkService) ListChunkByParentID(
return chunks, nil
}
// UpdateDocumentChunk applies an optimistic, versioned edit. Retrieval
// questions are invalidated when the body changes because they describe the
// previous revision. The current row remains saved when reindexing fails and
// exposes index_status=failed so the UI never presents a false success state.
func (s *chunkService) UpdateDocumentChunk(
ctx context.Context, chunkID string, content *string, isEnabled *bool, expectedRevision *int,
) (*types.Chunk, error) {
tenantID := types.MustTenantIDFromContext(ctx)
chunk, err := s.chunkRepository.GetChunkByID(ctx, tenantID, chunkID)
if err != nil {
return nil, err
}
if chunk.ChunkType != types.ChunkTypeText {
return nil, fmt.Errorf("only text chunks can be edited")
}
if expectedRevision != nil && *expectedRevision != chunk.ContentRevision {
return nil, ErrChunkRevisionConflict
}
newContent := chunk.Content
if content != nil {
newContent = strings.TrimSpace(*content)
if newContent == "" {
return nil, fmt.Errorf("chunk content cannot be empty")
}
if len(newContent) > maxEditableChunkLength {
return nil, fmt.Errorf("chunk content exceeds %d bytes", maxEditableChunkLength)
}
}
newEnabled := chunk.IsEnabled
if isEnabled != nil {
newEnabled = *isEnabled
}
if newContent == chunk.Content && newEnabled == chunk.IsEnabled {
if chunk.IndexStatus == "failed" {
chunk.IndexStatus = "processing"
_ = s.chunkRepository.UpdateChunk(ctx, chunk)
if err := s.syncChunkIndex(ctx, chunk); err != nil {
chunk.IndexStatus = "failed"
_ = s.chunkRepository.UpdateChunk(ctx, chunk)
return chunk, nil
}
chunk.IndexStatus = "ready"
if err := s.chunkRepository.UpdateChunk(ctx, chunk); err != nil {
return nil, err
}
}
return chunk, nil
}
actorID, _ := types.UserIDFromContext(ctx)
now := time.Now()
oldRevision := chunk.ContentRevision
revision := &types.ChunkRevision{
ID: uuid.NewString(), TenantID: chunk.TenantID,
KnowledgeBaseID: chunk.KnowledgeBaseID, KnowledgeID: chunk.KnowledgeID,
ChunkID: chunk.ID, Revision: oldRevision, Content: chunk.Content,
IsEnabled: chunk.IsEnabled, EditorID: chunk.LastEditorID,
EditSource: "user", EditedAt: chunk.UpdatedAt, CreatedAt: now,
}
if chunk.SourceContent == "" {
chunk.SourceContent = chunk.Content
}
bodyChanged := newContent != chunk.Content
chunk.Content = newContent
chunk.IsEnabled = newEnabled
chunk.ContentRevision++
chunk.LastEditorID = actorID
chunk.IndexStatus = "processing"
chunk.UpdatedAt = now
if bodyChanged {
meta, metaErr := chunk.DocumentMetadata()
if metaErr != nil {
return nil, fmt.Errorf("parse chunk metadata: %w", metaErr)
}
if meta != nil {
meta.GeneratedQuestions = nil
if err := chunk.SetDocumentMetadata(meta); err != nil {
return nil, err
}
}
}
if err := s.chunkRepository.SaveChunkRevision(ctx, chunk, revision, oldRevision); err != nil {
return nil, err
}
if bodyChanged && chunk.ParentChunkID != "" {
if err := s.rebuildParentContent(ctx, chunk); err != nil {
logger.Warnf(ctx, "Failed to rebuild parent chunk after edit: %v", err)
}
}
if bodyChanged {
knowledge, getErr := s.knowledgeRepo.GetKnowledgeByID(ctx, tenantID, chunk.KnowledgeID)
if getErr == nil && knowledge.SummaryStatus == types.SummaryStatusCompleted {
_ = s.knowledgeRepo.UpdateKnowledgeColumn(ctx, knowledge.ID, "summary_status", types.SummaryStatusPending)
}
}
if err := s.syncChunkIndex(ctx, chunk); err != nil {
chunk.IndexStatus = "failed"
_ = s.chunkRepository.UpdateChunk(ctx, chunk)
logger.Errorf(ctx, "Chunk %s saved but reindex failed: %v", chunk.ID, err)
return chunk, nil
}
chunk.IndexStatus = "ready"
if err := s.chunkRepository.UpdateChunk(ctx, chunk); err != nil {
return nil, err
}
return chunk, nil
}
func (s *chunkService) ListChunkRevisions(ctx context.Context, chunkID string) ([]*types.ChunkRevision, error) {
return s.chunkRepository.ListChunkRevisions(ctx, types.MustTenantIDFromContext(ctx), chunkID)
}
func (s *chunkService) RevertDocumentChunk(
ctx context.Context, chunkID string, revision int, expectedRevision *int,
) (*types.Chunk, error) {
item, err := s.chunkRepository.GetChunkRevision(ctx, types.MustTenantIDFromContext(ctx), chunkID, revision)
if err != nil {
return nil, err
}
content := item.Content
enabled := item.IsEnabled
return s.UpdateDocumentChunk(ctx, chunkID, &content, &enabled, expectedRevision)
}
// rebuildParentContent overlays manually edited child ranges on the immutable
// parent source. Applying replacements in reverse offset order preserves the
// parser coordinate system even when edited text changes length.
func (s *chunkService) rebuildParentContent(ctx context.Context, edited *types.Chunk) error {
parent, err := s.chunkRepository.GetChunkByID(ctx, edited.TenantID, edited.ParentChunkID)
if err != nil {
return err
}
children, err := s.chunkRepository.ListChunkByParentID(ctx, edited.TenantID, parent.ID)
if err != nil {
return err
}
base := parent.SourceContent
if base == "" {
base = parent.Content
parent.SourceContent = base
}
baseRunes := []rune(base)
type replacement struct {
start, end int
content string
updatedAt time.Time
}
replacements := make([]replacement, 0)
for _, child := range children {
if child.ContentRevision == 0 {
continue
}
start, end := child.StartAt-parent.StartAt, child.EndAt-parent.StartAt
if start >= 0 && end >= start && end <= len(baseRunes) {
replacements = append(replacements, replacement{start, end, child.Content, child.UpdatedAt})
}
}
// Overlapping child windows cannot both be projected after arbitrary text
// replacement. Keep the most recently edited window and ignore older
// overlaps so the parent remains deterministic.
sort.Slice(replacements, func(i, j int) bool { return replacements[i].updatedAt.After(replacements[j].updatedAt) })
selected := make([]replacement, 0, len(replacements))
for _, candidate := range replacements {
overlaps := false
for _, existing := range selected {
if candidate.start < existing.end && candidate.end > existing.start {
overlaps = true
break
}
}
if !overlaps {
selected = append(selected, candidate)
}
}
replacements = selected
sort.Slice(replacements, func(i, j int) bool { return replacements[i].start > replacements[j].start })
for _, repl := range replacements {
baseRunes = append(append(append([]rune{}, baseRunes[:repl.start]...), []rune(repl.content)...), baseRunes[repl.end:]...)
}
parent.Content = string(baseRunes)
parent.UpdatedAt = time.Now()
return s.chunkRepository.UpdateChunk(ctx, parent)
}
func (s *chunkService) syncChunkIndex(ctx context.Context, chunk *types.Chunk) error {
kb, err := s.kbRepository.GetKnowledgeBaseByID(ctx, chunk.KnowledgeBaseID)
if err != nil {
return err
}
if !kb.NeedsEmbeddingModel() {
return nil
}
embedder, err := s.modelService.GetEmbeddingModel(ctx, kb.EmbeddingModelID)
if err != nil {
return err
}
engine, err := retriever.CreateRetrieveEngineForKB(ctx, s.retrieveEngine, s.ownership, chunk.TenantID, kb.VectorStoreID)
if err != nil {
return err
}
if err := engine.DeleteByChunkIDList(ctx, []string{chunk.ID}, embedder.GetDimensions(), kb.Type); err != nil {
return err
}
if !chunk.IsEnabled {
return nil
}
knowledge, err := s.knowledgeRepo.GetKnowledgeByID(ctx, chunk.TenantID, chunk.KnowledgeID)
if err != nil {
return err
}
prefix := ""
if knowledge.Title != "" {
prefix = strings.TrimSpace(knowledge.Title) + "\n"
}
if metadata := knowledge.CustomMetadataText(); metadata != "" {
prefix += "Metadata:\n" + metadata + "\n"
}
items := []*types.IndexInfo{{
Content: prefix + chunk.EmbeddingContent(), SourceID: chunk.ID,
SourceType: types.ChunkSourceType, ChunkID: chunk.ID,
KnowledgeID: chunk.KnowledgeID, KnowledgeBaseID: chunk.KnowledgeBaseID,
KnowledgeType: kb.Type, IsEnabled: true,
}}
meta, err := chunk.DocumentMetadata()
if err != nil {
return err
}
if meta != nil {
for _, question := range meta.GeneratedQuestions {
if strings.TrimSpace(question.Question) == "" {
continue
}
items = append(items, &types.IndexInfo{
Content: prefix + question.Question, SourceID: fmt.Sprintf("%s-%s", chunk.ID, question.ID),
SourceType: types.ChunkSourceType, ChunkID: chunk.ID,
KnowledgeID: chunk.KnowledgeID, KnowledgeBaseID: chunk.KnowledgeBaseID,
KnowledgeType: kb.Type, IsEnabled: true,
})
}
}
return engine.BatchIndex(ctx, embedder, items)
}
func (s *chunkService) UpsertGeneratedQuestion(
ctx context.Context, chunkID string, questionID string, question string,
) (*types.GeneratedQuestion, error) {
question = strings.TrimSpace(question)
if question == "" {
return nil, fmt.Errorf("question cannot be empty")
}
chunk, err := s.chunkRepository.GetChunkByID(ctx, types.MustTenantIDFromContext(ctx), chunkID)
if err != nil {
return nil, err
}
meta, err := chunk.DocumentMetadata()
if err != nil {
return nil, err
}
if meta == nil {
meta = &types.DocumentChunkMetadata{}
}
meta.GeneratedQuestionsRevision = chunk.ContentRevision
if questionID == "" {
questionID = uuid.NewString()
meta.GeneratedQuestions = append(meta.GeneratedQuestions, types.GeneratedQuestion{ID: questionID, Question: question})
} else {
found := false
for i := range meta.GeneratedQuestions {
if meta.GeneratedQuestions[i].ID == questionID {
meta.GeneratedQuestions[i].Question = question
found = true
break
}
}
if !found {
return nil, fmt.Errorf("question not found")
}
}
if err := chunk.SetDocumentMetadata(meta); err != nil {
return nil, err
}
if err := s.chunkRepository.UpdateChunk(ctx, chunk); err != nil {
return nil, err
}
if err := s.syncChunkIndex(ctx, chunk); err != nil {
return nil, err
}
for i := range meta.GeneratedQuestions {
if meta.GeneratedQuestions[i].ID == questionID {
return &meta.GeneratedQuestions[i], nil
}
}
return nil, fmt.Errorf("question not found")
}
// DeleteGeneratedQuestion deletes a single generated question from a chunk by question ID
// This updates the chunk metadata and removes the corresponding vector index
func (s *chunkService) DeleteGeneratedQuestion(ctx context.Context, chunkID string, questionID string) error {
+34
View File
@@ -2,6 +2,7 @@ package service
import (
"context"
"encoding/json"
"errors"
"fmt"
"io"
@@ -609,12 +610,45 @@ func (s *knowledgeService) UpdateKnowledge(ctx context.Context, knowledge *types
if knowledge.Description != "" {
record.Description = knowledge.Description
}
metadataChanged := knowledge.CustomMetadata != nil
if metadataChanged {
var custom map[string]interface{}
if err := json.Unmarshal(knowledge.CustomMetadata, &custom); err != nil {
return fmt.Errorf("custom_metadata must be a JSON object: %w", err)
}
if len(custom) > 20 {
return fmt.Errorf("custom_metadata supports at most 20 fields")
}
for key, value := range custom {
if len(strings.TrimSpace(key)) == 0 || len(key) > 64 || len(fmt.Sprint(value)) > 1000 {
return fmt.Errorf("invalid custom_metadata field %q", key)
}
switch value.(type) {
case string, float64, bool, nil:
default:
return fmt.Errorf("custom_metadata field %q must be a string, number, boolean, or null", key)
}
}
record.CustomMetadata = knowledge.CustomMetadata
}
// Update knowledge record in the repository
if err := s.repo.UpdateKnowledge(ctx, record); err != nil {
logger.Errorf(ctx, "Failed to update knowledge: %v", err)
return err
}
if metadataChanged {
chunks, err := s.chunkRepo.ListChunksByKnowledgeID(ctx, record.TenantID, record.ID)
if err != nil {
return fmt.Errorf("list chunks for metadata reindex: %w", err)
}
if err := s.updateChunkVector(ctx, record.KnowledgeBaseID, chunks); err != nil {
return fmt.Errorf("metadata saved but reindex failed: %w", err)
}
if record.SummaryStatus == types.SummaryStatusCompleted {
_ = s.repo.UpdateKnowledgeColumn(ctx, record.ID, "summary_status", types.SummaryStatusPending)
}
}
logger.Infof(ctx, "Knowledge updated successfully, ID: %s", knowledge.ID)
return nil
}
+298 -11
View File
@@ -533,6 +533,9 @@ func (s *knowledgeService) processChunks(ctx context.Context,
if t := strings.TrimSpace(knowledge.Title); t != "" {
titlePrefix = t + "\n"
}
if metadata := knowledge.CustomMetadataText(); metadata != "" {
titlePrefix += "Metadata:\n" + metadata + "\n"
}
for _, chunk := range textChunks {
// chunk.EmbeddingContent prepends ContextHeader (heading breadcrumb)
// when the chunker populated it during Tier-1 splitting; falls back
@@ -764,12 +767,32 @@ func (s *knowledgeService) getSummary(ctx context.Context,
// happen AFTER concatenation because StartAt is based on original document
// offsets — enriched (longer) content would break the positioning.
chunkContents := ""
hasEditedChunk := false
for _, chunk := range sortedChunks {
runes := []rune(chunkContents)
if chunk.StartAt <= len(runes) {
chunkContents = string(runes[:chunk.StartAt]) + chunk.Content
} else {
chunkContents = chunkContents + chunk.Content
if chunk.ContentRevision > 0 {
hasEditedChunk = true
break
}
}
if hasEditedChunk {
// Parser offsets describe the immutable source. Once a replacement has
// changed length they can no longer be applied to the effective content;
// concatenate current chunks instead of truncating at stale offsets.
parts := make([]string, 0, len(sortedChunks))
for _, chunk := range sortedChunks {
if chunk.IsEnabled && strings.TrimSpace(chunk.Content) != "" {
parts = append(parts, chunk.Content)
}
}
chunkContents = strings.Join(parts, "\n\n")
} else {
for _, chunk := range sortedChunks {
runes := []rune(chunkContents)
if chunk.StartAt <= len(runes) {
chunkContents = string(runes[:chunk.StartAt]) + chunk.Content
} else {
chunkContents += chunk.Content
}
}
}
@@ -815,8 +838,13 @@ func (s *knowledgeService) getSummary(ctx context.Context,
return "", err
}
// Pass the raw chunk text to the LLM with no filename / file-type framing.
// User-authored metadata is trusted document context. Internal ingestion
// metadata remains excluded because it contains IDs and pipeline controls.
contentWithMetadata := chunkContents
if custom := knowledge.CustomMetadataText(); custom != "" {
contentWithMetadata = "Document metadata:\n" + custom + "\n\nDocument content:\n" + chunkContents
}
contentWithMetadata = sampleLongContent(contentWithMetadata, maxInputChars)
// Determine max output tokens from config
maxTokens := 2048
@@ -1050,6 +1078,7 @@ func (s *knowledgeService) ProcessSummaryGeneration(ctx context.Context, t *asyn
}
// Generate summary
summaryMetadataVersion := string(knowledge.CustomMetadata)
summary, err := s.getSummary(ctx, chatModel, knowledge, textChunks)
if err != nil {
logger.Errorf(ctx, "Failed to generate summary for knowledge %s: %v", payload.KnowledgeID, err)
@@ -1089,6 +1118,25 @@ func (s *knowledgeService) ProcessSummaryGeneration(ctx context.Context, t *asyn
summaryOut["fallback"] = "first_chunk"
}
}
// Do not publish an answer derived from a superseded chunk or metadata
// version. A user can explicitly refresh again from the latest revision.
latestKnowledge, latestErr := s.repo.GetKnowledgeByID(ctx, payload.TenantID, payload.KnowledgeID)
staleSummary := latestErr != nil || string(latestKnowledge.CustomMetadata) != summaryMetadataVersion
if !staleSummary {
for _, sourceChunk := range textChunks {
latestChunk, getErr := s.chunkRepo.GetChunkByID(ctx, payload.TenantID, sourceChunk.ID)
if getErr != nil || latestChunk.ContentRevision != sourceChunk.ContentRevision || latestChunk.IsEnabled != sourceChunk.IsEnabled {
staleSummary = true
break
}
}
}
if staleSummary {
logger.Infof(ctx, "Discarding stale summary for knowledge %s", payload.KnowledgeID)
_ = s.repo.UpdateKnowledgeColumn(ctx, payload.KnowledgeID, "summary_status", types.SummaryStatusPending)
summaryOut["skipped"] = "content_revision_changed"
return nil
}
// Update knowledge description
knowledge.Description = summary
@@ -1499,6 +1547,7 @@ func (s *knowledgeService) processQuestionGenerationForKnowledge(ctx context.Con
nextContent = enrichContent(textChunks[i+1])
}
generationRevision := chunk.ContentRevision
llmCallAttempts++
questions, err := s.generateQuestionsWithContext(ctx, chatModel, enrichContent(chunk), prevContent, nextContent,
knowledge.Title, questionCount, customInstructions)
@@ -1512,6 +1561,12 @@ func (s *knowledgeService) processQuestionGenerationForKnowledge(ctx context.Con
llmCallEmpty++
continue
}
latestChunk, latestErr := s.chunkRepo.GetChunkByID(ctx, payload.TenantID, chunk.ID)
if latestErr != nil || latestChunk.ContentRevision != generationRevision {
logger.Infof(ctx, "Skipping stale generated questions for chunk %s (revision changed)", chunk.ID)
continue
}
chunk = latestChunk
llmCallSuccess++
generatedQuestionsTotal += len(questions)
if sampleQuestion == "" && len(questions) > 0 {
@@ -1528,7 +1583,7 @@ func (s *knowledgeService) processQuestionGenerationForKnowledge(ctx context.Con
}
}
meta := &types.DocumentChunkMetadata{
GeneratedQuestions: generatedQuestions,
GeneratedQuestions: generatedQuestions, GeneratedQuestionsRevision: chunk.ContentRevision,
}
if err := chunk.SetDocumentMetadata(meta); err != nil {
chunkMetadataSetFailed++
@@ -1830,6 +1885,7 @@ func (s *knowledgeService) processQuestionGenerationForChunks(ctx context.Contex
continue
}
generationRevision := chunk.ContentRevision
questions, gerr := s.generateQuestionsWithContext(
ctx, chatModel, enrich(chunk), prevContentAt(i), nextContentAt(i), knowledge.Title, questionCount,
customInstructions)
@@ -1841,6 +1897,12 @@ func (s *knowledgeService) processQuestionGenerationForChunks(ctx context.Contex
if len(questions) == 0 {
continue
}
latestChunk, latestErr := s.chunkRepo.GetChunkByID(ctx, payload.TenantID, chunk.ID)
if latestErr != nil || latestChunk.ContentRevision != generationRevision {
logger.Infof(ctx, "Skipping stale generated questions for chunk %s (revision changed)", chunk.ID)
continue
}
chunk = latestChunk
chunksProcessed++
generatedQuestionsTotal += len(questions)
if sampleQuestion == "" {
@@ -1854,7 +1916,9 @@ func (s *knowledgeService) processQuestionGenerationForChunks(ctx context.Contex
Question: question,
}
}
meta := &types.DocumentChunkMetadata{GeneratedQuestions: generatedQuestions}
meta := &types.DocumentChunkMetadata{
GeneratedQuestions: generatedQuestions, GeneratedQuestionsRevision: chunk.ContentRevision,
}
if err := chunk.SetDocumentMetadata(meta); err != nil {
logger.Warnf(ctx, "Failed to set document metadata for chunk %s: %v", chunk.ID, err)
continue
@@ -1965,6 +2029,190 @@ func (s *knowledgeService) generateQuestionsWithContext(ctx context.Context,
return questions, nil
}
// RegenerateChunkQuestions synchronously refreshes Doc2Query-style auxiliary
// questions for the current chunk revision and atomically replaces their
// retrieval entries through updateChunkVector.
func (s *knowledgeService) RegenerateChunkQuestions(
ctx context.Context, chunkID string,
) ([]types.GeneratedQuestion, error) {
tenantID := types.MustTenantIDFromContext(ctx)
chunk, err := s.chunkRepo.GetChunkByID(ctx, tenantID, chunkID)
if err != nil {
return nil, err
}
if chunk.ChunkType != types.ChunkTypeText {
return nil, fmt.Errorf("questions can only be generated for text chunks")
}
generationRevision := chunk.ContentRevision
knowledge, err := s.repo.GetKnowledgeByID(ctx, tenantID, chunk.KnowledgeID)
if err != nil {
return nil, err
}
kb, err := s.kbService.GetKnowledgeBaseByID(ctx, chunk.KnowledgeBaseID)
if err != nil {
return nil, err
}
if kb.SummaryModelID == "" {
return nil, fmt.Errorf("summary model is required for question generation")
}
chatModel, err := s.modelService.GetChatModel(ctx, kb.SummaryModelID)
if err != nil {
return nil, err
}
resolveNeighbor := func(id string) string {
if id == "" {
return ""
}
neighbor, getErr := s.chunkRepo.GetChunkByID(ctx, tenantID, id)
if getErr != nil {
return ""
}
return neighbor.Content
}
overrides, _ := knowledge.ProcessOverrides()
config := ResolveProcessConfig(kb, overrides).QuestionGenerationConfig
count := config.QuestionCount
if count <= 0 {
count = 3
}
if count > 10 {
count = 10
}
questions, err := s.generateQuestionsWithContext(
ctx, chatModel, chunk.Content, resolveNeighbor(chunk.PreChunkID),
resolveNeighbor(chunk.NextChunkID), knowledge.Title, count, config.CustomInstructions,
)
if err != nil {
return nil, err
}
latestChunk, err := s.chunkRepo.GetChunkByID(ctx, tenantID, chunkID)
if err != nil {
return nil, err
}
if latestChunk.ContentRevision != generationRevision {
return nil, ErrChunkRevisionConflict
}
chunk = latestChunk
generated := make([]types.GeneratedQuestion, 0, len(questions))
for _, question := range questions {
generated = append(generated, types.GeneratedQuestion{ID: uuid.NewString(), Question: question})
}
meta := &types.DocumentChunkMetadata{
GeneratedQuestions: generated, GeneratedQuestionsRevision: chunk.ContentRevision,
}
if err := chunk.SetDocumentMetadata(meta); err != nil {
return nil, err
}
if err := s.chunkRepo.UpdateChunk(ctx, chunk); err != nil {
return nil, err
}
if err := s.updateChunkVector(ctx, chunk.KnowledgeBaseID, []*types.Chunk{chunk}); err != nil {
return nil, err
}
return generated, nil
}
// RegenerateKnowledgeSummary refreshes both the knowledge description and the
// summary chunk(s), then reindexes all retrieval artifacts with current
// custom metadata. It is intentionally idempotent.
func (s *knowledgeService) RegenerateKnowledgeSummary(
ctx context.Context, knowledgeID string,
) (*types.Knowledge, error) {
tenantID := types.MustTenantIDFromContext(ctx)
knowledge, err := s.repo.GetKnowledgeByID(ctx, tenantID, knowledgeID)
if err != nil {
return nil, err
}
kb, err := s.kbService.GetKnowledgeBaseByID(ctx, knowledge.KnowledgeBaseID)
if err != nil {
return nil, err
}
if kb.SummaryModelID == "" {
return nil, fmt.Errorf("summary model is not configured")
}
allChunks, err := s.chunkRepo.ListChunksByKnowledgeID(ctx, tenantID, knowledgeID)
if err != nil {
return nil, err
}
textChunks := make([]*types.Chunk, 0)
for _, chunk := range allChunks {
if chunk.ChunkType == types.ChunkTypeText && chunk.IsEnabled {
textChunks = append(textChunks, chunk)
}
}
if len(textChunks) == 0 {
return nil, fmt.Errorf("no enabled text chunks to summarize")
}
chatModel, err := s.modelService.GetChatModel(ctx, kb.SummaryModelID)
if err != nil {
return nil, err
}
metadataVersion := string(knowledge.CustomMetadata)
knowledge.SummaryStatus = types.SummaryStatusProcessing
if err := s.repo.UpdateKnowledge(ctx, knowledge); err != nil {
return nil, err
}
summary, err := s.getSummary(ctx, chatModel, knowledge, textChunks)
if err != nil {
knowledge.SummaryStatus = types.SummaryStatusFailed
_ = s.repo.UpdateKnowledge(ctx, knowledge)
return nil, err
}
latestKnowledge, err := s.repo.GetKnowledgeByID(ctx, tenantID, knowledgeID)
if err != nil || string(latestKnowledge.CustomMetadata) != metadataVersion {
_ = s.repo.UpdateKnowledgeColumn(ctx, knowledgeID, "summary_status", types.SummaryStatusPending)
return nil, ErrChunkRevisionConflict
}
for _, sourceChunk := range textChunks {
latestChunk, getErr := s.chunkRepo.GetChunkByID(ctx, tenantID, sourceChunk.ID)
if getErr != nil || latestChunk.ContentRevision != sourceChunk.ContentRevision || latestChunk.IsEnabled != sourceChunk.IsEnabled {
_ = s.repo.UpdateKnowledgeColumn(ctx, knowledgeID, "summary_status", types.SummaryStatusPending)
return nil, ErrChunkRevisionConflict
}
}
knowledge.Description = summary
knowledge.SummaryStatus = types.SummaryStatusCompleted
knowledge.UpdatedAt = time.Now()
if err := s.repo.UpdateKnowledge(ctx, knowledge); err != nil {
return nil, err
}
if kb.NeedsEmbeddingModel() {
found := false
maxIndex := 0
for _, chunk := range allChunks {
if chunk.ChunkIndex > maxIndex {
maxIndex = chunk.ChunkIndex
}
if chunk.ChunkType == types.ChunkTypeSummary {
chunk.Content = "# Summary\n" + summary
chunk.SourceContent = chunk.Content
chunk.IsEnabled = true
chunk.UpdatedAt = time.Now()
if err := s.chunkRepo.UpdateChunk(ctx, chunk); err != nil {
return nil, err
}
found = true
}
}
if !found {
summaryChunk := &types.Chunk{
ID: uuid.NewString(), TenantID: tenantID, KnowledgeID: knowledge.ID,
KnowledgeBaseID: knowledge.KnowledgeBaseID, Content: "# Summary\n" + summary,
ChunkIndex: maxIndex + 1, IsEnabled: true, ChunkType: types.ChunkTypeSummary,
ParentChunkID: textChunks[0].ID, CreatedAt: time.Now(), UpdatedAt: time.Now(),
}
if err := s.chunkRepo.CreateChunks(ctx, []*types.Chunk{summaryChunk}); err != nil {
return nil, err
}
allChunks = append(allChunks, summaryChunk)
}
if err := s.updateChunkVector(ctx, knowledge.KnowledgeBaseID, allChunks); err != nil {
return nil, err
}
}
return knowledge, nil
}
// ReparseKnowledge deletes existing document content and re-parses the knowledge asynchronously.
// This method reuses the logic from UpdateManualKnowledge for resource cleanup and async parsing.
func (s *knowledgeService) ReparseKnowledge(
@@ -2386,6 +2634,9 @@ func (s *knowledgeService) updateChunkVector(ctx context.Context, kbID string, c
if err != nil {
return err
}
if !sourceKB.NeedsEmbeddingModel() {
return nil
}
embeddingModel, err := s.modelService.GetEmbeddingModel(ctx, sourceKB.EmbeddingModelID)
if err != nil {
return err
@@ -2394,21 +2645,57 @@ func (s *knowledgeService) updateChunkVector(ctx context.Context, kbID string, c
// Initialize composite retrieve engine from tenant configuration
indexInfo := make([]*types.IndexInfo, 0, len(chunks))
ids := make([]string, 0, len(chunks))
knowledgeCache := make(map[string]*types.Knowledge)
for _, chunk := range chunks {
if chunk.KnowledgeBaseID != kbID {
logger.Warnf(ctx, "Knowledge base ID mismatch: %s != %s", chunk.KnowledgeBaseID, kbID)
continue
}
ids = append(ids, chunk.ID)
if !chunk.IsEnabled || chunk.ChunkType == types.ChunkTypeParentText {
continue
}
knowledge := knowledgeCache[chunk.KnowledgeID]
if knowledge == nil {
knowledge, err = s.repo.GetKnowledgeByID(ctx, chunk.TenantID, chunk.KnowledgeID)
if err != nil {
return err
}
knowledgeCache[chunk.KnowledgeID] = knowledge
}
prefix := ""
if title := strings.TrimSpace(knowledge.Title); title != "" {
prefix = title + "\n"
}
if metadata := knowledge.CustomMetadataText(); metadata != "" {
prefix += "Metadata:\n" + metadata + "\n"
}
indexInfo = append(indexInfo, &types.IndexInfo{
Content: chunk.Content,
Content: prefix + chunk.EmbeddingContent(),
SourceID: chunk.ID,
SourceType: types.ChunkSourceType,
ChunkID: chunk.ID,
KnowledgeID: chunk.KnowledgeID,
KnowledgeBaseID: chunk.KnowledgeBaseID,
IsEnabled: true,
KnowledgeType: sourceKB.Type,
IsEnabled: chunk.IsEnabled,
})
ids = append(ids, chunk.ID)
meta, metaErr := chunk.DocumentMetadata()
if metaErr != nil {
return metaErr
}
if meta != nil {
for _, q := range meta.GeneratedQuestions {
if strings.TrimSpace(q.Question) != "" {
indexInfo = append(indexInfo, &types.IndexInfo{
Content: prefix + q.Question, SourceID: fmt.Sprintf("%s-%s", chunk.ID, q.ID),
SourceType: types.ChunkSourceType, ChunkID: chunk.ID,
KnowledgeID: chunk.KnowledgeID, KnowledgeBaseID: chunk.KnowledgeBaseID,
KnowledgeType: sourceKB.Type, IsEnabled: true,
})
}
}
}
}
retrieveEngine, err := retriever.CreateRetrieveEngineForKB(
@@ -307,27 +307,28 @@ func (s *knowledgeBaseService) buildSearchResult(chunk *types.Chunk,
matchedContent string,
) *types.SearchResult {
return &types.SearchResult{
ID: chunk.ID,
Content: chunk.Content,
KnowledgeID: chunk.KnowledgeID,
ChunkIndex: chunk.ChunkIndex,
KnowledgeTitle: knowledge.Title,
StartAt: chunk.StartAt,
EndAt: chunk.EndAt,
Seq: chunk.ChunkIndex,
Score: score,
MatchType: matchType,
Metadata: knowledge.GetMetadata(),
ChunkType: string(chunk.ChunkType),
ParentChunkID: chunk.ParentChunkID,
ImageInfo: chunk.ImageInfo,
KnowledgeFilename: knowledge.FileName,
KnowledgeSource: knowledge.Source,
KnowledgeChannel: knowledge.Channel,
KnowledgeDescription: knowledge.Description,
ChunkMetadata: chunk.Metadata,
MatchedContent: matchedContent,
KnowledgeBaseID: knowledge.KnowledgeBaseID,
ID: chunk.ID,
Content: chunk.Content,
KnowledgeID: chunk.KnowledgeID,
ChunkIndex: chunk.ChunkIndex,
KnowledgeTitle: knowledge.Title,
StartAt: chunk.StartAt,
EndAt: chunk.EndAt,
Seq: chunk.ChunkIndex,
Score: score,
MatchType: matchType,
Metadata: knowledge.GetMetadata(),
ChunkType: string(chunk.ChunkType),
ParentChunkID: chunk.ParentChunkID,
ImageInfo: chunk.ImageInfo,
KnowledgeFilename: knowledge.FileName,
KnowledgeSource: knowledge.Source,
KnowledgeChannel: knowledge.Channel,
KnowledgeDescription: knowledge.Description,
KnowledgeCustomMetadata: knowledge.CustomMetadataText(),
ChunkMetadata: chunk.Metadata,
MatchedContent: matchedContent,
KnowledgeBaseID: knowledge.KnowledgeBaseID,
}
}
+94 -14
View File
@@ -1,6 +1,7 @@
package handler
import (
stderrors "errors"
"net/http"
"github.com/Tencent/WeKnora/internal/application/service"
@@ -152,13 +153,9 @@ func (h *ChunkHandler) ListKnowledgeChunks(c *gin.Context) {
// UpdateChunkRequest defines the request structure for updating a chunk
type UpdateChunkRequest struct {
Content string `json:"content"`
Embedding []float32 `json:"embedding"`
ChunkIndex int `json:"chunk_index"`
IsEnabled bool `json:"is_enabled"`
StartAt int `json:"start_at"`
EndAt int `json:"end_at"`
ImageInfo string `json:"image_info"`
Content *string `json:"content"`
IsEnabled *bool `json:"is_enabled"`
ExpectedRevision *int `json:"expected_revision"`
}
// fetchChunkAndVerifyOwnership fetches a chunk by ID and verifies it
@@ -227,14 +224,13 @@ func (h *ChunkHandler) UpdateChunk(c *gin.Context) {
return
}
if req.Content != "" {
chunk.Content = req.Content
}
chunk.IsEnabled = req.IsEnabled
if err := h.service.UpdateChunk(ctx, chunk); err != nil {
chunk, err = h.service.UpdateDocumentChunk(ctx, chunk.ID, req.Content, req.IsEnabled, req.ExpectedRevision)
if err != nil {
logger.ErrorWithFields(ctx, err, nil)
if stderrors.Is(err, service.ErrChunkRevisionConflict) {
c.Error(errors.NewConflictError("Chunk was modified by another user; refresh and retry"))
return
}
c.Error(errors.NewInternalServerError(err.Error()))
return
}
@@ -247,6 +243,90 @@ func (h *ChunkHandler) UpdateChunk(c *gin.Context) {
})
}
func (h *ChunkHandler) ListChunkRevisions(c *gin.Context) {
chunk, _, err := h.fetchChunkAndVerifyOwnership(c)
if err != nil {
c.Error(err)
return
}
items, err := h.service.ListChunkRevisions(c.Request.Context(), chunk.ID)
if err != nil {
c.Error(errors.NewInternalServerError(err.Error()))
return
}
c.JSON(http.StatusOK, gin.H{"success": true, "data": items})
}
type RevertChunkRequest struct {
Revision *int `json:"revision" binding:"required"`
ExpectedRevision *int `json:"expected_revision"`
}
func (h *ChunkHandler) RevertChunk(c *gin.Context) {
chunk, _, err := h.fetchChunkAndVerifyOwnership(c)
if err != nil {
c.Error(err)
return
}
var req RevertChunkRequest
if err := c.ShouldBindJSON(&req); err != nil {
c.Error(errors.NewBadRequestError(err.Error()))
return
}
if req.Revision == nil || *req.Revision < 0 {
c.Error(errors.NewBadRequestError("revision must be a non-negative integer"))
return
}
updated, err := h.service.RevertDocumentChunk(c.Request.Context(), chunk.ID, *req.Revision, req.ExpectedRevision)
if stderrors.Is(err, service.ErrChunkRevisionConflict) {
c.Error(errors.NewConflictError("Chunk was modified by another user; refresh and retry"))
return
}
if err != nil {
c.Error(errors.NewBadRequestError(err.Error()))
return
}
c.JSON(http.StatusOK, gin.H{"success": true, "data": updated})
}
type UpsertGeneratedQuestionRequest struct {
QuestionID string `json:"question_id"`
Question string `json:"question" binding:"required"`
}
func (h *ChunkHandler) UpsertGeneratedQuestion(c *gin.Context) {
chunkID := secutils.SanitizeForLog(c.Param("id"))
if chunkID == "" {
c.Error(errors.NewBadRequestError("Chunk ID is required"))
return
}
var req UpsertGeneratedQuestionRequest
if err := c.ShouldBindJSON(&req); err != nil {
c.Error(errors.NewBadRequestError(err.Error()))
return
}
item, err := h.service.UpsertGeneratedQuestion(c.Request.Context(), chunkID, req.QuestionID, req.Question)
if err != nil {
c.Error(errors.NewBadRequestError(err.Error()))
return
}
c.JSON(http.StatusOK, gin.H{"success": true, "data": item})
}
func (h *ChunkHandler) RegenerateGeneratedQuestions(c *gin.Context) {
chunkID := secutils.SanitizeForLog(c.Param("id"))
if chunkID == "" {
c.Error(errors.NewBadRequestError("Chunk ID is required"))
return
}
items, err := h.kgService.RegenerateChunkQuestions(c.Request.Context(), chunkID)
if err != nil {
c.Error(errors.NewBadRequestError(err.Error()))
return
}
c.JSON(http.StatusOK, gin.H{"success": true, "data": items})
}
// DeleteChunk godoc
// @Summary 删除分块
// @Description 删除指定的分块
+22
View File
@@ -1573,6 +1573,28 @@ func (h *KnowledgeHandler) UpdateKnowledge(c *gin.Context) {
})
}
// RegenerateKnowledgeSummary refreshes a stale summary after chunk or metadata edits.
func (h *KnowledgeHandler) RegenerateKnowledgeSummary(c *gin.Context) {
ctx := c.Request.Context()
id := secutils.SanitizeForLog(c.Param("id"))
if id == "" {
c.Error(errors.NewBadRequestError("Knowledge ID cannot be empty"))
return
}
_, effCtx, err := h.resolveKnowledgeAndValidateKBAccess(c, id, types.OrgRoleEditor)
if err != nil {
c.Error(err)
return
}
knowledge, err := h.kgService.RegenerateKnowledgeSummary(effCtx, id)
if err != nil {
logger.ErrorWithFields(ctx, err, nil)
c.Error(errors.NewBadRequestError(err.Error()))
return
}
c.JSON(http.StatusOK, gin.H{"success": true, "data": knowledge})
}
// UpdateManualKnowledge godoc
// @Summary 更新手工知识
// @Description 更新手工录入的Markdown知识内容
+5
View File
@@ -35,18 +35,22 @@ func RegisterChunkRoutes(r *gin.RouterGroup, handler *handler.ChunkHandler, g *r
chunkRead.GET("/:knowledge_id", g.Viewer(), g.KBAccessReadFromKnowledgeIDParam("knowledge_id"), handler.ListKnowledgeChunks)
// 通过chunk_id获取单个chunk(不需要knowledge_id) — Viewer+ 且对父 KB 有 read 权限
chunkRead.GET("/by-id/:id", g.Viewer(), g.KBAccessReadFromChunkIDParam("id"), handler.GetChunkByIDOnly)
chunkRead.GET("/:knowledge_id/:id/revisions", g.Viewer(), g.KBAccessReadFromKnowledgeIDParam("knowledge_id"), handler.ListChunkRevisions)
// 删除分块 — KB owner OR Admin+,且对父 KB 有 write 权限
chunks.DELETE("/:knowledge_id/:id", g.OwnedChunkKBOrAdmin(), g.KBAccessWriteFromKnowledgeIDParam("knowledge_id"), handler.DeleteChunk)
// 删除知识下的所有分块 — KB owner OR Admin+,且对父 KB 有 write 权限
chunks.DELETE("/:knowledge_id", g.OwnedChunkKBOrAdmin(), g.KBAccessWriteFromKnowledgeIDParam("knowledge_id"), handler.DeleteChunksByKnowledgeID)
// 更新分块信息 — KB owner OR Admin+,且对父 KB 有 write 权限
chunks.PUT("/:knowledge_id/:id", g.OwnedChunkKBOrAdmin(), g.KBAccessWriteFromKnowledgeIDParam("knowledge_id"), handler.UpdateChunk)
chunks.POST("/:knowledge_id/:id/revert", g.OwnedChunkKBOrAdmin(), g.KBAccessWriteFromKnowledgeIDParam("knowledge_id"), handler.RevertChunk)
// 删除单个生成的问题(通过分块 id) — 与其它 chunk mutation 一致:
// KB owner OR Admin+。早期这里因为链路 (chunk_id -> knowledge_id ->
// kb -> creator_id) 还没接通,被临时降级成 Contributor,导致一个
// 「能编辑所有 chunk 的同样规则在这条路由上反而更宽松」的不一致。
// 现在通过 KBCreatorLookupFromChunkIDParam 把那一跳补上,统一矩阵。
chunks.DELETE("/by-id/:id/questions", g.OwnedChunkKBOrAdminFromChunkID(), g.KBAccessWriteFromChunkIDParam("id"), handler.DeleteGeneratedQuestion)
chunks.PUT("/by-id/:id/questions", g.OwnedChunkKBOrAdminFromChunkID(), g.KBAccessWriteFromChunkIDParam("id"), handler.UpsertGeneratedQuestion)
chunks.POST("/by-id/:id/questions/regenerate", g.OwnedChunkKBOrAdminFromChunkID(), g.KBAccessWriteFromChunkIDParam("id"), handler.RegenerateGeneratedQuestions)
}
}
@@ -99,6 +103,7 @@ func RegisterKnowledgeRoutes(r *gin.RouterGroup, handler *handler.KnowledgeHandl
kRead.GET("/:id/spans", g.Viewer(), g.KBAccessReadFromKnowledgeIDParam("id"), handler.GetKnowledgeSpans)
k.DELETE("/:id", g.OwnedKnowledgeKBOrAdmin(), g.KBAccessWriteFromKnowledgeIDParam("id"), handler.DeleteKnowledge)
k.PUT("/:id", g.OwnedKnowledgeKBOrAdmin(), g.KBAccessWriteFromKnowledgeIDParam("id"), handler.UpdateKnowledge)
k.POST("/:id/regenerate-summary", g.OwnedKnowledgeKBOrAdmin(), g.KBAccessWriteFromKnowledgeIDParam("id"), handler.RegenerateKnowledgeSummary)
k.PUT("/manual/:id", g.OwnedKnowledgeKBOrAdmin(), g.KBAccessWriteFromKnowledgeIDParam("id"), handler.UpdateManualKnowledge)
k.POST("/:id/reparse", g.OwnedKnowledgeKBOrAdmin(), g.KBAccessWriteFromKnowledgeIDParam("id"), handler.ReparseKnowledge)
k.POST("/:id/cancel-parse", g.OwnedKnowledgeKBOrAdmin(), g.KBAccessWriteFromKnowledgeIDParam("id"), handler.CancelKnowledgeParse)
+30 -5
View File
@@ -125,6 +125,16 @@ type Chunk struct {
TagID string `json:"tag_id" gorm:"type:varchar(36);index"`
// Actual text content of the chunk
Content string `json:"content"`
// SourceContent is the immutable parser output. Legacy rows are lazily
// backfilled from Content on the first manual edit.
SourceContent string `json:"-"`
// ContentRevision is incremented for every user edit or rollback.
ContentRevision int `json:"content_revision" gorm:"not null;default:0"`
// IndexStatus reports whether the current content is reflected in the
// retrieval stores: ready | processing | failed.
IndexStatus string `json:"index_status" gorm:"type:varchar(16);not null;default:'ready'"`
// LastEditorID records the actor that produced the current revision.
LastEditorID string `json:"last_editor_id" gorm:"type:varchar(64);not null;default:''"`
// Index position of the chunk in the original document
ChunkIndex int `json:"chunk_index"`
// Whether the chunk is enabled, can be used to temporarily disable certain chunks
@@ -162,11 +172,26 @@ type Chunk struct {
UpdatedAt time.Time `json:"updated_at"`
// Soft delete marker, supports data recovery
DeletedAt gorm.DeletedAt `json:"deleted_at" gorm:"index"`
// ContextHeader is an in-memory-only context string (e.g. a Markdown
// heading breadcrumb) that the indexing pipeline prepends to Content
// when generating embeddings. NOT persisted — populated by the chunker
// during initial splitting and discarded after indexing.
ContextHeader string `json:"-" gorm:"-"`
// ContextHeader is a Markdown heading breadcrumb prepended when indexing.
// It is persisted so a later content edit can rebuild the same index input.
ContextHeader string `json:"-" gorm:"type:text"`
}
// ChunkRevision is an immutable snapshot of a superseded chunk revision.
// The current content lives on Chunk; this table stores prior versions.
type ChunkRevision struct {
ID string `json:"id" gorm:"type:varchar(36);primaryKey"`
TenantID uint64 `json:"tenant_id" gorm:"index"`
KnowledgeBaseID string `json:"knowledge_base_id" gorm:"type:varchar(36);index"`
KnowledgeID string `json:"knowledge_id" gorm:"type:varchar(36);index"`
ChunkID string `json:"chunk_id" gorm:"type:varchar(36);uniqueIndex:idx_chunk_revision"`
Revision int `json:"revision" gorm:"uniqueIndex:idx_chunk_revision"`
Content string `json:"content" gorm:"type:text"`
IsEnabled bool `json:"is_enabled"`
EditorID string `json:"editor_id" gorm:"type:varchar(64)"`
EditSource string `json:"edit_source" gorm:"type:varchar(16)"`
EditedAt time.Time `json:"edited_at"`
CreatedAt time.Time `json:"created_at"`
}
// EmbeddingContent returns the chunk content with ContextHeader prepended
+2
View File
@@ -36,6 +36,8 @@ type DocumentChunkMetadata struct {
// GeneratedQuestions 存储AI为该Chunk生成的相关问题
// 这些问题会被独立索引以提高召回率
GeneratedQuestions []GeneratedQuestion `json:"generated_questions,omitempty"`
// GeneratedQuestionsRevision ties the questions to Chunk.ContentRevision.
GeneratedQuestionsRevision int `json:"generated_questions_revision,omitempty"`
}
// GetQuestionStrings 返回问题内容字符串列表(兼容旧代码)
+17
View File
@@ -54,6 +54,15 @@ type ChunkRepository interface {
ListChunksByParentIDs(ctx context.Context, tenantID uint64, parentIDs []string) ([]*types.Chunk, error)
// UpdateChunk updates a chunk
UpdateChunk(ctx context.Context, chunk *types.Chunk) error
// CreateChunkRevision stores an immutable snapshot of a superseded revision.
CreateChunkRevision(ctx context.Context, revision *types.ChunkRevision) error
// SaveChunkRevision atomically snapshots the old row and applies the new
// row only when its revision still matches expectedRevision.
SaveChunkRevision(ctx context.Context, chunk *types.Chunk, revision *types.ChunkRevision, expectedRevision int) error
// ListChunkRevisions returns snapshots ordered newest first.
ListChunkRevisions(ctx context.Context, tenantID uint64, chunkID string) ([]*types.ChunkRevision, error)
// GetChunkRevision returns one historical snapshot.
GetChunkRevision(ctx context.Context, tenantID uint64, chunkID string, revision int) (*types.ChunkRevision, error)
// UpdateChunks updates chunks in batch
UpdateChunks(ctx context.Context, chunks []*types.Chunk) error
// SaveChunks persists full chunk objects in a single transaction using GORM Save (UPDATE).
@@ -148,4 +157,12 @@ type ChunkService interface {
// DeleteGeneratedQuestion deletes a single generated question from a chunk by question ID
// This updates the chunk metadata and removes the corresponding vector index
DeleteGeneratedQuestion(ctx context.Context, chunkID string, questionID string) error
// UpdateDocumentChunk applies a revision-checked edit and synchronizes retrieval indices.
UpdateDocumentChunk(ctx context.Context, chunkID string, content *string, isEnabled *bool, expectedRevision *int) (*types.Chunk, error)
// ListChunkRevisions lists immutable snapshots for a chunk.
ListChunkRevisions(ctx context.Context, chunkID string) ([]*types.ChunkRevision, error)
// RevertDocumentChunk restores a historical revision as a new current revision.
RevertDocumentChunk(ctx context.Context, chunkID string, revision int, expectedRevision *int) (*types.Chunk, error)
// UpsertGeneratedQuestion creates or updates a generated retrieval question.
UpsertGeneratedQuestion(ctx context.Context, chunkID string, questionID string, question string) (*types.GeneratedQuestion, error)
}
+4
View File
@@ -92,6 +92,10 @@ type KnowledgeService interface {
GetKnowledgeFile(ctx context.Context, id string) (io.ReadCloser, string, error)
// UpdateKnowledge updates knowledge information.
UpdateKnowledge(ctx context.Context, knowledge *types.Knowledge) error
// RegenerateKnowledgeSummary refreshes the document description and summary retrieval chunk.
RegenerateKnowledgeSummary(ctx context.Context, knowledgeID string) (*types.Knowledge, error)
// RegenerateChunkQuestions rebuilds the auxiliary questions for one current chunk revision.
RegenerateChunkQuestions(ctx context.Context, chunkID string) ([]types.GeneratedQuestion, error)
// UpdateManualKnowledge updates manual Markdown knowledge content.
UpdateManualKnowledge(
ctx context.Context,
+33
View File
@@ -3,6 +3,8 @@ package types
import (
"encoding/json"
"fmt"
"sort"
"strings"
"time"
"github.com/google/uuid"
@@ -154,6 +156,9 @@ type Knowledge struct {
StorageSize int64 `json:"storage_size"`
// Metadata of the knowledge
Metadata JSON `json:"metadata" gorm:"type:json"`
// CustomMetadata is user-authored descriptive metadata. It is deliberately
// separate from Metadata, which contains internal ingestion state and IDs.
CustomMetadata JSON `json:"custom_metadata" gorm:"type:json"`
// Last FAQ import result (for FAQ type knowledge only)
LastFAQImportResult JSON `json:"last_faq_import_result" gorm:"type:json"`
// Creation time of the knowledge
@@ -170,6 +175,34 @@ type Knowledge struct {
KnowledgeBaseName string `json:"knowledge_base_name" gorm:"-"`
}
// CustomMetadataText returns stable human-readable metadata for retrieval and
// model context. Internal ingestion metadata is intentionally excluded.
func (k *Knowledge) CustomMetadataText() string {
if k == nil || len(k.CustomMetadata) == 0 {
return ""
}
var values map[string]interface{}
if err := json.Unmarshal(k.CustomMetadata, &values); err != nil || len(values) == 0 {
return ""
}
keys := make([]string, 0, len(values))
for key := range values {
keys = append(keys, key)
}
sort.Strings(keys)
lines := make([]string, 0, len(keys))
for _, key := range keys {
if values[key] == nil {
continue
}
value := strings.TrimSpace(fmt.Sprint(values[key]))
if strings.TrimSpace(key) != "" && value != "" {
lines = append(lines, fmt.Sprintf("%s: %s", strings.TrimSpace(key), value))
}
}
return strings.Join(lines, "\n")
}
// GetMetadata returns the metadata as a map[string]string.
func (k *Knowledge) GetMetadata() map[string]string {
metadata := make(map[string]string)
@@ -0,0 +1,16 @@
package types
import (
"testing"
"github.com/stretchr/testify/require"
)
func TestKnowledgeCustomMetadataTextIsStableAndSeparate(t *testing.T) {
knowledge := &Knowledge{
Metadata: JSON(`{"external_id":"secret-internal-id"}`),
CustomMetadata: JSON(`{"region":"Shanghai","department":"R&D"}`),
}
require.Equal(t, "department: R&D\nregion: Shanghai", knowledge.CustomMetadataText())
require.NotContains(t, knowledge.CustomMetadataText(), "external_id")
}
+3
View File
@@ -202,6 +202,9 @@ type SearchResult struct {
// KnowledgeDescription is the description of the knowledge document
KnowledgeDescription string `json:"knowledge_description,omitempty"`
// KnowledgeCustomMetadata is user-authored context safe to expose to models.
KnowledgeCustomMetadata string `json:"knowledge_custom_metadata,omitempty"`
// KnowledgeBaseID is the ID of the knowledge base this result belongs to
KnowledgeBaseID string `json:"knowledge_base_id,omitempty"`
}
+23
View File
@@ -77,6 +77,7 @@ CREATE TABLE knowledges (
file_hash VARCHAR(64),
storage_size BIGINT NOT NULL DEFAULT 0,
metadata JSON,
custom_metadata JSON,
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
updated_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP ON UPDATE CURRENT_TIMESTAMP,
deleted_at TIMESTAMP NULL DEFAULT NULL,
@@ -185,6 +186,11 @@ CREATE TABLE chunks (
knowledge_base_id VARCHAR(36) NOT NULL,
knowledge_id VARCHAR(36) NOT NULL,
content TEXT NOT NULL,
source_content TEXT NOT NULL,
content_revision INT NOT NULL DEFAULT 0,
index_status VARCHAR(16) NOT NULL DEFAULT 'ready',
last_editor_id VARCHAR(64) NOT NULL DEFAULT '',
context_header TEXT NOT NULL,
chunk_index INTEGER NOT NULL,
is_enabled BOOLEAN NOT NULL DEFAULT TRUE,
start_at INTEGER NOT NULL,
@@ -204,3 +210,20 @@ CREATE TABLE chunks (
CREATE INDEX idx_chunks_tenant_knowledge ON chunks(tenant_id, knowledge_id);
CREATE INDEX idx_chunks_parent_id ON chunks(parent_chunk_id);
CREATE INDEX idx_chunks_chunk_type ON chunks(chunk_type);
CREATE TABLE chunk_revisions (
id VARCHAR(36) PRIMARY KEY,
tenant_id BIGINT NOT NULL,
knowledge_base_id VARCHAR(36) NOT NULL,
knowledge_id VARCHAR(36) NOT NULL,
chunk_id VARCHAR(36) NOT NULL,
revision INT NOT NULL,
content TEXT NOT NULL,
is_enabled BOOLEAN NOT NULL DEFAULT TRUE,
editor_id VARCHAR(64) NOT NULL DEFAULT '',
edit_source VARCHAR(16) NOT NULL DEFAULT 'user',
edited_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP,
created_at TIMESTAMP NOT NULL DEFAULT CURRENT_TIMESTAMP,
UNIQUE KEY idx_chunk_revisions_chunk_revision (chunk_id, revision),
KEY idx_chunk_revisions_tenant_chunk (tenant_id, chunk_id)
) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4;
+23
View File
@@ -105,6 +105,7 @@ CREATE TABLE IF NOT EXISTS knowledges (
file_hash VARCHAR(64),
storage_size BIGINT NOT NULL DEFAULT 0,
metadata TEXT,
custom_metadata TEXT NOT NULL DEFAULT '{}',
tag_id VARCHAR(36),
summary_status VARCHAR(32) DEFAULT 'none',
last_faq_import_result TEXT DEFAULT NULL,
@@ -242,6 +243,11 @@ CREATE TABLE IF NOT EXISTS chunks (
knowledge_base_id VARCHAR(36) NOT NULL,
knowledge_id VARCHAR(36) NOT NULL,
content TEXT NOT NULL,
source_content TEXT NOT NULL DEFAULT '',
content_revision INTEGER NOT NULL DEFAULT 0,
index_status VARCHAR(16) NOT NULL DEFAULT 'ready',
last_editor_id VARCHAR(64) NOT NULL DEFAULT '',
context_header TEXT NOT NULL DEFAULT '',
chunk_index INTEGER NOT NULL,
is_enabled BOOLEAN NOT NULL DEFAULT 1,
start_at INTEGER NOT NULL,
@@ -274,6 +280,23 @@ CREATE UNIQUE INDEX IF NOT EXISTS idx_chunks_seq_id ON chunks(seq_id);
CREATE INDEX IF NOT EXISTS idx_chunks_kb_tenant ON chunks(knowledge_base_id, tenant_id);
CREATE INDEX IF NOT EXISTS idx_chunks_knowledge_enabled ON chunks(knowledge_id, is_enabled, deleted_at);
CREATE TABLE IF NOT EXISTS chunk_revisions (
id VARCHAR(36) PRIMARY KEY,
tenant_id INTEGER NOT NULL,
knowledge_base_id VARCHAR(36) NOT NULL,
knowledge_id VARCHAR(36) NOT NULL,
chunk_id VARCHAR(36) NOT NULL,
revision INTEGER NOT NULL,
content TEXT NOT NULL DEFAULT '',
is_enabled BOOLEAN NOT NULL DEFAULT 1,
editor_id VARCHAR(64) NOT NULL DEFAULT '',
edit_source VARCHAR(16) NOT NULL DEFAULT 'user',
edited_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP,
created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP,
UNIQUE(chunk_id, revision)
);
CREATE INDEX IF NOT EXISTS idx_chunk_revisions_tenant_chunk ON chunk_revisions(tenant_id, chunk_id);
CREATE TABLE IF NOT EXISTS users (
id VARCHAR(36) PRIMARY KEY,
username VARCHAR(100) NOT NULL UNIQUE,
@@ -0,0 +1,8 @@
-- Roll back migration 000078.
DROP TABLE IF EXISTS chunk_revisions;
ALTER TABLE knowledges DROP COLUMN IF EXISTS custom_metadata;
ALTER TABLE chunks DROP COLUMN IF EXISTS context_header;
ALTER TABLE chunks DROP COLUMN IF EXISTS last_editor_id;
ALTER TABLE chunks DROP COLUMN IF EXISTS index_status;
ALTER TABLE chunks DROP COLUMN IF EXISTS content_revision;
ALTER TABLE chunks DROP COLUMN IF EXISTS source_content;
@@ -0,0 +1,26 @@
-- Migration 000078: editable chunks, revision history, and custom document metadata.
ALTER TABLE chunks ADD COLUMN IF NOT EXISTS source_content TEXT NOT NULL DEFAULT '';
ALTER TABLE chunks ADD COLUMN IF NOT EXISTS content_revision INT NOT NULL DEFAULT 0;
ALTER TABLE chunks ADD COLUMN IF NOT EXISTS index_status VARCHAR(16) NOT NULL DEFAULT 'ready';
ALTER TABLE chunks ADD COLUMN IF NOT EXISTS last_editor_id VARCHAR(64) NOT NULL DEFAULT '';
ALTER TABLE chunks ADD COLUMN IF NOT EXISTS context_header TEXT NOT NULL DEFAULT '';
UPDATE chunks SET source_content = content WHERE source_content = '';
ALTER TABLE knowledges ADD COLUMN IF NOT EXISTS custom_metadata JSONB NOT NULL DEFAULT '{}'::JSONB;
CREATE TABLE IF NOT EXISTS chunk_revisions (
id VARCHAR(36) PRIMARY KEY,
tenant_id BIGINT NOT NULL,
knowledge_base_id VARCHAR(36) NOT NULL,
knowledge_id VARCHAR(36) NOT NULL,
chunk_id VARCHAR(36) NOT NULL,
revision INT NOT NULL,
content TEXT NOT NULL DEFAULT '',
is_enabled BOOLEAN NOT NULL DEFAULT TRUE,
editor_id VARCHAR(64) NOT NULL DEFAULT '',
edit_source VARCHAR(16) NOT NULL DEFAULT 'user',
edited_at TIMESTAMP WITH TIME ZONE NOT NULL DEFAULT NOW(),
created_at TIMESTAMP WITH TIME ZONE NOT NULL DEFAULT NOW()
);
CREATE UNIQUE INDEX IF NOT EXISTS idx_chunk_revisions_chunk_revision ON chunk_revisions (chunk_id, revision);
CREATE INDEX IF NOT EXISTS idx_chunk_revisions_tenant_chunk ON chunk_revisions (tenant_id, chunk_id);