From 54a77db95dc36a9bb59950f575d695b6a95d5457 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E7=A8=8B=E5=BA=8F=E5=91=98=E9=98=BF=E6=B1=9F-Relakkes?= Date: Sat, 3 Oct 2026 22:27:55 +0800 Subject: [PATCH] fix(desktop): read CJK file names in prose whole, settled by the disk (#1431) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The prose scan excludes CJK so a verb is not swallowed into a path, which cut an unquoted `测试文档1.docx` to `1.docx` and `D:/资料/测试文档1.docx` to `1.docx`. A match that is recognisably the tail of a CJK name is now widened to the whole token. Names the text cannot bound are settled against what exists: `报告v2.docx`, names glued together without a space, names with spaces or full-width brackets, and names made only of CJK plus an extension. The extractor offers the other readings; a changed file settles them, otherwise one cached workspace listing per folder does, longest existing name first. CJK-only names read like prose about formats (`后缀为.docx的文件`), so they are never linked and appear as cards or images only once confirmed. Fixes #1423 --- .../chat/AssistantMessage.filepaths.test.tsx | 33 ++++- .../src/components/chat/AssistantMessage.tsx | 7 +- .../chat/InlineImageGallery.test.tsx | 32 ++++ .../components/chat/InlineImageGallery.tsx | 29 ++-- .../hooks/useDiskConfirmedTargets.test.tsx | 139 +++++++++++++++++ desktop/src/hooks/useDiskConfirmedTargets.ts | 109 ++++++++++++++ .../src/lib/assistantOutputTargets.test.ts | 71 +++++++++ desktop/src/lib/assistantOutputTargets.ts | 73 ++++++++- desktop/src/lib/filePathBoundary.test.ts | 87 +++++++++++ desktop/src/lib/filePathBoundary.ts | 140 +++++++++++++++++- 10 files changed, 698 insertions(+), 22 deletions(-) create mode 100644 desktop/src/hooks/useDiskConfirmedTargets.test.tsx create mode 100644 desktop/src/hooks/useDiskConfirmedTargets.ts diff --git a/desktop/src/components/chat/AssistantMessage.filepaths.test.tsx b/desktop/src/components/chat/AssistantMessage.filepaths.test.tsx index 54cd45ee..85a51557 100644 --- a/desktop/src/components/chat/AssistantMessage.filepaths.test.tsx +++ b/desktop/src/components/chat/AssistantMessage.filepaths.test.tsx @@ -49,8 +49,9 @@ vi.mock('../../stores/workspaceContentStore', () => { }) const getWorkspaceFile = vi.hoisted(() => vi.fn().mockResolvedValue({ state: 'ok', content: 'file body' })) +const getWorkspaceTree = vi.hoisted(() => vi.fn().mockResolvedValue({ state: 'missing', path: '', entries: [] })) vi.mock('../../api/sessions', () => ({ - sessionsApi: { getWorkspaceFile }, + sessionsApi: { getWorkspaceFile, getWorkspaceTree }, })) const copyTextToClipboard = vi.hoisted(() => vi.fn().mockResolvedValue(true)) @@ -75,6 +76,7 @@ vi.mock('../../stores/settingsStore', () => ({ })) import { AssistantMessage } from './AssistantMessage' +import { resetDiskListingCacheForTests } from '../../hooks/useDiskConfirmedTargets' afterEach(() => { openPath.mockClear() @@ -85,6 +87,8 @@ afterEach(() => { openPreviewFn.mockReset().mockResolvedValue(undefined) copyTextToClipboard.mockReset().mockResolvedValue(true) getWorkspaceFile.mockReset().mockResolvedValue({ state: 'ok', content: 'file body' }) + getWorkspaceTree.mockReset().mockResolvedValue({ state: 'missing', path: '', entries: [] }) + resetDiskListingCacheForTests() }) describe('AssistantMessage file references', () => { @@ -134,6 +138,33 @@ describe('AssistantMessage file references', () => { await waitFor(() => expect(openPath).toHaveBeenCalledWith('/other/promo/public/audio/track.wav')) }) + it('names and opens the card by the file that exists when the prose could not bound it (#1423)', async () => { + getWorkspaceTree.mockResolvedValue({ + state: 'ok', + path: '', + entries: [{ name: '报告v2.docx', path: '报告v2.docx', isDirectory: false }], + }) + render() + + const card = await screen.findByText('报告v2.docx', { selector: 'span' }) + fireEvent.click(card.closest('button')!) + + await waitFor(() => expect(openPreviewFn).toHaveBeenCalledWith('s1', '报告v2.docx', {})) + }) + + it('shows a card for a CJK-and-extension name only once it is on disk, and links neither', async () => { + getWorkspaceTree.mockResolvedValue({ + state: 'ok', + path: '', + entries: [{ name: '开题报告.docx', path: '开题报告.docx', isDirectory: false }], + }) + render() + + expect(await screen.findByText('开题报告.docx', { selector: 'span' })).toBeInTheDocument() + expect(screen.queryByText(/后缀为\.docx/, { selector: 'span' })).not.toBeInTheDocument() + expect(screen.queryByRole('link', { name: /docx/ })).not.toBeInTheDocument() + }) + it('keeps prose and card destinations equal when the project root is stated later', async () => { render() fireEvent.click(screen.getByRole('link', { name: 'track.wav' })) diff --git a/desktop/src/components/chat/AssistantMessage.tsx b/desktop/src/components/chat/AssistantMessage.tsx index 6351813b..e4ad6a12 100644 --- a/desktop/src/components/chat/AssistantMessage.tsx +++ b/desktop/src/components/chat/AssistantMessage.tsx @@ -21,6 +21,7 @@ import { getServerBaseUrl } from '../../lib/desktopRuntime' import { isManagedGeneratedImagePath } from '../../lib/attachmentImages' import { useWorkspaceContentStore } from '../../stores/workspaceContentStore' import { useTranslation, type TranslationKey } from '../../i18n' +import { useDiskConfirmedTargets } from '../../hooks/useDiskConfirmedTargets' type Props = { content: string @@ -90,7 +91,7 @@ export const AssistantMessage = memo(function AssistantMessage({ [content, sessionId, t, workDir], ) - const outputTargets = useMemo( + const extractedTargets = useMemo( () => isStreaming || !sessionId ? [] @@ -99,11 +100,15 @@ export const AssistantMessage = memo(function AssistantMessage({ workDir, changedFiles: turnChangedFiles, includeChangedFileFallback: isTurnOutputOwner, + // Confirmed against the disk by useDiskConfirmedTargets before showing. + includeUnconfirmedNames: true, }).filter( (target) => target.kind !== 'image' && target.kind !== 'video', ), [content, isStreaming, isTurnOutputOwner, sessionId, workDir, turnChangedFiles], ) + // A bare name the text could not bound is settled against the workspace listing. + const outputTargets = useDiskConfirmedTargets(sessionId, extractedTargets) const resolveAssistantImageSrc = useMemo( () => { if (isStreaming || !sessionId) return undefined diff --git a/desktop/src/components/chat/InlineImageGallery.test.tsx b/desktop/src/components/chat/InlineImageGallery.test.tsx index a790eb51..507db689 100644 --- a/desktop/src/components/chat/InlineImageGallery.test.tsx +++ b/desktop/src/components/chat/InlineImageGallery.test.tsx @@ -19,9 +19,16 @@ vi.mock('../../lib/desktopRuntime', () => ({ const fetchServerImageBlobUrl = vi.hoisted(() => vi.fn()) vi.mock('../../lib/authedImage', () => ({ fetchServerImageBlobUrl })) +// Bare names glued to prose are settled against a workspace listing. +const getWorkspaceTree = vi.hoisted(() => vi.fn()) +vi.mock('../../api/sessions', () => ({ sessionsApi: { getWorkspaceTree } })) + import { InlineImageGallery } from './InlineImageGallery' +import { resetDiskListingCacheForTests } from '../../hooks/useDiskConfirmedTargets' beforeEach(() => { + resetDiskListingCacheForTests() + getWorkspaceTree.mockReset().mockResolvedValue({ state: 'missing', path: '', entries: [] }) fetchServerImageBlobUrl.mockReset().mockRejectedValue(new Error('403')) // jsdom ships no object-URL support. Object.defineProperty(URL, 'revokeObjectURL', { value: vi.fn(), configurable: true, writable: true }) @@ -219,6 +226,31 @@ describe('InlineImageGallery', () => { expect(await screen.findByRole('alert')).toHaveTextContent('frame.png') }) + it('shows the image the workspace really holds when a verb is glued to its name', async () => { + getWorkspaceTree.mockResolvedValue({ + state: 'ok', + path: '', + entries: [{ name: '1.png', path: '1.png', isDirectory: false }], + }) + render() + + await waitFor(() => expect(imgSrcs()).toEqual(['http://127.0.0.1:4321/preview-fs/s1/1.png'])) + }) + + it('shows a CJK-named image only once the workspace confirms it', async () => { + getWorkspaceTree.mockResolvedValue({ + state: 'ok', + path: '', + entries: [{ name: '流程图.png', path: '流程图.png', isDirectory: false }], + }) + render() + + expect(screen.queryAllByRole('img')).toHaveLength(0) + await waitFor(() => expect(imgSrcs()).toEqual([ + `http://127.0.0.1:4321/preview-fs/s1/${encodeURIComponent('流程图.png')}`, + ])) + }) + it('still raises the error block for a name the turn really wrote', async () => { render( sessionId + ? extractAssistantOutputTargets(text, { + workDir, + changedFiles: changedFileEvidence, + includeUnconfirmedNames: true, + }).filter( + (target) => ( + target.kind === 'image' && + target.source !== 'markdown-link' && + !markdownImageSources.has(normalizeImageReference(target.href)) && + !markdownImageSources.has(normalizeImageReference(target.normalizedPath ?? '')) + ), + ) + : [], [changedFileEvidence, markdownImageSources, sessionId, text, workDir]) + // `截图保存为1.png` may be `1.png` saved by a glued verb, or one CJK name; the + // workspace listing settles which, as it does for the output cards. + const relativeTargets = useDiskConfirmedTargets(sessionId, extractedRelativeTargets) + const images = useMemo(() => { // 1. Absolute paths (legacy behavior) — served via /api/filesystem/file. const absolute: GalleryImage[] = imagePaths.map((p) => ({ src: localImageFileUrl(p), name: fileName(p), path: p })) @@ -130,14 +149,6 @@ export function InlineImageGallery({ text, sessionId, workDir, changedFiles, sup // build a /preview-fs URL. Reuses the sandboxed target extractor instead of // a bespoke relative-path regex. const base = getServerBaseUrl() - const relativeTargets = extractAssistantOutputTargets(text, { workDir, changedFiles: changedFileEvidence }).filter( - (target) => ( - target.kind === 'image' && - target.source !== 'markdown-link' && - !markdownImageSources.has(normalizeImageReference(target.href)) && - !markdownImageSources.has(normalizeImageReference(target.normalizedPath ?? '')) - ), - ) // Dedup: an absolute path inside the workspace can be caught by BOTH sources. // Skip a relative target whose basename already appears among the absolute @@ -170,7 +181,7 @@ export function InlineImageGallery({ text, sessionId, workDir, changedFiles, sup } return [...absolute, ...relative] - }, [changedFileEvidence, imagePaths, markdownImageSources, sessionId, text, workDir]) + }, [changedFileEvidence, imagePaths, relativeTargets, sessionId, text, workDir]) // A guessed image that failed to load leaves no trace: there is nothing to retry // when the file was never claimed to be there. diff --git a/desktop/src/hooks/useDiskConfirmedTargets.test.tsx b/desktop/src/hooks/useDiskConfirmedTargets.test.tsx new file mode 100644 index 00000000..2fefc7a9 --- /dev/null +++ b/desktop/src/hooks/useDiskConfirmedTargets.test.tsx @@ -0,0 +1,139 @@ +import { renderHook, waitFor } from '@testing-library/react' +import { afterEach, beforeEach, describe, expect, it, vi, type MockInstance } from 'vitest' +import { sessionsApi, type WorkspaceTreeResult } from '@/api/sessions' +import { extractAssistantOutputTargets } from '@/lib/assistantOutputTargets' +import { resetDiskListingCacheForTests, useDiskConfirmedTargets } from './useDiskConfirmedTargets' + +let getWorkspaceTree: MockInstance + +function listing(path: string, files: string[]): WorkspaceTreeResult { + return { + state: 'ok', + path, + entries: files.map((name) => ({ name, path: path ? `${path}/${name}` : name, isDirectory: false })), + } as WorkspaceTreeResult +} + +function onDisk(files: Record) { + getWorkspaceTree.mockImplementation(async (_sessionId, dir = '') => listing(dir, files[dir] ?? [])) +} + +// No changed file corroborates anything, so every reading is left for the disk. +const cards = (content: string) => extractAssistantOutputTargets(content, { + workDir: '/w', + changedFiles: [], + includeUnconfirmedNames: true, +}) +const named = (targets: { title: string; href: string }[]) => targets.map((target) => [target.title, target.href]) + +beforeEach(() => { + resetDiskListingCacheForTests() + getWorkspaceTree = vi.spyOn(sessionsApi, 'getWorkspaceTree') +}) + +afterEach(() => { + getWorkspaceTree.mockRestore() +}) + +describe('useDiskConfirmedTargets', () => { + it('renames a card to the longest reading that exists on disk', async () => { + onDisk({ '': ['报告v2.docx', 'v2.docx.bak'] }) + const targets = cards('已生成报告v2.docx') + + const { result } = renderHook(() => useDiskConfirmedTargets('s1', targets)) + + expect(named(result.current)).toEqual([['v2.docx', 'v2.docx']]) + await waitFor(() => expect(named(result.current)).toEqual([['报告v2.docx', '报告v2.docx']])) + expect(getWorkspaceTree).toHaveBeenCalledTimes(1) + expect(getWorkspaceTree).toHaveBeenCalledWith('s1', '') + }) + + it('recovers names glued together and names with spaces and full-width brackets', async () => { + onDisk({ '': ['测试文档1.docx', '测试文档2.docx', '毕业设计(论文)任务书 张三.docx'] }) + const targets = cards('已找到测试文档1.docx和测试文档2.docx。任务书是 毕业设计(论文)任务书 张三.docx') + + const { result } = renderHook(() => useDiskConfirmedTargets('s1', targets)) + + await waitFor(() => expect(named(result.current)).toEqual([ + ['测试文档1.docx', '测试文档1.docx'], + ['测试文档2.docx', '测试文档2.docx'], + ['毕业设计(论文)任务书 张三.docx', '毕业设计(论文)任务书 张三.docx'], + ])) + }) + + it('prefers the longer name when a shorter one exists too', async () => { + // `1.docx` also sits in the folder, but the prose spelled out more of the name. + onDisk({ '': ['1.docx', '测试文档1.docx'] }) + const targets = cards('已找到测试文档1.docx') + + const { result } = renderHook(() => useDiskConfirmedTargets('s1', targets)) + + await waitFor(() => expect(named(result.current)).toEqual([['测试文档1.docx', '测试文档1.docx']])) + }) + + it('matches a name the file system stored decomposed (NFD)', async () => { + onDisk({ '': ['résumé报告v2.docx'.normalize('NFD')] }) + const targets = cards('见 résumé报告v2.docx') + + const { result } = renderHook(() => useDiskConfirmedTargets('s1', targets)) + + await waitFor(() => expect(result.current[0]?.title).toBe('résumé报告v2.docx')) + }) + + it('keeps the scan reading when nothing exists or the listing fails', async () => { + onDisk({ '': ['other.docx'] }) + const { result } = renderHook(() => useDiskConfirmedTargets('s1', cards('已生成报告v2.docx'))) + await waitFor(() => expect(getWorkspaceTree).toHaveBeenCalled()) + expect(named(result.current)).toEqual([['v2.docx', 'v2.docx']]) + + getWorkspaceTree.mockRejectedValue(new Error('403')) + const failed = renderHook(() => useDiskConfirmedTargets('s2', cards('已生成报告v3.docx'))) + await waitFor(() => expect(getWorkspaceTree).toHaveBeenCalledWith('s2', '')) + expect(named(failed.result.current)).toEqual([['v3.docx', 'v3.docx']]) + }) + + it('collapses two mentions that settle on the same file', async () => { + onDisk({ '': ['报告v2.docx'] }) + const targets = cards('已生成报告v2.docx,即 `报告v2.docx`') + + const { result } = renderHook(() => useDiskConfirmedTargets('s1', targets)) + + await waitFor(() => expect(named(result.current)).toEqual([['报告v2.docx', '报告v2.docx']])) + }) + + it('shows a CJK-and-extension name only once the disk confirms it', async () => { + onDisk({ '': ['开题报告.docx'] }) + const targets = cards('已生成 开题报告.docx。只支持后缀为.docx的文件') + + const { result } = renderHook(() => useDiskConfirmedTargets('s1', targets)) + + expect(result.current).toEqual([]) + await waitFor(() => expect(named(result.current)).toEqual([['开题报告.docx', '开题报告.docx']])) + }) + + it('never shows an unconfirmed name when the listing fails', async () => { + getWorkspaceTree.mockRejectedValue(new Error('403')) + const { result } = renderHook(() => useDiskConfirmedTargets('s1', cards('已生成 开题报告.docx'))) + + await waitFor(() => expect(getWorkspaceTree).toHaveBeenCalled()) + expect(result.current).toEqual([]) + }) + + it('lists a folder once for every message that names files in it', async () => { + onDisk({ '': ['报告v2.docx', '报告v3.docx'] }) + const first = renderHook(() => useDiskConfirmedTargets('s1', cards('已生成报告v2.docx'))) + const second = renderHook(() => useDiskConfirmedTargets('s1', cards('已生成报告v3.docx'))) + + await waitFor(() => expect(second.result.current[0]?.title).toBe('报告v3.docx')) + expect(first.result.current[0]?.title).toBe('报告v2.docx') + expect(getWorkspaceTree).toHaveBeenCalledTimes(1) + }) + + it('asks the disk nothing when every name is already bounded', () => { + const targets = cards('见 `报告v2.docx` 和 out/a.docx') + + renderHook(() => useDiskConfirmedTargets('s1', targets)) + + expect(getWorkspaceTree).not.toHaveBeenCalled() + }) +}) diff --git a/desktop/src/hooks/useDiskConfirmedTargets.ts b/desktop/src/hooks/useDiskConfirmedTargets.ts new file mode 100644 index 00000000..7e526cf4 --- /dev/null +++ b/desktop/src/hooks/useDiskConfirmedTargets.ts @@ -0,0 +1,109 @@ +import { useEffect, useMemo, useState } from 'react' +import { sessionsApi } from '@/api/sessions' +import type { AssistantOutputTarget } from '@/lib/assistantOutputTargets' + +type Listing = { key: string; namesByDir: Map> } + +const DIR_SEPARATOR = '\u0000' +const LISTING_TTL_MS = 5_000 + +// A long reply history mounts many messages naming files in the same folder; one +// listing serves them all for a few seconds, then a fresh one sees new files. +const listingCache = new Map | null> }>() + +export function resetDiskListingCacheForTests() { + listingCache.clear() +} + +function listFileNames(sessionId: string, directory: string): Promise | null> { + const cacheKey = `${sessionId}${DIR_SEPARATOR}${directory}` + const cached = listingCache.get(cacheKey) + if (cached && Date.now() - cached.at < LISTING_TTL_MS) return cached.names + + const names = sessionsApi.getWorkspaceTree(sessionId, directory) + .then((tree) => tree.state === 'ok' + ? new Set(tree.entries.filter((entry) => !entry.isDirectory).map((entry) => entry.name.normalize('NFC'))) + : null) + .catch(() => null) + listingCache.set(cacheKey, { at: Date.now(), names }) + return names +} + +function splitDirectory(path: string): [string, string] { + const cut = Math.max(path.lastIndexOf('/'), path.lastIndexOf('\\')) + return cut === -1 ? ['', path] : [path.slice(0, cut), path.slice(cut + 1)] +} + +/** + * Let the disk settle names the text could not bound. + * + * A bare prose mention such as `报告v2.docx` reaches the card as the scan's best + * reading (`v2.docx`) plus the other readings it may be. One workspace listing of + * each directory involved tells which of them exists; the longest that does wins, + * as it is the one the prose spelled out in full. Without an answer — no session, + * a failed listing, no reading on disk — the card keeps the scan's reading, and a + * target that {@link AssistantOutputTarget.awaitsConfirmation awaits confirmation} + * (`开题报告.docx`, which reads like `后缀为.docx的文件`) is not shown at all. + */ +export function useDiskConfirmedTargets( + sessionId: string | undefined, + targets: AssistantOutputTarget[], +): AssistantOutputTarget[] { + const directories = useMemo(() => [...new Set( + targets + .filter((target) => target.nameCandidates?.length || target.awaitsConfirmation) + .map((target) => splitDirectory(target.normalizedPath ?? target.href)[0]), + )].sort(), [targets]) + const key = sessionId && directories.length > 0 ? `${sessionId}${DIR_SEPARATOR}${directories.join(DIR_SEPARATOR)}` : '' + const [listing, setListing] = useState(null) + + useEffect(() => { + if (!key || !sessionId) return + let current = true + void Promise.all(directories.map(async (directory) => { + const names = await listFileNames(sessionId, directory) + return names ? [directory, names] as const : null + })).then((results) => { + if (!current) return + setListing({ key, namesByDir: new Map(results.filter((result) => result !== null)) }) + }) + return () => { + current = false + } + // `key` encodes the session and every directory. + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [key]) + + return useMemo(() => { + if (!listing || listing.key !== key) return targets.filter((target) => !target.awaitsConfirmation) + const seen = new Set() + return targets.map((target): AssistantOutputTarget | null => { + if (!target.nameCandidates?.length && !target.awaitsConfirmation) return target + const [directory, name] = splitDirectory(target.normalizedPath ?? target.href) + const names = listing.namesByDir.get(directory) + const confirmed = names && [...(target.nameCandidates ?? []), name] + .sort((left, right) => right.length - left.length) + .find((candidate) => names.has(candidate.normalize('NFC'))) + // A name only the disk could vouch for is not shown on a guess. + if (!confirmed) return target.awaitsConfirmation ? null : target + + const { awaitsConfirmation: _awaitsConfirmation, ...settled } = target + if (confirmed === name) return settled + const corrected = directory ? `${directory}/${confirmed}` : confirmed + return { + ...settled, + id: `${target.kind}:${corrected}`, + title: confirmed, + href: corrected, + normalizedPath: corrected, + subtitle: corrected, + } + }).filter((target): target is AssistantOutputTarget => { + if (!target) return false + // `v2.docx` and `报告v2.docx` in one reply can settle on the same file. + if (seen.has(target.id)) return false + seen.add(target.id) + return true + }) + }, [key, listing, targets]) +} diff --git a/desktop/src/lib/assistantOutputTargets.test.ts b/desktop/src/lib/assistantOutputTargets.test.ts index 7667ebc7..07e850f9 100644 --- a/desktop/src/lib/assistantOutputTargets.test.ts +++ b/desktop/src/lib/assistantOutputTargets.test.ts @@ -443,6 +443,77 @@ describe('extractAssistantOutputTargets with changedFiles reconciliation', () => .toEqual(['开题报告2.docx', '开题报告3.docx', '开题报告_v2.docx']) }) + describe('a name the text alone cannot bound', () => { + const names = (content: string, changedFiles: string[]) => + extractAssistantOutputTargets(content, { + workDir: '/w', + changedFiles, + includeChangedFileFallback: false, + includeUnconfirmedNames: true, + }) + .map((target) => [target.title, target.normalizedPath]) + + it('takes the longer reading the turn really wrote', () => { + expect(names('已生成报告v2.docx', ['/w/报告v2.docx'])).toEqual([['报告v2.docx', '报告v2.docx']]) + }) + + it('splits names glued together without a space', () => { + expect(names('已找到测试文档1.docx和测试文档2.docx', ['/w/测试文档1.docx', '/w/测试文档2.docx'])) + .toEqual([['测试文档1.docx', '测试文档1.docx'], ['测试文档2.docx', '测试文档2.docx']]) + }) + + it('recovers a name with spaces and full-width brackets', () => { + expect(names('已生成 毕业设计(论文)任务书 张三.docx', ['/w/毕业设计(论文)任务书 张三.docx'])) + .toEqual([['毕业设计(论文)任务书 张三.docx', '毕业设计(论文)任务书 张三.docx']]) + }) + + it('prefers the mention over a shorter name that also exists', () => { + expect(names('已找到 测试文档1.docx', ['/w/1.docx', '/w/测试文档1.docx'])) + .toEqual([['测试文档1.docx', '测试文档1.docx']]) + }) + + describe('a name made only of CJK and an extension', () => { + it('is not guessed at by default, as prose about formats looks the same', () => { + for (const content of ['已生成 开题报告.docx', '只支持后缀为.docx的文件']) { + expect(extractAssistantOutputTargets(content, { workDir: '/w', changedFiles: [] })).toEqual([]) + } + }) + + it('is offered for confirmation when asked, and settled by a changed file', () => { + const unconfirmed = extractAssistantOutputTargets('已生成 开题报告.docx', { + workDir: '/w', changedFiles: [], includeUnconfirmedNames: true, + }) + expect(unconfirmed).toMatchObject([{ normalizedPath: '开题报告.docx', awaitsConfirmation: true }]) + + const written = extractAssistantOutputTargets('已生成 开题报告.docx', { + workDir: '/w', changedFiles: ['/w/开题报告.docx'], includeUnconfirmedNames: true, + }) + expect(written).toHaveLength(1) + expect(written[0]).toMatchObject({ title: '开题报告.docx', normalizedPath: '开题报告.docx' }) + expect(written[0]!.awaitsConfirmation).toBeUndefined() + }) + }) + + it('leaves the readings open for the disk when no changed file settles them', () => { + const [target] = extractAssistantOutputTargets('已生成报告v2.docx', { workDir: '/w', changedFiles: [] }) + expect(target).toMatchObject({ normalizedPath: 'v2.docx' }) + expect(target!.nameCandidates).toEqual(expect.arrayContaining(['已生成报告v2.docx', '报告v2.docx'])) + }) + }) + + it('keeps the whole CJK name when the prose does not quote it (#1423)', () => { + const targets = extractAssistantOutputTargets( + '已找到 测试文档1.docx 和 测试文档2.docx,另一份在 C:\\Users\\a\\Desktop\\资料\\测试文档3.docx', + { workDir: 'C:\\Users\\a\\Desktop', changedFiles: [] }, + ) + + expect(targets.map((target) => [target.title, target.normalizedPath])).toEqual([ + ['测试文档1.docx', '测试文档1.docx'], + ['测试文档2.docx', '测试文档2.docx'], + ['测试文档3.docx', 'C:/Users/a/Desktop/资料/测试文档3.docx'], + ]) + }) + it('keeps a CJK directory', () => { const targets = extractAssistantOutputTargets( '见 `论文/开题报告终稿.docx`', diff --git a/desktop/src/lib/assistantOutputTargets.ts b/desktop/src/lib/assistantOutputTargets.ts index 9276e909..9f44d816 100644 --- a/desktop/src/lib/assistantOutputTargets.ts +++ b/desktop/src/lib/assistantOutputTargets.ts @@ -1,6 +1,12 @@ import { resolveAssistantFileHref } from './assistantFileContext' import { trimTrailingPunctuation } from './urlBoundary' -import { isLinkableFilePath, parseFilePathRef, splitTextByFilePaths } from './filePathBoundary' +import { + bareNameCandidates, + findUnicodeExtensionNames, + isLinkableFilePath, + parseFilePathRef, + splitTextByFilePaths, +} from './filePathBoundary' import { isGeneratedArtifactFile, isOutputResourceFile, isShellProducedDeliverable } from './fileCapabilities' export type AssistantOutputTargetKind = @@ -26,6 +32,14 @@ export type AssistantOutputTarget = { normalizedPath?: string confidence: 'high' source: AssistantOutputTargetSource + /** + * Other names a bare prose mention may really be (`报告v2.docx` for `v2.docx`), + * longest first, in the same directory. Text cannot choose between them; the + * changed files or the disk can — see {@link useDiskConfirmedTargets}. + */ + nameCandidates?: string[] + /** Not to be shown until a changed file or the disk confirms one of its names. */ + awaitsConfirmation?: boolean } export type ExtractAssistantOutputTargetOptions = { @@ -48,6 +62,13 @@ export type ExtractAssistantOutputTargetOptions = { * did not mention them. Reconciliation still runs when false. Defaults to true. */ includeChangedFileFallback?: boolean + /** + * Also return names made only of CJK and an extension (`开题报告.docx`), marked + * {@link AssistantOutputTarget.awaitsConfirmation}. Their text reads the same + * as prose about formats (`后缀为.docx的文件`), so only a caller that confirms + * them against the disk before showing them should ask. Defaults to false. + */ + includeUnconfirmedNames?: boolean } type FileTargetMatch = { @@ -206,7 +227,12 @@ export function extractAssistantOutputTargets( }, createFileKey(fileTarget), treeMatch.position) } - const queuePlainPath = (path: string, position: number) => { + const queuePlainPath = ( + path: string, + position: number, + nameCandidates?: string[], + awaitsConfirmation = false, + ) => { const href = resolveAssistantFileHref(path, content) const fileTarget = toWorkspaceFileTarget(href, workDir) @@ -223,6 +249,8 @@ export function extractAssistantOutputTargets( normalizedPath: fileTarget.normalizedPath, confidence: 'high', source: 'plain-path', + ...(nameCandidates?.length ? { nameCandidates } : {}), + ...(awaitsConfirmation ? { awaitsConfirmation } : {}), }, createFileKey(fileTarget), position) } @@ -240,6 +268,8 @@ export function extractAssistantOutputTargets( } let plainTextPosition = 0 + let previousPathEnd = 0 + const pathRanges: Array<{ start: number; end: number }> = [] for (const segment of splitTextByFilePaths(content)) { const position = plainTextPosition plainTextPosition += segment.value.length @@ -247,6 +277,9 @@ export function extractAssistantOutputTargets( if (segment.type !== 'path') { continue } + const floor = previousPathEnd + previousPathEnd = position + segment.value.length + pathRanges.push({ start: position, end: previousPathEnd }) if (isInMarkdownLink(position, markdownLinks)) { continue @@ -260,7 +293,25 @@ export function extractAssistantOutputTargets( continue } - queuePlainPath(segment.ref.path, position) + // Only a bare name is ambiguous at its edges; a path with a directory is not. + const nameCandidates = /[\\/]/.test(segment.ref.path) + ? undefined + : bareNameCandidates(content, position, position + segment.ref.path.length, floor) + queuePlainPath(segment.ref.path, position, nameCandidates) + } + + if (options.includeUnconfirmedNames) { + const overlaps = (start: number, end: number) => [...pathRanges, ...codeSpans] + .some((range) => start < range.end && end > range.start) + for (const name of findUnicodeExtensionNames(content)) { + if (overlaps(name.start, name.end)) continue + if (isInMarkdownLink(name.start, markdownLinks) || isInCodeBlock(name.start, codeBlocks)) continue + + const nameCandidates = /[\\/]/.test(name.ref.path) + ? undefined + : bareNameCandidates(content, name.start, name.start + name.ref.path.length, 0) + queuePlainPath(name.ref.path, name.start, nameCandidates, true) + } } candidates.sort((left, right) => { @@ -370,7 +421,14 @@ function reconcileTargetsWithChangedFiles( // An authored absolute path (including an explicit prose root) is identity, // not a basename hint. A checkpoint cannot disprove a shell-created output. const explicitPath = isAbsoluteFilePath(target.href) - const match = explicitPath ? null : matchChangedFile(mentioned, changedFiles) + // The longer reading of an ambiguous name wins when the turn really wrote it + // (`报告v2.docx`, not the `v2.docx` the prose scan settled on). + const match = explicitPath + ? null + : [...(target.nameCandidates ?? []), mentioned] + .sort((left, right) => getBasename(right).length - getBasename(left).length) + .map((name) => matchChangedFile(name, changedFiles)) + .find(Boolean) ?? null // Checkpoints record editing tools, not arbitrary shell output. Explicit // identities and document/media deliverables survive absent evidence; bare // source-like mentions still need corroboration to avoid resource-strip noise. @@ -391,12 +449,17 @@ function reconcileTargetsWithChangedFiles( } seen.add(key) + const { nameCandidates, awaitsConfirmation, ...rest } = target out.push({ - ...target, + ...rest, id: createId(target.kind, corrected), + ...(target.source === 'plain-path' ? { title: getBasename(corrected) } : {}), href: corrected, normalizedPath: corrected, subtitle: corrected, + // A changed-file match is settled; only an unmatched guess stays open. + ...(!match && nameCandidates ? { nameCandidates } : {}), + ...(!match && awaitsConfirmation ? { awaitsConfirmation } : {}), }) if (out.length >= limit) break } diff --git a/desktop/src/lib/filePathBoundary.test.ts b/desktop/src/lib/filePathBoundary.test.ts index 6219f370..2575b7de 100644 --- a/desktop/src/lib/filePathBoundary.test.ts +++ b/desktop/src/lib/filePathBoundary.test.ts @@ -1,6 +1,8 @@ import { describe, expect, it } from 'vitest' import { AMBIGUOUS_STANDALONE_EXTENSIONS, + bareNameCandidates, + findUnicodeExtensionNames, LINKABLE_FILE_EXTENSIONS, isFilePathOnly, isLinkableFilePath, @@ -104,6 +106,91 @@ describe('splitTextByFilePaths', () => { ]) }) + describe('a CJK file name in prose (#1423)', () => { + const paths = (text: string) => splitTextByFilePaths(text) + .filter((s) => s.type === 'path') + .map((s) => s.value) + + it.each([ + ['已找到 测试文档1.docx 和 测试文档2.docx', ['测试文档1.docx', '测试文档2.docx']], + ['- 测试文档1.docx', ['测试文档1.docx']], + ['| 测试文档1.docx | 12KB |', ['测试文档1.docx']], + ['已生成:开题报告_v2.docx。', ['开题报告_v2.docx']], + ['见 资料/report.docx', ['资料/report.docx']], + ['改了 src/中文/a.ts:3', ['src/中文/a.ts:3']], + ['已找到 D:/资料/测试文档1.docx', ['D:/资料/测试文档1.docx']], + ['已找到 C:\\Users\\a\\Desktop\\测试文档1.docx', ['C:\\Users\\a\\Desktop\\测试文档1.docx']], + ['在 /Users/a/论文/开题报告4.pdf', ['/Users/a/论文/开题报告4.pdf']], + ])('keeps the whole name: %s', (text, expected) => { + expect(paths(text)).toEqual(expected) + }) + + it('still leaves a CJK verb glued to an ASCII name in the sentence', () => { + expect(paths('修改了foo.ts')).toEqual(['foo.ts']) + expect(paths('修改了/Users/a/foo.ts')).toEqual(['/Users/a/foo.ts']) + }) + + it('preserves the original text exactly', () => { + const text = '已找到 测试文档1.docx,以及 D:/资料/测试文档2.docx。' + expect(splitTextByFilePaths(text).map((s) => s.value).join('')).toBe(text) + }) + }) + + it('links no name made only of CJK and an extension: prose about formats reads the same', () => { + // `开题报告.docx` and `后缀为.docx的文件` cannot be told apart by their text. + for (const text of ['已生成 开题报告.docx', '只支持后缀为.docx的文件', '把它另存为.pdf格式']) { + expect(splitTextByFilePaths(text).some((s) => s.type === 'path')).toBe(false) + } + }) + + describe('findUnicodeExtensionNames', () => { + const names = (text: string) => findUnicodeExtensionNames(text).map((match) => text.slice(match.start, match.end)) + + it('reads the whole token in front of the extension, for a caller that checks the disk', () => { + expect(names('已生成 开题报告.docx 和 摘要.pdf。')).toEqual(['开题报告.docx', '摘要.pdf']) + expect(names('只支持后缀为.docx的文件')).toEqual(['只支持后缀为.docx']) + }) + + it('leaves names the prose scan already reads, and non-file extensions', () => { + expect(names('见 测试文档1.docx、report.docx、Node.js、版本2.0')).toEqual([]) + }) + }) + + describe('bareNameCandidates', () => { + const at = (text: string, name: string) => { + const start = text.indexOf(name) + return bareNameCandidates(text, start, start + name.length, 0) + } + + it('offers the longer names a space or full-width bracket may belong to', () => { + const text = '任务书在 毕业设计(论文)任务书 张三.docx 里' + expect(at(text, '三.docx')).toEqual(expect.arrayContaining([ + '毕业设计(论文)任务书 张三.docx', + '任务书 张三.docx', + '张三.docx', + ])) + }) + + it('offers the shorter names at each CJK boundary of a glued token', () => { + const text = '已找到测试文档1.docx和测试文档2.docx' + const candidates = at(text, '已找到测试文档1.docx') + expect(candidates).toEqual(expect.arrayContaining(['测试文档1.docx', '1.docx'])) + expect(candidates).not.toContain('已找到测试文档1.docx') + }) + + it('lists longer names first and never crosses a sentence mark, a line or the previous path', () => { + const text = '前一句。\n见:报告 v2.docx' + const candidates = at(text, 'v2.docx') + expect(candidates[0]).toBe('报告 v2.docx') + expect(candidates.every((name) => !/[。:\n]/.test(name))).toBe(true) + expect(bareNameCandidates('a.docx 报告v2.docx', 9, 16, 6)).toEqual(['报告v2.docx', '告v2.docx']) + }) + + it('does not trim inside an ASCII word', () => { + expect(at('见 report.docx', 'report.docx')).toEqual(['见 report.docx']) + }) + }) + it('finds every path in a sentence', () => { const segments = splitTextByFilePaths('先看 a/b.ts:10,再看 c/d.py') expect(segments.filter((s) => s.type === 'path').map((s) => s.value)).toEqual(['a/b.ts:10', 'c/d.py']) diff --git a/desktop/src/lib/filePathBoundary.ts b/desktop/src/lib/filePathBoundary.ts index e46cbf1d..89b4acf8 100644 --- a/desktop/src/lib/filePathBoundary.ts +++ b/desktop/src/lib/filePathBoundary.ts @@ -20,6 +20,11 @@ * from `修`), while {@link parseFilePathRef} — an href we generated ourselves, * a backtick-quoted span — accepts them, because `README-拍摄大纲.md` is a real * file the turn really wrote and its delimiter is the span, not the sentence. + * When the ASCII match is recognisably the tail of a CJK name (`测试文档1.docx`), + * the prose scanner widens it to the whole token — see + * {@link widenAcrossUnicodeLetters}. A name with no ASCII part at all + * (`开题报告.docx`) reads exactly like prose about formats, so it is never linked + * here; {@link findUnicodeExtensionNames} offers it to callers that check the disk. * CJK *punctuation* is excluded in both — it is sentence material either way. */ @@ -314,15 +319,19 @@ export function splitTextByFilePaths(text: string): FilePathSegment[] { continue } - if (start > cursor) { - segments.push({ type: 'text', value: text.slice(cursor, start) }) + const widened = 'url' in found ? null : widenAcrossUnicodeLetters(text, start, found, cursor) + const pathStart = widened?.start ?? start + const ref = widened?.ref ?? found + + if (pathStart > cursor) { + segments.push({ type: 'text', value: text.slice(cursor, pathStart) }) } segments.push( - 'url' in found - ? { type: 'github', value: found.raw, ref: found } - : { type: 'path', value: found.raw, ref: found }, + 'url' in ref + ? { type: 'github', value: ref.raw, ref } + : { type: 'path', value: ref.raw, ref }, ) - cursor = start + found.raw.length + cursor = pathStart + ref.raw.length FILE_PATH_SCAN_RE.lastIndex = cursor } @@ -333,6 +342,125 @@ export function splitTextByFilePaths(text: string): FilePathSegment[] { return segments } +const UNICODE_LETTER_RE = /[^\x00-\x7F]/ +const IS_UNICODE_WORD_RE = /[\p{L}\p{N}]/u +const TOKEN_CHAR_RE = /[\p{L}\p{N}_.\-@+/\\~]/u + +/** + * The ASCII scan stops at the first CJK letter, so a match flush against one is + * either a path glued to a CJK verb (`修改了lib/foo.ts`, keep it) or the ASCII + * tail of a CJK name (`测试文档1.docx` → `1.docx`, #1423). A tail is recognisable: + * it opens with a digit or symbol, sits in a later segment of the token + * (`资料/测试文档v1.docx`), or is a lone `/name` cut from a relative path + * (`资料/report.docx`). Then the whole whitespace/punctuation-bounded token is + * read with the Unicode segment set, and used only if it is one linkable path + * ending exactly where the ASCII match did. + * + * A verb glued to a name that opens with an ASCII letter (`报告v2.docx`) stays + * ambiguous with `修改了foo.ts`, and is left as the ASCII match. + */ +function widenAcrossUnicodeLetters( + text: string, + start: number, + found: FilePathRef, + floor: number, +): { start: number; ref: FilePathRef } | null { + const previous = start > 0 ? text.charAt(start - 1) : '' + if (!UNICODE_LETTER_RE.test(previous) || !IS_UNICODE_WORD_RE.test(previous)) return null + + let tokenStart = start + while (tokenStart > floor && TOKEN_CHAR_RE.test(text.charAt(tokenStart - 1))) tokenStart -= 1 + // A drive root: `:` is not a token character, but `D:` opens the path. + if ( + tokenStart - 2 >= floor + && text.charAt(tokenStart - 1) === ':' + && /[A-Za-z]/.test(text.charAt(tokenStart - 2)) + && (tokenStart - 2 === floor || !TOKEN_CHAR_RE.test(text.charAt(tokenStart - 3))) + ) { + tokenStart -= 2 + } + + const leading = text.slice(tokenStart, start) + const rooted = /^(?:~|\.{1,2})?[\\/]|^[A-Za-z]:[\\/]/.test(leading) + const startsAtSeparator = /^[\\/]/.test(found.raw) + const isTail = rooted + || /[\\/]/.test(leading) + || /^[\d_.\-@+]/.test(found.raw) + || (startsAtSeparator && (found.path.match(/[\\/]/g)?.length ?? 0) === 1) + if (!isTail) return null + + const end = start + found.raw.length + const ref = parseFilePathRef(text.slice(tokenStart, end)) + return ref && ref.raw.length === end - tokenStart ? { start: tokenStart, ref } : null +} + +export type UnicodeExtensionName = { start: number; end: number; ref: FilePathRef } + +/** + * Names made only of non-ASCII letters and an extension: `开题报告.docx`. + * + * The prose scan never links these, because their text is the same as prose + * about formats — `只支持后缀为.docx的文件`, `另存为.pdf格式`. They are returned + * for a caller that can confirm them against the disk before showing anything. + * A name with an ASCII part in front of the dot (`测试文档1.docx`) is the prose + * scan's to read. + */ +export function findUnicodeExtensionNames(text: string): UnicodeExtensionName[] { + const names: UnicodeExtensionName[] = [] + let floor = 0 + + for (const match of text.matchAll(/\.[A-Za-z0-9]+(?![A-Za-z0-9_])/g)) { + const start = match.index ?? 0 + if (start < floor) continue + const previous = start > 0 ? text.charAt(start - 1) : '' + if (!UNICODE_LETTER_RE.test(previous) || !IS_UNICODE_WORD_RE.test(previous)) continue + + const widened = widenAcrossUnicodeLetters(text, start, { raw: match[0], path: match[0] }, floor) + if (!widened || 'url' in widened.ref) continue + const end = widened.start + widened.ref.raw.length + names.push({ start: widened.start, end, ref: widened.ref }) + floor = end + } + + return names +} + +const CANDIDATE_STOP_RE = /[\n\r|`*"'<>::,,.。;;、!!??“”‘’「」『』《》\[\]\\/]/ +const CANDIDATE_REACH = 60 + +/** + * Every name a bare file mention in prose may really be, longest first, for a + * caller that can check them against the disk. + * + * Text alone cannot settle `报告v2.docx` (verb glued to `v2.docx`, or one CJK + * name?), `测试文档1.docx和测试文档2.docx` (where does the second start?) or + * `毕业设计(论文)任务书 张三.docx` (spaces and full-width brackets are sentence + * material in the scan). So this widens leftwards across spaces and brackets up + * to a sentence mark, a line, `floor` or {@link CANDIDATE_REACH} characters, and + * trims rightwards at each CJK boundary inside the name. The extension never + * moves, and the mention itself (`start`..`end`) is not listed. + */ +export function bareNameCandidates(text: string, start: number, end: number, floor: number): string[] { + const name = text.slice(start, end) + const dot = name.lastIndexOf('.') + if (dot <= 0 && !name.startsWith('.')) return [] + + let left = start + const reach = Math.max(floor, start - CANDIDATE_REACH) + while (left > reach && !CANDIDATE_STOP_RE.test(text.charAt(left - 1))) left -= 1 + + const candidates: string[] = [] + for (let i = left; i < start; i += 1) { + if (!/\s/.test(text.charAt(i))) candidates.push(text.slice(i, end)) + } + for (let i = start + 1; i < start + Math.max(dot, 0); i += 1) { + if (UNICODE_LETTER_RE.test(text.charAt(i - 1)) && !/\s/.test(text.charAt(i))) { + candidates.push(text.slice(i, end)) + } + } + return candidates +} + /** * Parse a reference that is already known to be one — the href/data attribute * round-trip, and inline code spans.