mirror of
https://github.com/stablyai/orca.git
synced 2026-10-01 08:01:56 +00:00
Fix browser history matching for workspace docs and recency rankings
Promote path-prefix matches to top tier for entries without a host (workspace docs), and clamp the recency bonus so future timestamps cannot outrank fresh visits. Includes tests for both path-prefix promotion and recency bonus clamping behavior.
This commit is contained in:
+18
@@ -338,4 +338,22 @@ describe('workspace document suggestions', () => {
|
||||
})
|
||||
expect(byPath.some((row) => row.docLocation)).toBe(true)
|
||||
})
|
||||
|
||||
it('ranks a path-prefix document above a heavily visited url-tail history match', () => {
|
||||
const rows = buildBrowserAddressBarSuggestions({
|
||||
value: '/repo/docs',
|
||||
browserUrlHistory: [
|
||||
{
|
||||
url: 'https://example.com/repo/docs/index.html',
|
||||
normalizedUrl: 'https://example.com/repo/docs/index.html',
|
||||
title: 'Docs Index',
|
||||
lastVisitedAt: Date.now(),
|
||||
visitCount: 500
|
||||
}
|
||||
],
|
||||
workspaceDocHistory: [DOC_ENTRY]
|
||||
})
|
||||
|
||||
expect(rows.find((row) => !row.isSearch)?.docLocation).toEqual(DOC_ENTRY.docLocation)
|
||||
})
|
||||
})
|
||||
|
||||
@@ -138,6 +138,29 @@ describe('browser history match', () => {
|
||||
expect(match([broken], 'brok')).toEqual([{ tier: 'title', url: 'not a url at all' }])
|
||||
})
|
||||
|
||||
it('promotes a path prefix to the top tier when the entry has no host', () => {
|
||||
const doc = entry({ url: '/repo/docs/report.html', title: 'Quarterly Report' })
|
||||
|
||||
expect(match([doc], '/repo/docs')).toEqual([
|
||||
{ tier: 'host-prefix', url: '/repo/docs/report.html' }
|
||||
])
|
||||
expect(match([doc], 'docs/report')).toEqual([
|
||||
{ tier: 'url-tail', url: '/repo/docs/report.html' }
|
||||
])
|
||||
})
|
||||
|
||||
it('clamps the recency bonus so a future lastVisitedAt cannot outrank a fresh visit', () => {
|
||||
const rows = match(
|
||||
[
|
||||
entry({ url: 'https://acme.dev/future', visitCount: 1, lastVisitedAt: NOW + 500 * HOUR }),
|
||||
entry({ url: 'https://acme.dev/now', visitCount: 2, lastVisitedAt: NOW })
|
||||
],
|
||||
'acme'
|
||||
)
|
||||
|
||||
expect(rows.map((row) => row.url)).toEqual(['https://acme.dev/now', 'https://acme.dev/future'])
|
||||
})
|
||||
|
||||
it('agrees with a naive reference scan on which entries match', () => {
|
||||
const corpus = Array.from({ length: 300 }, (_, index) =>
|
||||
entry({
|
||||
|
||||
@@ -66,7 +66,10 @@ function matchTier(
|
||||
prepared: PreparedBrowserHistoryEntry,
|
||||
lowerQuery: string
|
||||
): BrowserHistoryMatchTier | null {
|
||||
if (prepared.lowerHost.startsWith(lowerQuery)) {
|
||||
// A workspace-doc entry's url is a filesystem path, so it has no host: a path
|
||||
// prefix is as deliberate as a host prefix and earns the same top tier.
|
||||
const prefixTarget = prepared.lowerHost === '' ? prepared.lowerUrl : prepared.lowerHost
|
||||
if (prefixTarget.startsWith(lowerQuery)) {
|
||||
return 'host-prefix'
|
||||
}
|
||||
if (prepared.lowerHost.includes(lowerQuery)) {
|
||||
@@ -106,7 +109,10 @@ export function matchBrowserHistory({
|
||||
matches.push({
|
||||
entry: candidate.entry,
|
||||
tier,
|
||||
score: candidate.frecencyBase + Math.max(0, RECENCY_BONUS_HOURS - ageHours)
|
||||
// Clamped both ends: a future lastVisitedAt must not buy more than a fresh visit.
|
||||
score:
|
||||
candidate.frecencyBase +
|
||||
Math.min(RECENCY_BONUS_HOURS, Math.max(0, RECENCY_BONUS_HOURS - ageHours))
|
||||
})
|
||||
}
|
||||
if (matches.length === 0) {
|
||||
|
||||
Reference in New Issue
Block a user