2025-12-13 04:02:00 +08:00
|
|
|
/**
|
|
|
|
|
* CLI Integration Tests
|
|
|
|
|
*
|
|
|
|
|
* Tests all qmd CLI commands using a temporary test database via INDEX_PATH.
|
|
|
|
|
* These tests spawn actual qmd processes to verify end-to-end functionality.
|
|
|
|
|
*/
|
|
|
|
|
|
2026-02-16 04:57:13 +08:00
|
|
|
import { describe, test, expect, beforeAll, afterAll, beforeEach } from "vitest";
|
2025-12-13 04:02:00 +08:00
|
|
|
import { mkdtemp, rm, writeFile, mkdir } from "fs/promises";
|
2026-02-21 09:52:53 +08:00
|
|
|
import { existsSync, readFileSync, writeFileSync, unlinkSync } from "fs";
|
2025-12-13 04:02:00 +08:00
|
|
|
import { tmpdir } from "os";
|
2026-02-16 04:57:13 +08:00
|
|
|
import { join, dirname } from "path";
|
|
|
|
|
import { fileURLToPath } from "url";
|
|
|
|
|
import { spawn } from "child_process";
|
|
|
|
|
import { setTimeout as sleep } from "timers/promises";
|
2025-12-13 04:02:00 +08:00
|
|
|
|
|
|
|
|
// Test fixtures directory and database path
|
|
|
|
|
let testDir: string;
|
|
|
|
|
let testDbPath: string;
|
2025-12-14 02:28:50 +08:00
|
|
|
let testConfigDir: string;
|
2025-12-13 04:02:00 +08:00
|
|
|
let fixturesDir: string;
|
|
|
|
|
let testCounter = 0; // Unique counter for each test run
|
|
|
|
|
|
2026-02-16 09:46:45 +08:00
|
|
|
// Get the directory where this test file lives
|
|
|
|
|
const thisDir = dirname(fileURLToPath(import.meta.url));
|
|
|
|
|
const projectRoot = join(thisDir, "..");
|
|
|
|
|
const qmdScript = join(projectRoot, "src", "qmd.ts");
|
2026-02-16 04:57:13 +08:00
|
|
|
// Resolve tsx binary from project's node_modules (not cwd-dependent)
|
|
|
|
|
const tsxBin = (() => {
|
|
|
|
|
const candidate = join(projectRoot, "node_modules", ".bin", "tsx");
|
|
|
|
|
if (existsSync(candidate)) {
|
|
|
|
|
return candidate;
|
|
|
|
|
}
|
|
|
|
|
return join(process.cwd(), "node_modules", ".bin", "tsx");
|
|
|
|
|
})();
|
2025-12-13 04:02:00 +08:00
|
|
|
|
|
|
|
|
// Helper to run qmd command with test database
|
|
|
|
|
async function runQmd(
|
|
|
|
|
args: string[],
|
Add path normalization, output format tests, and fix test isolation
- Add support for collection/path.md format in get command (checks if
first component is a known collection before treating as filesystem path)
- Add comprehensive output format tests verifying qmd:// URIs, docid,
and context in JSON, CSV, MD, XML, files, and CLI formats
- Add path normalization tests for various input formats:
qmd://, //, qmd:////, collection/path, and path:line suffix
- Add isolated test environments (createIsolatedTestEnv) to prevent
YAML config conflicts between test suites
- Add test fixture files test1.md and test2.md for path tests
- Update runQmd helper to accept custom configDir parameter
🤖 Generated with [Claude Code](https://claude.com/claude-code)
Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2025-12-21 02:45:18 +08:00
|
|
|
options: { cwd?: string; env?: Record<string, string>; dbPath?: string; configDir?: string } = {}
|
2025-12-13 04:02:00 +08:00
|
|
|
): Promise<{ stdout: string; stderr: string; exitCode: number }> {
|
|
|
|
|
const workingDir = options.cwd || fixturesDir;
|
|
|
|
|
const dbPath = options.dbPath || testDbPath;
|
Add path normalization, output format tests, and fix test isolation
- Add support for collection/path.md format in get command (checks if
first component is a known collection before treating as filesystem path)
- Add comprehensive output format tests verifying qmd:// URIs, docid,
and context in JSON, CSV, MD, XML, files, and CLI formats
- Add path normalization tests for various input formats:
qmd://, //, qmd:////, collection/path, and path:line suffix
- Add isolated test environments (createIsolatedTestEnv) to prevent
YAML config conflicts between test suites
- Add test fixture files test1.md and test2.md for path tests
- Update runQmd helper to accept custom configDir parameter
🤖 Generated with [Claude Code](https://claude.com/claude-code)
Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2025-12-21 02:45:18 +08:00
|
|
|
const configDir = options.configDir || testConfigDir;
|
2026-02-16 04:57:13 +08:00
|
|
|
const proc = spawn(tsxBin, [qmdScript, ...args], {
|
2025-12-13 04:02:00 +08:00
|
|
|
cwd: workingDir,
|
|
|
|
|
env: {
|
|
|
|
|
...process.env,
|
|
|
|
|
INDEX_PATH: dbPath,
|
Add path normalization, output format tests, and fix test isolation
- Add support for collection/path.md format in get command (checks if
first component is a known collection before treating as filesystem path)
- Add comprehensive output format tests verifying qmd:// URIs, docid,
and context in JSON, CSV, MD, XML, files, and CLI formats
- Add path normalization tests for various input formats:
qmd://, //, qmd:////, collection/path, and path:line suffix
- Add isolated test environments (createIsolatedTestEnv) to prevent
YAML config conflicts between test suites
- Add test fixture files test1.md and test2.md for path tests
- Update runQmd helper to accept custom configDir parameter
🤖 Generated with [Claude Code](https://claude.com/claude-code)
Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2025-12-21 02:45:18 +08:00
|
|
|
QMD_CONFIG_DIR: configDir, // Use test config directory
|
2025-12-13 04:02:00 +08:00
|
|
|
PWD: workingDir, // Must explicitly set PWD since getPwd() checks this
|
|
|
|
|
...options.env,
|
|
|
|
|
},
|
2026-02-16 04:57:13 +08:00
|
|
|
stdio: ["ignore", "pipe", "pipe"],
|
2025-12-13 04:02:00 +08:00
|
|
|
});
|
|
|
|
|
|
2026-02-16 04:57:13 +08:00
|
|
|
const stdoutPromise = new Promise<string>((resolve, reject) => {
|
|
|
|
|
let data = "";
|
|
|
|
|
proc.stdout?.on("data", (chunk: Buffer) => { data += chunk.toString(); });
|
|
|
|
|
proc.once("error", reject);
|
|
|
|
|
proc.stdout?.once("end", () => resolve(data));
|
|
|
|
|
});
|
|
|
|
|
const stderrPromise = new Promise<string>((resolve, reject) => {
|
|
|
|
|
let data = "";
|
|
|
|
|
proc.stderr?.on("data", (chunk: Buffer) => { data += chunk.toString(); });
|
|
|
|
|
proc.once("error", reject);
|
|
|
|
|
proc.stderr?.once("end", () => resolve(data));
|
|
|
|
|
});
|
|
|
|
|
const exitCode = await new Promise<number>((resolve, reject) => {
|
|
|
|
|
proc.once("error", reject);
|
|
|
|
|
proc.on("close", (code) => resolve(code ?? 1));
|
|
|
|
|
});
|
|
|
|
|
const stdout = await stdoutPromise;
|
|
|
|
|
const stderr = await stderrPromise;
|
2025-12-13 04:02:00 +08:00
|
|
|
|
|
|
|
|
return { stdout, stderr, exitCode };
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Get a fresh database path for isolated tests
|
|
|
|
|
function getFreshDbPath(): string {
|
|
|
|
|
testCounter++;
|
|
|
|
|
return join(testDir, `test-${testCounter}.sqlite`);
|
|
|
|
|
}
|
|
|
|
|
|
Add path normalization, output format tests, and fix test isolation
- Add support for collection/path.md format in get command (checks if
first component is a known collection before treating as filesystem path)
- Add comprehensive output format tests verifying qmd:// URIs, docid,
and context in JSON, CSV, MD, XML, files, and CLI formats
- Add path normalization tests for various input formats:
qmd://, //, qmd:////, collection/path, and path:line suffix
- Add isolated test environments (createIsolatedTestEnv) to prevent
YAML config conflicts between test suites
- Add test fixture files test1.md and test2.md for path tests
- Update runQmd helper to accept custom configDir parameter
🤖 Generated with [Claude Code](https://claude.com/claude-code)
Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2025-12-21 02:45:18 +08:00
|
|
|
// Create an isolated test environment (db + config dir)
|
|
|
|
|
async function createIsolatedTestEnv(prefix: string): Promise<{ dbPath: string; configDir: string }> {
|
|
|
|
|
testCounter++;
|
|
|
|
|
const dbPath = join(testDir, `${prefix}-${testCounter}.sqlite`);
|
|
|
|
|
const configDir = join(testDir, `${prefix}-config-${testCounter}`);
|
|
|
|
|
await mkdir(configDir, { recursive: true });
|
|
|
|
|
await writeFile(join(configDir, "index.yml"), "collections: {}\n");
|
|
|
|
|
return { dbPath, configDir };
|
|
|
|
|
}
|
|
|
|
|
|
2025-12-13 04:02:00 +08:00
|
|
|
// Setup test fixtures
|
|
|
|
|
beforeAll(async () => {
|
|
|
|
|
// Create temp directory structure
|
|
|
|
|
testDir = await mkdtemp(join(tmpdir(), "qmd-test-"));
|
|
|
|
|
testDbPath = join(testDir, "test.sqlite");
|
2025-12-14 02:28:50 +08:00
|
|
|
testConfigDir = join(testDir, "config");
|
2025-12-13 04:02:00 +08:00
|
|
|
fixturesDir = join(testDir, "fixtures");
|
|
|
|
|
|
2025-12-14 02:28:50 +08:00
|
|
|
await mkdir(testConfigDir, { recursive: true });
|
2025-12-13 04:02:00 +08:00
|
|
|
await mkdir(fixturesDir, { recursive: true });
|
|
|
|
|
await mkdir(join(fixturesDir, "notes"), { recursive: true });
|
|
|
|
|
await mkdir(join(fixturesDir, "docs"), { recursive: true });
|
|
|
|
|
|
2025-12-14 02:28:50 +08:00
|
|
|
// Create empty YAML config for tests
|
|
|
|
|
await writeFile(
|
|
|
|
|
join(testConfigDir, "index.yml"),
|
|
|
|
|
"collections: {}\n"
|
|
|
|
|
);
|
|
|
|
|
|
2025-12-13 04:02:00 +08:00
|
|
|
// Create test markdown files
|
|
|
|
|
await writeFile(
|
|
|
|
|
join(fixturesDir, "README.md"),
|
|
|
|
|
`# Test Project
|
|
|
|
|
|
|
|
|
|
This is a test project for QMD CLI testing.
|
|
|
|
|
|
|
|
|
|
## Features
|
|
|
|
|
|
|
|
|
|
- Full-text search with BM25
|
|
|
|
|
- Vector similarity search
|
|
|
|
|
- Hybrid search with reranking
|
|
|
|
|
`
|
|
|
|
|
);
|
|
|
|
|
|
|
|
|
|
await writeFile(
|
|
|
|
|
join(fixturesDir, "notes", "meeting.md"),
|
|
|
|
|
`# Team Meeting Notes
|
|
|
|
|
|
|
|
|
|
Date: 2024-01-15
|
|
|
|
|
|
|
|
|
|
## Attendees
|
|
|
|
|
- Alice
|
|
|
|
|
- Bob
|
|
|
|
|
- Charlie
|
|
|
|
|
|
|
|
|
|
## Discussion Topics
|
|
|
|
|
- Project timeline review
|
|
|
|
|
- Resource allocation
|
|
|
|
|
- Technical debt prioritization
|
|
|
|
|
|
|
|
|
|
## Action Items
|
|
|
|
|
1. Alice to update documentation
|
|
|
|
|
2. Bob to fix authentication bug
|
|
|
|
|
3. Charlie to review pull requests
|
|
|
|
|
`
|
|
|
|
|
);
|
|
|
|
|
|
|
|
|
|
await writeFile(
|
|
|
|
|
join(fixturesDir, "notes", "ideas.md"),
|
|
|
|
|
`# Product Ideas
|
|
|
|
|
|
|
|
|
|
## Feature Requests
|
|
|
|
|
- Dark mode support
|
|
|
|
|
- Keyboard shortcuts
|
|
|
|
|
- Export to PDF
|
|
|
|
|
|
|
|
|
|
## Technical Improvements
|
|
|
|
|
- Improve search performance
|
|
|
|
|
- Add caching layer
|
|
|
|
|
- Optimize database queries
|
|
|
|
|
`
|
|
|
|
|
);
|
|
|
|
|
|
|
|
|
|
await writeFile(
|
|
|
|
|
join(fixturesDir, "docs", "api.md"),
|
|
|
|
|
`# API Documentation
|
|
|
|
|
|
|
|
|
|
## Endpoints
|
|
|
|
|
|
|
|
|
|
### GET /search
|
|
|
|
|
Search for documents.
|
|
|
|
|
|
|
|
|
|
Parameters:
|
|
|
|
|
- q: Search query (required)
|
|
|
|
|
- limit: Max results (default: 10)
|
|
|
|
|
|
|
|
|
|
### GET /document/:id
|
|
|
|
|
Retrieve a specific document.
|
|
|
|
|
|
|
|
|
|
### POST /index
|
|
|
|
|
Index new documents.
|
Add path normalization, output format tests, and fix test isolation
- Add support for collection/path.md format in get command (checks if
first component is a known collection before treating as filesystem path)
- Add comprehensive output format tests verifying qmd:// URIs, docid,
and context in JSON, CSV, MD, XML, files, and CLI formats
- Add path normalization tests for various input formats:
qmd://, //, qmd:////, collection/path, and path:line suffix
- Add isolated test environments (createIsolatedTestEnv) to prevent
YAML config conflicts between test suites
- Add test fixture files test1.md and test2.md for path tests
- Update runQmd helper to accept custom configDir parameter
🤖 Generated with [Claude Code](https://claude.com/claude-code)
Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2025-12-21 02:45:18 +08:00
|
|
|
`
|
|
|
|
|
);
|
|
|
|
|
|
|
|
|
|
// Create test files for path normalization tests
|
|
|
|
|
await writeFile(
|
|
|
|
|
join(fixturesDir, "test1.md"),
|
|
|
|
|
`# Test Document 1
|
|
|
|
|
|
|
|
|
|
This is the first test document.
|
|
|
|
|
|
|
|
|
|
It has multiple lines for testing line numbers.
|
|
|
|
|
Line 6 is here.
|
|
|
|
|
Line 7 is here.
|
|
|
|
|
`
|
|
|
|
|
);
|
|
|
|
|
|
|
|
|
|
await writeFile(
|
|
|
|
|
join(fixturesDir, "test2.md"),
|
|
|
|
|
`# Test Document 2
|
|
|
|
|
|
|
|
|
|
This is the second test document.
|
2025-12-13 04:02:00 +08:00
|
|
|
`
|
|
|
|
|
);
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
// Cleanup after all tests
|
|
|
|
|
afterAll(async () => {
|
|
|
|
|
if (testDir) {
|
|
|
|
|
await rm(testDir, { recursive: true, force: true });
|
|
|
|
|
}
|
|
|
|
|
});
|
|
|
|
|
|
2025-12-14 02:40:20 +08:00
|
|
|
// Reset YAML config before each test to ensure isolation
|
|
|
|
|
beforeEach(async () => {
|
|
|
|
|
// Reset to empty collections config
|
|
|
|
|
await writeFile(
|
|
|
|
|
join(testConfigDir, "index.yml"),
|
|
|
|
|
"collections: {}\n"
|
|
|
|
|
);
|
|
|
|
|
});
|
|
|
|
|
|
2025-12-13 04:02:00 +08:00
|
|
|
describe("CLI Help", () => {
|
|
|
|
|
test("shows help with --help flag", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["--help"]);
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("Usage:");
|
2025-12-13 05:07:01 +08:00
|
|
|
expect(stdout).toContain("qmd collection add");
|
2025-12-13 04:02:00 +08:00
|
|
|
expect(stdout).toContain("qmd search");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("shows help with no arguments", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd([]);
|
|
|
|
|
expect(exitCode).toBe(1);
|
|
|
|
|
expect(stdout).toContain("Usage:");
|
|
|
|
|
});
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
describe("CLI Add Command", () => {
|
|
|
|
|
test("adds files from current directory", async () => {
|
2025-12-13 05:07:01 +08:00
|
|
|
const { stdout, exitCode } = await runQmd(["collection", "add", "."]);
|
2025-12-13 04:02:00 +08:00
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("Collection:");
|
|
|
|
|
expect(stdout).toContain("Indexed:");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("adds files with custom glob pattern", async () => {
|
2025-12-14 02:40:20 +08:00
|
|
|
const { stdout, stderr, exitCode } = await runQmd(["collection", "add", ".", "--mask", "notes/*.md"]);
|
|
|
|
|
if (exitCode !== 0) {
|
|
|
|
|
console.error("Command failed:", stderr);
|
|
|
|
|
}
|
2025-12-13 04:02:00 +08:00
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("Collection:");
|
|
|
|
|
// Should find meeting.md and ideas.md in notes/
|
|
|
|
|
expect(stdout).toContain("notes/*.md");
|
|
|
|
|
});
|
|
|
|
|
|
2025-12-13 05:07:01 +08:00
|
|
|
test("can recreate collection with remove and add", async () => {
|
2025-12-13 04:02:00 +08:00
|
|
|
// First add
|
2025-12-13 05:07:01 +08:00
|
|
|
await runQmd(["collection", "add", "."]);
|
|
|
|
|
// Remove it
|
|
|
|
|
await runQmd(["collection", "remove", "fixtures"]);
|
|
|
|
|
// Re-add
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["collection", "add", "."]);
|
2025-12-13 04:02:00 +08:00
|
|
|
expect(exitCode).toBe(0);
|
2025-12-13 05:07:01 +08:00
|
|
|
expect(stdout).toContain("Collection 'fixtures' created successfully");
|
2025-12-13 04:02:00 +08:00
|
|
|
});
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
describe("CLI Status Command", () => {
|
|
|
|
|
beforeEach(async () => {
|
|
|
|
|
// Ensure we have indexed files
|
2025-12-13 05:07:01 +08:00
|
|
|
await runQmd(["collection", "add", "."]);
|
2025-12-13 04:02:00 +08:00
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("shows index status", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["status"]);
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
// Should show collection info
|
|
|
|
|
expect(stdout).toContain("Collection");
|
|
|
|
|
});
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
describe("CLI Search Command", () => {
|
|
|
|
|
beforeEach(async () => {
|
|
|
|
|
// Ensure we have indexed files
|
2025-12-13 05:07:01 +08:00
|
|
|
await runQmd(["collection", "add", "."]);
|
2025-12-13 04:02:00 +08:00
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("searches for documents with BM25", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["search", "meeting"]);
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
// Should find meeting.md
|
|
|
|
|
expect(stdout.toLowerCase()).toContain("meeting");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("searches with limit option", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["search", "-n", "1", "test"]);
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("searches with all results option", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["search", "--all", "the"]);
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("returns no results message for non-matching query", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["search", "xyznonexistent123"]);
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("No results");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("requires query argument", async () => {
|
|
|
|
|
const { stdout, stderr, exitCode } = await runQmd(["search"]);
|
|
|
|
|
expect(exitCode).toBe(1);
|
|
|
|
|
// Error message goes to stderr
|
|
|
|
|
expect(stderr).toContain("Usage:");
|
|
|
|
|
});
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
describe("CLI Get Command", () => {
|
|
|
|
|
beforeEach(async () => {
|
|
|
|
|
// Ensure we have indexed files
|
2025-12-13 05:07:01 +08:00
|
|
|
await runQmd(["collection", "add", "."]);
|
2025-12-13 04:02:00 +08:00
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("retrieves document content by path", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["get", "README.md"]);
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("Test Project");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("retrieves document from subdirectory", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["get", "notes/meeting.md"]);
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("Team Meeting");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("handles non-existent file", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["get", "nonexistent.md"]);
|
|
|
|
|
// Should indicate file not found
|
|
|
|
|
expect(exitCode).toBe(1);
|
|
|
|
|
});
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
describe("CLI Multi-Get Command", () => {
|
2025-12-15 05:51:16 +08:00
|
|
|
let localDbPath: string;
|
|
|
|
|
|
2025-12-13 04:02:00 +08:00
|
|
|
beforeEach(async () => {
|
2025-12-15 05:51:16 +08:00
|
|
|
// Use fresh database for each test
|
|
|
|
|
localDbPath = getFreshDbPath();
|
2025-12-13 04:02:00 +08:00
|
|
|
// Ensure we have indexed files
|
2025-12-15 05:51:16 +08:00
|
|
|
const addResult = await runQmd(["collection", "add", ".", "--name", "fixtures"], { dbPath: localDbPath });
|
|
|
|
|
if (addResult.exitCode !== 0) {
|
|
|
|
|
throw new Error(`Failed to add collection: ${addResult.stderr}`);
|
|
|
|
|
}
|
2025-12-13 04:02:00 +08:00
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("retrieves multiple documents by pattern", async () => {
|
2025-12-15 05:51:16 +08:00
|
|
|
// Test glob pattern matching
|
|
|
|
|
const { stdout, stderr, exitCode } = await runQmd(["multi-get", "notes/*.md"], { dbPath: localDbPath });
|
2025-12-13 04:02:00 +08:00
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
// Should contain content from both notes files
|
|
|
|
|
expect(stdout).toContain("Meeting");
|
|
|
|
|
expect(stdout).toContain("Ideas");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("retrieves documents by comma-separated paths", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd([
|
|
|
|
|
"multi-get",
|
|
|
|
|
"README.md,notes/meeting.md",
|
2025-12-15 05:51:16 +08:00
|
|
|
], { dbPath: localDbPath });
|
2025-12-13 04:02:00 +08:00
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("Test Project");
|
|
|
|
|
expect(stdout).toContain("Team Meeting");
|
|
|
|
|
});
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
describe("CLI Update Command", () => {
|
|
|
|
|
let localDbPath: string;
|
|
|
|
|
|
|
|
|
|
beforeEach(async () => {
|
|
|
|
|
// Use a fresh database for this test suite
|
|
|
|
|
localDbPath = getFreshDbPath();
|
|
|
|
|
// Ensure we have indexed files
|
2025-12-13 05:07:01 +08:00
|
|
|
await runQmd(["collection", "add", "."], { dbPath: localDbPath });
|
2025-12-13 04:02:00 +08:00
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("updates all collections", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["update"], { dbPath: localDbPath });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("Updating");
|
|
|
|
|
});
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
describe("CLI Add-Context Command", () => {
|
Add path normalization, output format tests, and fix test isolation
- Add support for collection/path.md format in get command (checks if
first component is a known collection before treating as filesystem path)
- Add comprehensive output format tests verifying qmd:// URIs, docid,
and context in JSON, CSV, MD, XML, files, and CLI formats
- Add path normalization tests for various input formats:
qmd://, //, qmd:////, collection/path, and path:line suffix
- Add isolated test environments (createIsolatedTestEnv) to prevent
YAML config conflicts between test suites
- Add test fixture files test1.md and test2.md for path tests
- Update runQmd helper to accept custom configDir parameter
🤖 Generated with [Claude Code](https://claude.com/claude-code)
Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2025-12-21 02:45:18 +08:00
|
|
|
let localDbPath: string;
|
|
|
|
|
let localConfigDir: string;
|
|
|
|
|
const collName = "fixtures";
|
|
|
|
|
|
|
|
|
|
beforeAll(async () => {
|
|
|
|
|
const env = await createIsolatedTestEnv("context-cmd");
|
|
|
|
|
localDbPath = env.dbPath;
|
|
|
|
|
localConfigDir = env.configDir;
|
|
|
|
|
|
|
|
|
|
// Add collection with known name
|
|
|
|
|
const { exitCode, stderr } = await runQmd(
|
|
|
|
|
["collection", "add", fixturesDir, "--name", collName],
|
|
|
|
|
{ dbPath: localDbPath, configDir: localConfigDir }
|
|
|
|
|
);
|
|
|
|
|
if (exitCode !== 0) console.error("collection add failed:", stderr);
|
|
|
|
|
expect(exitCode).toBe(0);
|
2025-12-13 04:02:00 +08:00
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("adds context to a path", async () => {
|
2025-12-14 03:19:41 +08:00
|
|
|
// Add context to the collection root using virtual path
|
2025-12-13 04:02:00 +08:00
|
|
|
const { stdout, exitCode } = await runQmd([
|
2025-12-14 03:19:41 +08:00
|
|
|
"context",
|
|
|
|
|
"add",
|
Add path normalization, output format tests, and fix test isolation
- Add support for collection/path.md format in get command (checks if
first component is a known collection before treating as filesystem path)
- Add comprehensive output format tests verifying qmd:// URIs, docid,
and context in JSON, CSV, MD, XML, files, and CLI formats
- Add path normalization tests for various input formats:
qmd://, //, qmd:////, collection/path, and path:line suffix
- Add isolated test environments (createIsolatedTestEnv) to prevent
YAML config conflicts between test suites
- Add test fixture files test1.md and test2.md for path tests
- Update runQmd helper to accept custom configDir parameter
🤖 Generated with [Claude Code](https://claude.com/claude-code)
Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2025-12-21 02:45:18 +08:00
|
|
|
`qmd://${collName}/`,
|
2025-12-13 04:02:00 +08:00
|
|
|
"Personal notes and meeting logs",
|
Add path normalization, output format tests, and fix test isolation
- Add support for collection/path.md format in get command (checks if
first component is a known collection before treating as filesystem path)
- Add comprehensive output format tests verifying qmd:// URIs, docid,
and context in JSON, CSV, MD, XML, files, and CLI formats
- Add path normalization tests for various input formats:
qmd://, //, qmd:////, collection/path, and path:line suffix
- Add isolated test environments (createIsolatedTestEnv) to prevent
YAML config conflicts between test suites
- Add test fixture files test1.md and test2.md for path tests
- Update runQmd helper to accept custom configDir parameter
🤖 Generated with [Claude Code](https://claude.com/claude-code)
Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2025-12-21 02:45:18 +08:00
|
|
|
], { dbPath: localDbPath, configDir: localConfigDir });
|
2025-12-13 04:02:00 +08:00
|
|
|
expect(exitCode).toBe(0);
|
2025-12-14 03:19:41 +08:00
|
|
|
expect(stdout).toContain("✓ Added context");
|
2025-12-13 04:02:00 +08:00
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("requires path and text arguments", async () => {
|
2025-12-22 02:50:17 +08:00
|
|
|
const { stderr, exitCode } = await runQmd(["context", "add"], { dbPath: localDbPath, configDir: localConfigDir });
|
2025-12-13 04:02:00 +08:00
|
|
|
expect(exitCode).toBe(1);
|
|
|
|
|
// Error message goes to stderr
|
|
|
|
|
expect(stderr).toContain("Usage:");
|
|
|
|
|
});
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
describe("CLI Cleanup Command", () => {
|
|
|
|
|
beforeEach(async () => {
|
|
|
|
|
// Ensure we have indexed files
|
2025-12-13 05:07:01 +08:00
|
|
|
await runQmd(["collection", "add", "."]);
|
2025-12-13 04:02:00 +08:00
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("cleans up orphaned entries", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["cleanup"]);
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
});
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
describe("CLI Error Handling", () => {
|
|
|
|
|
test("handles unknown command", async () => {
|
|
|
|
|
const { stderr, exitCode } = await runQmd(["unknowncommand"]);
|
|
|
|
|
expect(exitCode).toBe(1);
|
|
|
|
|
// Should indicate unknown command
|
|
|
|
|
expect(stderr).toContain("Unknown command");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("uses INDEX_PATH environment variable", async () => {
|
|
|
|
|
// Verify the test DB path is being used by creating a separate index
|
|
|
|
|
const customDbPath = join(testDir, "custom.sqlite");
|
2025-12-13 05:07:01 +08:00
|
|
|
const { exitCode } = await runQmd(["collection", "add", "."], {
|
2025-12-13 04:02:00 +08:00
|
|
|
env: { INDEX_PATH: customDbPath },
|
|
|
|
|
});
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
|
|
|
|
|
// The custom database should exist
|
2026-02-16 04:57:13 +08:00
|
|
|
expect(existsSync(customDbPath)).toBe(true);
|
2025-12-13 04:02:00 +08:00
|
|
|
});
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
describe("CLI Output Formats", () => {
|
|
|
|
|
beforeEach(async () => {
|
2025-12-13 05:07:01 +08:00
|
|
|
await runQmd(["collection", "add", "."]);
|
2025-12-13 04:02:00 +08:00
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("search with --json flag outputs JSON", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["search", "--json", "test"]);
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
// Should be valid JSON
|
|
|
|
|
const parsed = JSON.parse(stdout);
|
|
|
|
|
expect(Array.isArray(parsed)).toBe(true);
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("search with --files flag outputs file paths", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["search", "--files", "meeting"]);
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain(".md");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("search output includes snippets by default", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["search", "API"]);
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
// If results found, should have snippet content
|
|
|
|
|
if (!stdout.includes("No results")) {
|
|
|
|
|
expect(stdout.toLowerCase()).toContain("api");
|
|
|
|
|
}
|
|
|
|
|
});
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
describe("CLI Search with Collection Filter", () => {
|
|
|
|
|
let localDbPath: string;
|
|
|
|
|
|
|
|
|
|
beforeEach(async () => {
|
|
|
|
|
// Use a fresh database for this test suite
|
|
|
|
|
localDbPath = getFreshDbPath();
|
2025-12-14 03:19:41 +08:00
|
|
|
// Create multiple collections with explicit names
|
|
|
|
|
await runQmd(["collection", "add", ".", "--name", "notes", "--mask", "notes/*.md"], { dbPath: localDbPath });
|
|
|
|
|
await runQmd(["collection", "add", ".", "--name", "docs", "--mask", "docs/*.md"], { dbPath: localDbPath });
|
2025-12-13 04:02:00 +08:00
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("filters search by collection name", async () => {
|
2025-12-14 03:19:41 +08:00
|
|
|
const { stdout, stderr, exitCode } = await runQmd([
|
2025-12-13 04:02:00 +08:00
|
|
|
"search",
|
|
|
|
|
"-c",
|
|
|
|
|
"notes",
|
|
|
|
|
"meeting",
|
|
|
|
|
], { dbPath: localDbPath });
|
2025-12-14 03:19:41 +08:00
|
|
|
if (exitCode !== 0) {
|
|
|
|
|
console.log("Collection filter search failed:");
|
|
|
|
|
console.log("stdout:", stdout);
|
|
|
|
|
console.log("stderr:", stderr);
|
|
|
|
|
}
|
2025-12-13 04:02:00 +08:00
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
});
|
|
|
|
|
});
|
2025-12-13 04:47:42 +08:00
|
|
|
|
|
|
|
|
describe("CLI Context Management", () => {
|
|
|
|
|
let localDbPath: string;
|
|
|
|
|
|
|
|
|
|
beforeEach(async () => {
|
|
|
|
|
// Use a fresh database for this test suite
|
|
|
|
|
localDbPath = getFreshDbPath();
|
|
|
|
|
// Index some files first
|
2025-12-13 05:07:01 +08:00
|
|
|
await runQmd(["collection", "add", "."], { dbPath: localDbPath });
|
2025-12-13 04:47:42 +08:00
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("add global context with /", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd([
|
|
|
|
|
"context",
|
|
|
|
|
"add",
|
|
|
|
|
"/",
|
|
|
|
|
"Global system context",
|
|
|
|
|
], { dbPath: localDbPath });
|
|
|
|
|
expect(exitCode).toBe(0);
|
2025-12-14 03:16:48 +08:00
|
|
|
expect(stdout).toContain("✓ Set global context");
|
2025-12-13 04:47:42 +08:00
|
|
|
expect(stdout).toContain("Global system context");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("list contexts", async () => {
|
|
|
|
|
// Add a global context first
|
|
|
|
|
await runQmd([
|
|
|
|
|
"context",
|
|
|
|
|
"add",
|
|
|
|
|
"/",
|
|
|
|
|
"Test context",
|
|
|
|
|
], { dbPath: localDbPath });
|
|
|
|
|
|
|
|
|
|
const { stdout, exitCode } = await runQmd([
|
|
|
|
|
"context",
|
|
|
|
|
"list",
|
|
|
|
|
], { dbPath: localDbPath });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("Configured Contexts");
|
|
|
|
|
expect(stdout).toContain("Test context");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("add context to virtual path", async () => {
|
|
|
|
|
// Collection name should be "fixtures" (basename of the fixtures directory)
|
|
|
|
|
const { stdout, exitCode } = await runQmd([
|
|
|
|
|
"context",
|
|
|
|
|
"add",
|
|
|
|
|
"qmd://fixtures/notes",
|
|
|
|
|
"Context for notes subdirectory",
|
|
|
|
|
], { dbPath: localDbPath });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("✓ Added context for: qmd://fixtures/notes");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("remove global context", async () => {
|
|
|
|
|
// Add a global context first
|
|
|
|
|
await runQmd([
|
|
|
|
|
"context",
|
|
|
|
|
"add",
|
|
|
|
|
"/",
|
|
|
|
|
"Global context to remove",
|
|
|
|
|
], { dbPath: localDbPath });
|
|
|
|
|
|
|
|
|
|
const { stdout, exitCode } = await runQmd([
|
|
|
|
|
"context",
|
|
|
|
|
"rm",
|
|
|
|
|
"/",
|
|
|
|
|
], { dbPath: localDbPath });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("✓ Removed");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("remove virtual path context", async () => {
|
|
|
|
|
// Add a context first
|
|
|
|
|
await runQmd([
|
|
|
|
|
"context",
|
|
|
|
|
"add",
|
|
|
|
|
"qmd://fixtures/notes",
|
|
|
|
|
"Context to remove",
|
|
|
|
|
], { dbPath: localDbPath });
|
|
|
|
|
|
|
|
|
|
const { stdout, exitCode } = await runQmd([
|
|
|
|
|
"context",
|
|
|
|
|
"rm",
|
|
|
|
|
"qmd://fixtures/notes",
|
|
|
|
|
], { dbPath: localDbPath });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("✓ Removed context for: qmd://fixtures/notes");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("fails to remove non-existent context", async () => {
|
|
|
|
|
const { stdout, stderr, exitCode } = await runQmd([
|
|
|
|
|
"context",
|
|
|
|
|
"rm",
|
|
|
|
|
"qmd://nonexistent/path",
|
|
|
|
|
], { dbPath: localDbPath });
|
|
|
|
|
expect(exitCode).toBe(1);
|
|
|
|
|
expect(stderr || stdout).toContain("not found");
|
|
|
|
|
});
|
|
|
|
|
});
|
2025-12-13 04:55:21 +08:00
|
|
|
|
|
|
|
|
describe("CLI ls Command", () => {
|
|
|
|
|
let localDbPath: string;
|
|
|
|
|
|
|
|
|
|
beforeEach(async () => {
|
|
|
|
|
// Use a fresh database for this test suite
|
|
|
|
|
localDbPath = getFreshDbPath();
|
|
|
|
|
// Index some files first
|
2025-12-13 05:07:01 +08:00
|
|
|
await runQmd(["collection", "add", "."], { dbPath: localDbPath });
|
2025-12-13 04:55:21 +08:00
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("lists all collections", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["ls"], { dbPath: localDbPath });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("Collections:");
|
|
|
|
|
expect(stdout).toContain("qmd://fixtures/");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("lists files in a collection", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["ls", "fixtures"], { dbPath: localDbPath });
|
|
|
|
|
expect(exitCode).toBe(0);
|
Add docid, line-numbers, handelize, and fix displayPath format
Features:
- Add short document IDs (docid) - first 6 chars of hash - to all search outputs
- Add --line-numbers CLI option and lineNumbers param for MCP tools
- Add handelize() function for token-friendly filenames (lowercase, special chars to dash, preserves extension)
- Convert triple underscore `___` to folder separator in filenames
- Change displayPath format to include collection name (collection/path)
- Make line-numbers default for MCP search snippets
Changes:
- store.ts: Add getDocid(), findDocumentByDocid(), handelize() functions
- formatter.ts: Add docid to all formatters, addLineNumbers() helper
- qmd.ts: Add --line-numbers option, use handelize during indexing
- mcp.ts: Remove resource listing, lineNumbers default for snippets
- Update all tests to expect new displayPath format and handelize behavior
- Update CLAUDE.md with docid documentation
All 274 tests pass.
🤖 Generated with [Claude Code](https://claude.com/claude-code)
Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2025-12-17 01:26:39 +08:00
|
|
|
// handelize converts to lowercase
|
|
|
|
|
expect(stdout).toContain("qmd://fixtures/readme.md");
|
2025-12-13 04:55:21 +08:00
|
|
|
expect(stdout).toContain("qmd://fixtures/notes/meeting.md");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("lists files with path prefix", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["ls", "fixtures/notes"], { dbPath: localDbPath });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("qmd://fixtures/notes/meeting.md");
|
|
|
|
|
expect(stdout).toContain("qmd://fixtures/notes/ideas.md");
|
Add docid, line-numbers, handelize, and fix displayPath format
Features:
- Add short document IDs (docid) - first 6 chars of hash - to all search outputs
- Add --line-numbers CLI option and lineNumbers param for MCP tools
- Add handelize() function for token-friendly filenames (lowercase, special chars to dash, preserves extension)
- Convert triple underscore `___` to folder separator in filenames
- Change displayPath format to include collection name (collection/path)
- Make line-numbers default for MCP search snippets
Changes:
- store.ts: Add getDocid(), findDocumentByDocid(), handelize() functions
- formatter.ts: Add docid to all formatters, addLineNumbers() helper
- qmd.ts: Add --line-numbers option, use handelize during indexing
- mcp.ts: Remove resource listing, lineNumbers default for snippets
- Update all tests to expect new displayPath format and handelize behavior
- Update CLAUDE.md with docid documentation
All 274 tests pass.
🤖 Generated with [Claude Code](https://claude.com/claude-code)
Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2025-12-17 01:26:39 +08:00
|
|
|
// Should not include files outside the prefix (handelize converts to lowercase)
|
|
|
|
|
expect(stdout).not.toContain("qmd://fixtures/readme.md");
|
2025-12-13 04:55:21 +08:00
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("lists files with virtual path", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["ls", "qmd://fixtures/docs"], { dbPath: localDbPath });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("qmd://fixtures/docs/api.md");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("handles non-existent collection", async () => {
|
|
|
|
|
const { stderr, exitCode } = await runQmd(["ls", "nonexistent"], { dbPath: localDbPath });
|
|
|
|
|
expect(exitCode).toBe(1);
|
|
|
|
|
expect(stderr).toContain("Collection not found");
|
|
|
|
|
});
|
|
|
|
|
});
|
2025-12-13 05:02:16 +08:00
|
|
|
|
|
|
|
|
describe("CLI Collection Commands", () => {
|
|
|
|
|
let localDbPath: string;
|
|
|
|
|
|
|
|
|
|
beforeEach(async () => {
|
|
|
|
|
// Use a fresh database for this test suite
|
|
|
|
|
localDbPath = getFreshDbPath();
|
|
|
|
|
// Index some files first to create a collection
|
2025-12-13 05:07:01 +08:00
|
|
|
await runQmd(["collection", "add", "."], { dbPath: localDbPath });
|
2025-12-13 05:02:16 +08:00
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("lists collections", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["collection", "list"], { dbPath: localDbPath });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("Collections");
|
|
|
|
|
expect(stdout).toContain("fixtures");
|
Add path normalization, output format tests, and fix test isolation
- Add support for collection/path.md format in get command (checks if
first component is a known collection before treating as filesystem path)
- Add comprehensive output format tests verifying qmd:// URIs, docid,
and context in JSON, CSV, MD, XML, files, and CLI formats
- Add path normalization tests for various input formats:
qmd://, //, qmd:////, collection/path, and path:line suffix
- Add isolated test environments (createIsolatedTestEnv) to prevent
YAML config conflicts between test suites
- Add test fixture files test1.md and test2.md for path tests
- Update runQmd helper to accept custom configDir parameter
🤖 Generated with [Claude Code](https://claude.com/claude-code)
Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2025-12-21 02:45:18 +08:00
|
|
|
expect(stdout).toContain("qmd://fixtures/");
|
2025-12-13 05:02:16 +08:00
|
|
|
expect(stdout).toContain("Pattern:");
|
|
|
|
|
expect(stdout).toContain("Files:");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("removes a collection", async () => {
|
|
|
|
|
// First verify the collection exists
|
|
|
|
|
const { stdout: listBefore } = await runQmd(["collection", "list"], { dbPath: localDbPath });
|
|
|
|
|
expect(listBefore).toContain("fixtures");
|
|
|
|
|
|
|
|
|
|
// Remove it
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["collection", "remove", "fixtures"], { dbPath: localDbPath });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("✓ Removed collection 'fixtures'");
|
|
|
|
|
expect(stdout).toContain("Deleted");
|
|
|
|
|
|
|
|
|
|
// Verify it's gone
|
|
|
|
|
const { stdout: listAfter } = await runQmd(["collection", "list"], { dbPath: localDbPath });
|
|
|
|
|
expect(listAfter).not.toContain("fixtures");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("handles removing non-existent collection", async () => {
|
|
|
|
|
const { stderr, exitCode } = await runQmd(["collection", "remove", "nonexistent"], { dbPath: localDbPath });
|
|
|
|
|
expect(exitCode).toBe(1);
|
|
|
|
|
expect(stderr).toContain("Collection not found");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("handles missing remove argument", async () => {
|
|
|
|
|
const { stderr, exitCode } = await runQmd(["collection", "remove"], { dbPath: localDbPath });
|
|
|
|
|
expect(exitCode).toBe(1);
|
|
|
|
|
expect(stderr).toContain("Usage:");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("handles unknown subcommand", async () => {
|
|
|
|
|
const { stderr, exitCode } = await runQmd(["collection", "invalid"], { dbPath: localDbPath });
|
|
|
|
|
expect(exitCode).toBe(1);
|
|
|
|
|
expect(stderr).toContain("Unknown subcommand");
|
|
|
|
|
});
|
2025-12-13 05:29:49 +08:00
|
|
|
|
|
|
|
|
test("renames a collection", async () => {
|
|
|
|
|
// First verify the collection exists
|
|
|
|
|
const { stdout: listBefore } = await runQmd(["collection", "list"], { dbPath: localDbPath });
|
Add path normalization, output format tests, and fix test isolation
- Add support for collection/path.md format in get command (checks if
first component is a known collection before treating as filesystem path)
- Add comprehensive output format tests verifying qmd:// URIs, docid,
and context in JSON, CSV, MD, XML, files, and CLI formats
- Add path normalization tests for various input formats:
qmd://, //, qmd:////, collection/path, and path:line suffix
- Add isolated test environments (createIsolatedTestEnv) to prevent
YAML config conflicts between test suites
- Add test fixture files test1.md and test2.md for path tests
- Update runQmd helper to accept custom configDir parameter
🤖 Generated with [Claude Code](https://claude.com/claude-code)
Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2025-12-21 02:45:18 +08:00
|
|
|
expect(listBefore).toContain("qmd://fixtures/");
|
2025-12-13 05:29:49 +08:00
|
|
|
|
|
|
|
|
// Rename it
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["collection", "rename", "fixtures", "my-fixtures"], { dbPath: localDbPath });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("✓ Renamed collection 'fixtures' to 'my-fixtures'");
|
|
|
|
|
expect(stdout).toContain("qmd://fixtures/");
|
|
|
|
|
expect(stdout).toContain("qmd://my-fixtures/");
|
|
|
|
|
|
|
|
|
|
// Verify the new name exists and old name is gone
|
|
|
|
|
const { stdout: listAfter } = await runQmd(["collection", "list"], { dbPath: localDbPath });
|
Add path normalization, output format tests, and fix test isolation
- Add support for collection/path.md format in get command (checks if
first component is a known collection before treating as filesystem path)
- Add comprehensive output format tests verifying qmd:// URIs, docid,
and context in JSON, CSV, MD, XML, files, and CLI formats
- Add path normalization tests for various input formats:
qmd://, //, qmd:////, collection/path, and path:line suffix
- Add isolated test environments (createIsolatedTestEnv) to prevent
YAML config conflicts between test suites
- Add test fixture files test1.md and test2.md for path tests
- Update runQmd helper to accept custom configDir parameter
🤖 Generated with [Claude Code](https://claude.com/claude-code)
Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2025-12-21 02:45:18 +08:00
|
|
|
expect(listAfter).toContain("qmd://my-fixtures/");
|
|
|
|
|
expect(listAfter).not.toContain("qmd://fixtures/"); // Old collection should not appear
|
2025-12-13 05:29:49 +08:00
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("handles renaming non-existent collection", async () => {
|
|
|
|
|
const { stderr, exitCode } = await runQmd(["collection", "rename", "nonexistent", "newname"], { dbPath: localDbPath });
|
|
|
|
|
expect(exitCode).toBe(1);
|
|
|
|
|
expect(stderr).toContain("Collection not found");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("handles renaming to existing collection name", async () => {
|
|
|
|
|
// Create a second collection in a temp directory
|
|
|
|
|
const tempDir = await mkdtemp(join(tmpdir(), "qmd-second-"));
|
|
|
|
|
await writeFile(join(tempDir, "test.md"), "# Test");
|
|
|
|
|
const addResult = await runQmd(["collection", "add", tempDir, "--name", "second"], { dbPath: localDbPath });
|
|
|
|
|
|
|
|
|
|
if (addResult.exitCode !== 0) {
|
|
|
|
|
console.error("Failed to add second collection:", addResult.stderr);
|
|
|
|
|
}
|
|
|
|
|
expect(addResult.exitCode).toBe(0);
|
|
|
|
|
|
|
|
|
|
// Verify both collections exist
|
|
|
|
|
const { stdout: listBoth } = await runQmd(["collection", "list"], { dbPath: localDbPath });
|
Add path normalization, output format tests, and fix test isolation
- Add support for collection/path.md format in get command (checks if
first component is a known collection before treating as filesystem path)
- Add comprehensive output format tests verifying qmd:// URIs, docid,
and context in JSON, CSV, MD, XML, files, and CLI formats
- Add path normalization tests for various input formats:
qmd://, //, qmd:////, collection/path, and path:line suffix
- Add isolated test environments (createIsolatedTestEnv) to prevent
YAML config conflicts between test suites
- Add test fixture files test1.md and test2.md for path tests
- Update runQmd helper to accept custom configDir parameter
🤖 Generated with [Claude Code](https://claude.com/claude-code)
Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2025-12-21 02:45:18 +08:00
|
|
|
expect(listBoth).toContain("qmd://fixtures/");
|
|
|
|
|
expect(listBoth).toContain("qmd://second/");
|
2025-12-13 05:29:49 +08:00
|
|
|
|
|
|
|
|
// Try to rename fixtures to second (which already exists)
|
|
|
|
|
const { stderr, exitCode } = await runQmd(["collection", "rename", "fixtures", "second"], { dbPath: localDbPath });
|
|
|
|
|
expect(exitCode).toBe(1);
|
|
|
|
|
expect(stderr).toContain("Collection name already exists");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("handles missing rename arguments", async () => {
|
|
|
|
|
const { stderr: stderr1, exitCode: exitCode1 } = await runQmd(["collection", "rename"], { dbPath: localDbPath });
|
|
|
|
|
expect(exitCode1).toBe(1);
|
|
|
|
|
expect(stderr1).toContain("Usage:");
|
|
|
|
|
|
|
|
|
|
const { stderr: stderr2, exitCode: exitCode2 } = await runQmd(["collection", "rename", "fixtures"], { dbPath: localDbPath });
|
|
|
|
|
expect(exitCode2).toBe(1);
|
|
|
|
|
expect(stderr2).toContain("Usage:");
|
|
|
|
|
});
|
2025-12-13 05:02:16 +08:00
|
|
|
});
|
Add path normalization, output format tests, and fix test isolation
- Add support for collection/path.md format in get command (checks if
first component is a known collection before treating as filesystem path)
- Add comprehensive output format tests verifying qmd:// URIs, docid,
and context in JSON, CSV, MD, XML, files, and CLI formats
- Add path normalization tests for various input formats:
qmd://, //, qmd:////, collection/path, and path:line suffix
- Add isolated test environments (createIsolatedTestEnv) to prevent
YAML config conflicts between test suites
- Add test fixture files test1.md and test2.md for path tests
- Update runQmd helper to accept custom configDir parameter
🤖 Generated with [Claude Code](https://claude.com/claude-code)
Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2025-12-21 02:45:18 +08:00
|
|
|
|
|
|
|
|
// =============================================================================
|
|
|
|
|
// Output Format Tests - qmd:// URIs, context, and docid
|
|
|
|
|
// =============================================================================
|
|
|
|
|
|
|
|
|
|
describe("search output formats", () => {
|
|
|
|
|
let localDbPath: string;
|
|
|
|
|
let localConfigDir: string;
|
|
|
|
|
const collName = "fixtures";
|
|
|
|
|
|
|
|
|
|
beforeAll(async () => {
|
|
|
|
|
const env = await createIsolatedTestEnv("output-format");
|
|
|
|
|
localDbPath = env.dbPath;
|
|
|
|
|
localConfigDir = env.configDir;
|
|
|
|
|
|
|
|
|
|
// Add collection
|
|
|
|
|
const { exitCode, stderr } = await runQmd(
|
|
|
|
|
["collection", "add", fixturesDir, "--name", collName],
|
|
|
|
|
{ dbPath: localDbPath, configDir: localConfigDir }
|
|
|
|
|
);
|
|
|
|
|
if (exitCode !== 0) console.error("collection add failed:", stderr);
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
|
|
|
|
|
// Add context
|
|
|
|
|
await runQmd(["context", "add", `qmd://${collName}/`, "Test fixtures for QMD"], { dbPath: localDbPath, configDir: localConfigDir });
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("search --json includes qmd:// path, docid, and context", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["search", "test", "--json", "-n", "1"], { dbPath: localDbPath, configDir: localConfigDir });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
|
|
|
|
|
const results = JSON.parse(stdout);
|
|
|
|
|
expect(results.length).toBeGreaterThan(0);
|
|
|
|
|
|
|
|
|
|
const result = results[0];
|
|
|
|
|
expect(result.file).toMatch(new RegExp(`^qmd://${collName}/`));
|
|
|
|
|
expect(result.docid).toMatch(/^#[a-f0-9]{6}$/);
|
|
|
|
|
expect(result.context).toBe("Test fixtures for QMD");
|
|
|
|
|
// Ensure no full filesystem paths
|
|
|
|
|
expect(result.file).not.toMatch(/^\/Users\//);
|
|
|
|
|
expect(result.file).not.toMatch(/^\/home\//);
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("search --files includes qmd:// path, docid, and context", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["search", "test", "--files", "-n", "1"], { dbPath: localDbPath, configDir: localConfigDir });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
|
|
|
|
|
// Format: #docid,score,qmd://collection/path,"context"
|
|
|
|
|
expect(stdout).toMatch(new RegExp(`^#[a-f0-9]{6},[\\d.]+,qmd://${collName}/`, "m"));
|
|
|
|
|
expect(stdout).toContain("Test fixtures for QMD");
|
|
|
|
|
// Ensure no full filesystem paths
|
|
|
|
|
expect(stdout).not.toMatch(/\/Users\//);
|
|
|
|
|
expect(stdout).not.toMatch(/\/home\//);
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("search --csv includes qmd:// path, docid, and context", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["search", "test", "--csv", "-n", "1"], { dbPath: localDbPath, configDir: localConfigDir });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
|
|
|
|
|
// Header should include context
|
|
|
|
|
expect(stdout).toMatch(/^docid,score,file,title,context,line,snippet$/m);
|
|
|
|
|
// Data rows should have qmd:// paths and context
|
|
|
|
|
expect(stdout).toMatch(new RegExp(`#[a-f0-9]{6},[\\d.]+,qmd://${collName}/`));
|
|
|
|
|
expect(stdout).toContain("Test fixtures for QMD");
|
|
|
|
|
// Ensure no full filesystem paths
|
|
|
|
|
expect(stdout).not.toMatch(/\/Users\//);
|
|
|
|
|
expect(stdout).not.toMatch(/\/home\//);
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("search --md includes docid and context", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["search", "test", "--md", "-n", "1"], { dbPath: localDbPath, configDir: localConfigDir });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
|
|
|
|
|
expect(stdout).toMatch(/\*\*docid:\*\* `#[a-f0-9]{6}`/);
|
|
|
|
|
expect(stdout).toContain("**context:** Test fixtures for QMD");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("search --xml includes qmd:// path, docid, and context", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["search", "test", "--xml", "-n", "1"], { dbPath: localDbPath, configDir: localConfigDir });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
|
|
|
|
|
expect(stdout).toMatch(new RegExp(`<file docid="#[a-f0-9]{6}" name="qmd://${collName}/`));
|
|
|
|
|
expect(stdout).toContain('context="Test fixtures for QMD"');
|
|
|
|
|
// Ensure no full filesystem paths
|
|
|
|
|
expect(stdout).not.toMatch(/\/Users\//);
|
|
|
|
|
expect(stdout).not.toMatch(/\/home\//);
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("search default CLI format includes qmd:// path, docid, and context", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["search", "test", "-n", "1"], { dbPath: localDbPath, configDir: localConfigDir });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
|
|
|
|
|
// First line should have qmd:// path and docid
|
|
|
|
|
expect(stdout).toMatch(new RegExp(`^qmd://${collName}/.*#[a-f0-9]{6}`, "m"));
|
|
|
|
|
expect(stdout).toContain("Context: Test fixtures for QMD");
|
|
|
|
|
// Ensure no full filesystem paths
|
|
|
|
|
expect(stdout).not.toMatch(/\/Users\//);
|
|
|
|
|
expect(stdout).not.toMatch(/\/home\//);
|
|
|
|
|
});
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
// =============================================================================
|
|
|
|
|
// Get Command Path Normalization Tests
|
|
|
|
|
// =============================================================================
|
|
|
|
|
|
|
|
|
|
describe("get command path normalization", () => {
|
|
|
|
|
let localDbPath: string;
|
|
|
|
|
let localConfigDir: string;
|
|
|
|
|
const collName = "fixtures";
|
|
|
|
|
|
|
|
|
|
beforeAll(async () => {
|
|
|
|
|
const env = await createIsolatedTestEnv("get-paths");
|
|
|
|
|
localDbPath = env.dbPath;
|
|
|
|
|
localConfigDir = env.configDir;
|
|
|
|
|
|
|
|
|
|
const { exitCode, stderr } = await runQmd(
|
|
|
|
|
["collection", "add", fixturesDir, "--name", collName],
|
|
|
|
|
{ dbPath: localDbPath, configDir: localConfigDir }
|
|
|
|
|
);
|
|
|
|
|
if (exitCode !== 0) console.error("collection add failed:", stderr);
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("get with qmd://collection/path format", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["get", `qmd://${collName}/test1.md`, "-l", "3"], { dbPath: localDbPath, configDir: localConfigDir });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("Test Document 1");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("get with collection/path format (no scheme)", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["get", `${collName}/test1.md`, "-l", "3"], { dbPath: localDbPath, configDir: localConfigDir });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("Test Document 1");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("get with //collection/path format", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["get", `//${collName}/test1.md`, "-l", "3"], { dbPath: localDbPath, configDir: localConfigDir });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("Test Document 1");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("get with qmd:////collection/path format (extra slashes)", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["get", `qmd:////${collName}/test1.md`, "-l", "3"], { dbPath: localDbPath, configDir: localConfigDir });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("Test Document 1");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("get with path:line format", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["get", `${collName}/test1.md:3`, "-l", "2"], { dbPath: localDbPath, configDir: localConfigDir });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
// Should start from line 3, not line 1
|
|
|
|
|
expect(stdout).not.toMatch(/^# Test Document 1$/m);
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("get with qmd://path:line format", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["get", `qmd://${collName}/test1.md:3`, "-l", "2"], { dbPath: localDbPath, configDir: localConfigDir });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
// Should start from line 3, not line 1
|
|
|
|
|
expect(stdout).not.toMatch(/^# Test Document 1$/m);
|
|
|
|
|
});
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
// =============================================================================
|
|
|
|
|
// Status and Collection List - No Full Paths
|
|
|
|
|
// =============================================================================
|
|
|
|
|
|
|
|
|
|
describe("status and collection list hide filesystem paths", () => {
|
|
|
|
|
let localDbPath: string;
|
|
|
|
|
let localConfigDir: string;
|
|
|
|
|
const collName = "fixtures";
|
|
|
|
|
|
|
|
|
|
beforeAll(async () => {
|
|
|
|
|
const env = await createIsolatedTestEnv("status-paths");
|
|
|
|
|
localDbPath = env.dbPath;
|
|
|
|
|
localConfigDir = env.configDir;
|
|
|
|
|
|
|
|
|
|
const { exitCode, stderr } = await runQmd(
|
|
|
|
|
["collection", "add", fixturesDir, "--name", collName],
|
|
|
|
|
{ dbPath: localDbPath, configDir: localConfigDir }
|
|
|
|
|
);
|
|
|
|
|
if (exitCode !== 0) console.error("collection add failed:", stderr);
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("status does not show full filesystem paths", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["status"], { dbPath: localDbPath, configDir: localConfigDir });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
|
|
|
|
|
// Should show qmd:// URIs
|
|
|
|
|
expect(stdout).toContain(`qmd://${collName}/`);
|
|
|
|
|
// Should NOT show full filesystem paths (except for the index location which is ok)
|
|
|
|
|
const lines = stdout.split('\n').filter(l => !l.includes('Index:'));
|
|
|
|
|
const pathLines = lines.filter(l => l.includes('/Users/') || l.includes('/home/') || l.includes('/tmp/'));
|
|
|
|
|
expect(pathLines.length).toBe(0);
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("collection list does not show full filesystem paths", async () => {
|
|
|
|
|
const { stdout, exitCode } = await runQmd(["collection", "list"], { dbPath: localDbPath, configDir: localConfigDir });
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
|
|
|
|
|
// Should show qmd:// URIs
|
|
|
|
|
expect(stdout).toContain(`qmd://${collName}/`);
|
|
|
|
|
// Should NOT show Path: lines with filesystem paths
|
|
|
|
|
expect(stdout).not.toMatch(/Path:\s+\//);
|
|
|
|
|
});
|
|
|
|
|
});
|
MCP: Streamable HTTP, scoring fixes, tool improvements (#149)
* feat: MCP HTTP transport with daemon lifecycle
Add streaming HTTP transport as an alternative to stdio for the MCP
server. A long-lived HTTP server avoids reloading 3 GGUF models (~2GB)
on every client connection, reducing warm query latency from ~16s (CLI)
to ~10s.
New CLI surface:
qmd mcp --http [--port N] # foreground, default port 3000
qmd mcp --http --daemon # background, PID in ~/.cache/qmd/mcp.pid
qmd mcp stop # stop daemon via PID file
qmd status # now shows MCP daemon liveness
Server implementation (mcp.ts):
- Extract createMcpServer(store) shared by stdio and HTTP transports
- HTTP transport uses WebStandardStreamableHTTPServerTransport with
JSON responses (stateless, no SSE)
- /health endpoint with uptime, /mcp for MCP protocol, 404 otherwise
- Request logging to stderr with timestamps, tool names, query args
Daemon lifecycle (qmd.ts):
- PID file + log file management with stale PID detection
- Absolute paths in Bun.spawn (process.execPath + import.meta.path)
so daemon works regardless of cwd
- mkdirSync for cache dir on fresh installs
- Removes top-level SIGTERM/SIGINT handlers before starting HTTP
server so async cleanup in mcp.ts actually runs
Move hybridQuery() and vectorSearchQuery() into store.ts as standalone
functions that take a Store as first argument. Both CLI and MCP now
call the identical pipeline, eliminating the class of bugs where one
copy drifts from the other.
Shared pipeline (store.ts):
- hybridQuery(): BM25 probe → expand → FTS+vec search → RRF →
chunk → rerank (chunks only) → position-aware blending → dedup
- vectorSearchQuery(): expand → vec search → dedup → sort
- SearchHooks interface for optional progress callbacks
- Constants: STRONG_SIGNAL_MIN_SCORE, STRONG_SIGNAL_MIN_GAP,
RERANK_CANDIDATE_LIMIT (40), addLineNumbers()
Bugs fixed by unification:
- MCP now gets strong-signal short-circuit (was CLI-only)
- Reranker candidate limit unified at 40 (MCP had 30)
- File dedup added to hybrid query (MCP was missing it)
- Collection filter pushed into searchVec DB query
- Filter-then-slice ordering fixed (MCP was slice-then-filter)
* feat: type-routed query expansion — lex→FTS, vec/hyde→vector
expandQuery() now returns typed ExpandedQuery[] instead of string[],
preserving the lex/vec/hyde type info from the LLM's GBNF-structured
output. hybridQuery() and vectorSearchQuery() route searches by type:
lex queries go to FTS only, vec/hyde go to vector only.
Previously, every expanded query ran through BOTH backends — keyword
variants wasted embedding forward passes, semantic paraphrases wasted
BM25 lookups. Type routing eliminates ~4 calls/query with zero quality
loss (cross-backend noise actually hurt RRF fusion).
Cache format changed from newline-separated text to JSON (preserves
types). Old cache entries gracefully re-expand on first access.
CLI expansion tree now shows query types:
├─ original query
├─ lex: keyword variant
├─ vec: semantic meaning
└─ hyde: hypothetical document...
Benchmark (5 queries, 1756-doc index, warm LLM, Apple Silicon):
Metric Old (untyped) New (typed) Delta
Avg backend calls 10.0 6.0 -40%
Total wall time 1278ms 549ms -57%
Avg saved/query — — 146ms
"authentication setup" 12 → 7 calls 511 → 112ms
"database migration strategy" 10 → 6 calls 182 → 106ms
"how to handle errors in API" 10 → 6 calls 216 → 121ms
"meeting notes from last week" 10 → 6 calls 228 → 110ms
"performance optimization" 8 → 5 calls 141 → 100ms
Savings come from skipped embed() calls (~30-80ms each). FTS is
synchronous SQLite (~0ms), so lex→FTS routing is free while
vec/hyde→vector-only avoids wasted embedding passes.
* fix: MCP query snippets now use reranker's best chunk, not full body
extractSnippet() was scanning the entire document body for keyword
matches to build the snippet. But hybridQuery() already identified
the most relevant chunk via cross-attention reranking — rescanning
the full body is redundant and can land on a less relevant section
if the query terms appear elsewhere in the document.
CLI was already using bestChunk (set during the refactor). MCP was
still using body — a pre-existing inconsistency, not a regression.
* feat: dynamic MCP instructions + tool annotations
The MCP server now generates instructions at startup from actual index
state and injects them into the initialize response. LLMs see collection
names, document counts, content descriptions, and search strategy
guidance in their system prompt — zero tool calls needed for orientation.
Previously, the only guidance was generic static tool descriptions and
a user-invocable "query" prompt that no LLM would discover on its own.
An LLM connecting to QMD had no idea what collections existed, what they
contained, or how to scope searches effectively.
* change default port to 8181
* fix: BM25 score normalization was inverted
The normalization formula `1 / (1 + |bm25|)` is a decreasing function of
match strength. FTS5 BM25 scores are negative where more negative = better
match (e.g., -10 is strong, -0.5 is weak). The formula mapped:
strong match (raw -10) → 1/(1+10) = 9% ← should be highest
weak match (raw -0.5) → 1/(1+0.5) = 67% ← should be lowest
Three downstream effects:
1. `--min-score 0.5` (or MCP minScore: 0.5) filtered OUT strong matches
and kept only weak ones. The MCP instructions recommend this threshold.
2. CLI `formatScore()` color bands never showed green for BM25 results
(best matches scored ~9%, green threshold is 70%).
3. The strong signal optimization in hybridQuery (skip ~2s LLM expansion
when BM25 already has a clear winner) was dead code — strong matches
scored ~0.09, never reaching the 0.85 threshold.
Fix: `|x| / (1 + |x|)` — same (0,1) range, monotonic, no per-query
normalization needed, but now correctly maps strong → high, weak → low.
The normalization was born broken (Math.max(0, x) clamped all
negative BM25 to 0 → every score = 1.0), then PR #76 changed to
Math.abs which made scores vary but inverted the direction. Neither
state was ever correct.
* fix: rerank cache key ignores chunk content
The rerank cache key was (query, file, model) but the actual text sent
to the reranker is a keyword-selected chunk that varies by query terms.
Two different queries hitting the same file can select different chunks,
but the second query gets a stale cached score from the first chunk.
Example:
Query "auth flow" → selects chunk about authentication → score 0.92
Query "auth tokens" → same file, selects chunk about tokens
→ cache HIT on (query, file, model) → returns 0.92 from wrong chunk
Fix: include full chunk text in cache key. getCacheKey() already
SHA-256 hashes its inputs, so this adds no key bloat — just
disambiguation. Old cache entries become natural misses (different key
shape) and re-warm on next query.
* rename MCP tools for clarity, rewrite descriptions for LLM tool selection
Rename MCP tools: vsearch → vector_search, query → deep_search.
LLMs see these names — self-documenting names reduce reliance on
descriptions for tool selection. CLI commands stay unchanged
(qmd vsearch, qmd query) — different namespace, users type those.
Rewrite all search tool descriptions to be action-oriented:
- search: "Search by keyword. Finds documents containing exact
words and phrases in the query."
- vector_search: "Search by meaning. Finds relevant documents even
when they use different words than the query — handles synonyms,
paraphrases, and related concepts."
- deep_search: "Deep search. Auto-expands the query into variations,
searches each by keyword and meaning, and reranks for top hits
across all results."
Rewrite instructions ladder — each tool says what it does, no
"start here" / "escalate as needed" strategy language.
Delete the "query" prompt (registerPrompt) — it restated what
descriptions + instructions already cover. No LLM proactively
calls prompts/get to learn how to use tools.
* supress HTTP server logs during tests
2026-02-11 05:37:33 +08:00
|
|
|
|
|
|
|
|
// =============================================================================
|
|
|
|
|
// MCP HTTP Daemon Lifecycle
|
|
|
|
|
// =============================================================================
|
|
|
|
|
|
|
|
|
|
describe("mcp http daemon", () => {
|
|
|
|
|
let daemonTestDir: string;
|
|
|
|
|
let daemonCacheDir: string; // XDG_CACHE_HOME value (the qmd/ subdir is created automatically)
|
|
|
|
|
let daemonDbPath: string;
|
|
|
|
|
let daemonConfigDir: string;
|
|
|
|
|
|
|
|
|
|
// Track spawned PIDs for cleanup
|
|
|
|
|
const spawnedPids: number[] = [];
|
|
|
|
|
|
|
|
|
|
/** Get path to PID file inside the test cache dir */
|
|
|
|
|
function pidPath(): string {
|
|
|
|
|
return join(daemonCacheDir, "qmd", "mcp.pid");
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/** Run qmd with test-isolated env (cache, db, config) */
|
|
|
|
|
async function runDaemonQmd(
|
|
|
|
|
args: string[],
|
|
|
|
|
): Promise<{ stdout: string; stderr: string; exitCode: number }> {
|
|
|
|
|
return runQmd(args, {
|
|
|
|
|
dbPath: daemonDbPath,
|
|
|
|
|
configDir: daemonConfigDir,
|
|
|
|
|
env: { XDG_CACHE_HOME: daemonCacheDir },
|
|
|
|
|
});
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/** Spawn a foreground HTTP server (non-blocking) and return the process */
|
2026-02-16 04:57:13 +08:00
|
|
|
function spawnHttpServer(port: number): import("child_process").ChildProcess {
|
|
|
|
|
const proc = spawn(tsxBin, [qmdScript, "mcp", "--http", "--port", String(port)], {
|
MCP: Streamable HTTP, scoring fixes, tool improvements (#149)
* feat: MCP HTTP transport with daemon lifecycle
Add streaming HTTP transport as an alternative to stdio for the MCP
server. A long-lived HTTP server avoids reloading 3 GGUF models (~2GB)
on every client connection, reducing warm query latency from ~16s (CLI)
to ~10s.
New CLI surface:
qmd mcp --http [--port N] # foreground, default port 3000
qmd mcp --http --daemon # background, PID in ~/.cache/qmd/mcp.pid
qmd mcp stop # stop daemon via PID file
qmd status # now shows MCP daemon liveness
Server implementation (mcp.ts):
- Extract createMcpServer(store) shared by stdio and HTTP transports
- HTTP transport uses WebStandardStreamableHTTPServerTransport with
JSON responses (stateless, no SSE)
- /health endpoint with uptime, /mcp for MCP protocol, 404 otherwise
- Request logging to stderr with timestamps, tool names, query args
Daemon lifecycle (qmd.ts):
- PID file + log file management with stale PID detection
- Absolute paths in Bun.spawn (process.execPath + import.meta.path)
so daemon works regardless of cwd
- mkdirSync for cache dir on fresh installs
- Removes top-level SIGTERM/SIGINT handlers before starting HTTP
server so async cleanup in mcp.ts actually runs
Move hybridQuery() and vectorSearchQuery() into store.ts as standalone
functions that take a Store as first argument. Both CLI and MCP now
call the identical pipeline, eliminating the class of bugs where one
copy drifts from the other.
Shared pipeline (store.ts):
- hybridQuery(): BM25 probe → expand → FTS+vec search → RRF →
chunk → rerank (chunks only) → position-aware blending → dedup
- vectorSearchQuery(): expand → vec search → dedup → sort
- SearchHooks interface for optional progress callbacks
- Constants: STRONG_SIGNAL_MIN_SCORE, STRONG_SIGNAL_MIN_GAP,
RERANK_CANDIDATE_LIMIT (40), addLineNumbers()
Bugs fixed by unification:
- MCP now gets strong-signal short-circuit (was CLI-only)
- Reranker candidate limit unified at 40 (MCP had 30)
- File dedup added to hybrid query (MCP was missing it)
- Collection filter pushed into searchVec DB query
- Filter-then-slice ordering fixed (MCP was slice-then-filter)
* feat: type-routed query expansion — lex→FTS, vec/hyde→vector
expandQuery() now returns typed ExpandedQuery[] instead of string[],
preserving the lex/vec/hyde type info from the LLM's GBNF-structured
output. hybridQuery() and vectorSearchQuery() route searches by type:
lex queries go to FTS only, vec/hyde go to vector only.
Previously, every expanded query ran through BOTH backends — keyword
variants wasted embedding forward passes, semantic paraphrases wasted
BM25 lookups. Type routing eliminates ~4 calls/query with zero quality
loss (cross-backend noise actually hurt RRF fusion).
Cache format changed from newline-separated text to JSON (preserves
types). Old cache entries gracefully re-expand on first access.
CLI expansion tree now shows query types:
├─ original query
├─ lex: keyword variant
├─ vec: semantic meaning
└─ hyde: hypothetical document...
Benchmark (5 queries, 1756-doc index, warm LLM, Apple Silicon):
Metric Old (untyped) New (typed) Delta
Avg backend calls 10.0 6.0 -40%
Total wall time 1278ms 549ms -57%
Avg saved/query — — 146ms
"authentication setup" 12 → 7 calls 511 → 112ms
"database migration strategy" 10 → 6 calls 182 → 106ms
"how to handle errors in API" 10 → 6 calls 216 → 121ms
"meeting notes from last week" 10 → 6 calls 228 → 110ms
"performance optimization" 8 → 5 calls 141 → 100ms
Savings come from skipped embed() calls (~30-80ms each). FTS is
synchronous SQLite (~0ms), so lex→FTS routing is free while
vec/hyde→vector-only avoids wasted embedding passes.
* fix: MCP query snippets now use reranker's best chunk, not full body
extractSnippet() was scanning the entire document body for keyword
matches to build the snippet. But hybridQuery() already identified
the most relevant chunk via cross-attention reranking — rescanning
the full body is redundant and can land on a less relevant section
if the query terms appear elsewhere in the document.
CLI was already using bestChunk (set during the refactor). MCP was
still using body — a pre-existing inconsistency, not a regression.
* feat: dynamic MCP instructions + tool annotations
The MCP server now generates instructions at startup from actual index
state and injects them into the initialize response. LLMs see collection
names, document counts, content descriptions, and search strategy
guidance in their system prompt — zero tool calls needed for orientation.
Previously, the only guidance was generic static tool descriptions and
a user-invocable "query" prompt that no LLM would discover on its own.
An LLM connecting to QMD had no idea what collections existed, what they
contained, or how to scope searches effectively.
* change default port to 8181
* fix: BM25 score normalization was inverted
The normalization formula `1 / (1 + |bm25|)` is a decreasing function of
match strength. FTS5 BM25 scores are negative where more negative = better
match (e.g., -10 is strong, -0.5 is weak). The formula mapped:
strong match (raw -10) → 1/(1+10) = 9% ← should be highest
weak match (raw -0.5) → 1/(1+0.5) = 67% ← should be lowest
Three downstream effects:
1. `--min-score 0.5` (or MCP minScore: 0.5) filtered OUT strong matches
and kept only weak ones. The MCP instructions recommend this threshold.
2. CLI `formatScore()` color bands never showed green for BM25 results
(best matches scored ~9%, green threshold is 70%).
3. The strong signal optimization in hybridQuery (skip ~2s LLM expansion
when BM25 already has a clear winner) was dead code — strong matches
scored ~0.09, never reaching the 0.85 threshold.
Fix: `|x| / (1 + |x|)` — same (0,1) range, monotonic, no per-query
normalization needed, but now correctly maps strong → high, weak → low.
The normalization was born broken (Math.max(0, x) clamped all
negative BM25 to 0 → every score = 1.0), then PR #76 changed to
Math.abs which made scores vary but inverted the direction. Neither
state was ever correct.
* fix: rerank cache key ignores chunk content
The rerank cache key was (query, file, model) but the actual text sent
to the reranker is a keyword-selected chunk that varies by query terms.
Two different queries hitting the same file can select different chunks,
but the second query gets a stale cached score from the first chunk.
Example:
Query "auth flow" → selects chunk about authentication → score 0.92
Query "auth tokens" → same file, selects chunk about tokens
→ cache HIT on (query, file, model) → returns 0.92 from wrong chunk
Fix: include full chunk text in cache key. getCacheKey() already
SHA-256 hashes its inputs, so this adds no key bloat — just
disambiguation. Old cache entries become natural misses (different key
shape) and re-warm on next query.
* rename MCP tools for clarity, rewrite descriptions for LLM tool selection
Rename MCP tools: vsearch → vector_search, query → deep_search.
LLMs see these names — self-documenting names reduce reliance on
descriptions for tool selection. CLI commands stay unchanged
(qmd vsearch, qmd query) — different namespace, users type those.
Rewrite all search tool descriptions to be action-oriented:
- search: "Search by keyword. Finds documents containing exact
words and phrases in the query."
- vector_search: "Search by meaning. Finds relevant documents even
when they use different words than the query — handles synonyms,
paraphrases, and related concepts."
- deep_search: "Deep search. Auto-expands the query into variations,
searches each by keyword and meaning, and reranks for top hits
across all results."
Rewrite instructions ladder — each tool says what it does, no
"start here" / "escalate as needed" strategy language.
Delete the "query" prompt (registerPrompt) — it restated what
descriptions + instructions already cover. No LLM proactively
calls prompts/get to learn how to use tools.
* supress HTTP server logs during tests
2026-02-11 05:37:33 +08:00
|
|
|
cwd: fixturesDir,
|
|
|
|
|
env: {
|
|
|
|
|
...process.env,
|
|
|
|
|
INDEX_PATH: daemonDbPath,
|
|
|
|
|
QMD_CONFIG_DIR: daemonConfigDir,
|
|
|
|
|
},
|
2026-02-16 04:57:13 +08:00
|
|
|
stdio: ["ignore", "pipe", "pipe"],
|
MCP: Streamable HTTP, scoring fixes, tool improvements (#149)
* feat: MCP HTTP transport with daemon lifecycle
Add streaming HTTP transport as an alternative to stdio for the MCP
server. A long-lived HTTP server avoids reloading 3 GGUF models (~2GB)
on every client connection, reducing warm query latency from ~16s (CLI)
to ~10s.
New CLI surface:
qmd mcp --http [--port N] # foreground, default port 3000
qmd mcp --http --daemon # background, PID in ~/.cache/qmd/mcp.pid
qmd mcp stop # stop daemon via PID file
qmd status # now shows MCP daemon liveness
Server implementation (mcp.ts):
- Extract createMcpServer(store) shared by stdio and HTTP transports
- HTTP transport uses WebStandardStreamableHTTPServerTransport with
JSON responses (stateless, no SSE)
- /health endpoint with uptime, /mcp for MCP protocol, 404 otherwise
- Request logging to stderr with timestamps, tool names, query args
Daemon lifecycle (qmd.ts):
- PID file + log file management with stale PID detection
- Absolute paths in Bun.spawn (process.execPath + import.meta.path)
so daemon works regardless of cwd
- mkdirSync for cache dir on fresh installs
- Removes top-level SIGTERM/SIGINT handlers before starting HTTP
server so async cleanup in mcp.ts actually runs
Move hybridQuery() and vectorSearchQuery() into store.ts as standalone
functions that take a Store as first argument. Both CLI and MCP now
call the identical pipeline, eliminating the class of bugs where one
copy drifts from the other.
Shared pipeline (store.ts):
- hybridQuery(): BM25 probe → expand → FTS+vec search → RRF →
chunk → rerank (chunks only) → position-aware blending → dedup
- vectorSearchQuery(): expand → vec search → dedup → sort
- SearchHooks interface for optional progress callbacks
- Constants: STRONG_SIGNAL_MIN_SCORE, STRONG_SIGNAL_MIN_GAP,
RERANK_CANDIDATE_LIMIT (40), addLineNumbers()
Bugs fixed by unification:
- MCP now gets strong-signal short-circuit (was CLI-only)
- Reranker candidate limit unified at 40 (MCP had 30)
- File dedup added to hybrid query (MCP was missing it)
- Collection filter pushed into searchVec DB query
- Filter-then-slice ordering fixed (MCP was slice-then-filter)
* feat: type-routed query expansion — lex→FTS, vec/hyde→vector
expandQuery() now returns typed ExpandedQuery[] instead of string[],
preserving the lex/vec/hyde type info from the LLM's GBNF-structured
output. hybridQuery() and vectorSearchQuery() route searches by type:
lex queries go to FTS only, vec/hyde go to vector only.
Previously, every expanded query ran through BOTH backends — keyword
variants wasted embedding forward passes, semantic paraphrases wasted
BM25 lookups. Type routing eliminates ~4 calls/query with zero quality
loss (cross-backend noise actually hurt RRF fusion).
Cache format changed from newline-separated text to JSON (preserves
types). Old cache entries gracefully re-expand on first access.
CLI expansion tree now shows query types:
├─ original query
├─ lex: keyword variant
├─ vec: semantic meaning
└─ hyde: hypothetical document...
Benchmark (5 queries, 1756-doc index, warm LLM, Apple Silicon):
Metric Old (untyped) New (typed) Delta
Avg backend calls 10.0 6.0 -40%
Total wall time 1278ms 549ms -57%
Avg saved/query — — 146ms
"authentication setup" 12 → 7 calls 511 → 112ms
"database migration strategy" 10 → 6 calls 182 → 106ms
"how to handle errors in API" 10 → 6 calls 216 → 121ms
"meeting notes from last week" 10 → 6 calls 228 → 110ms
"performance optimization" 8 → 5 calls 141 → 100ms
Savings come from skipped embed() calls (~30-80ms each). FTS is
synchronous SQLite (~0ms), so lex→FTS routing is free while
vec/hyde→vector-only avoids wasted embedding passes.
* fix: MCP query snippets now use reranker's best chunk, not full body
extractSnippet() was scanning the entire document body for keyword
matches to build the snippet. But hybridQuery() already identified
the most relevant chunk via cross-attention reranking — rescanning
the full body is redundant and can land on a less relevant section
if the query terms appear elsewhere in the document.
CLI was already using bestChunk (set during the refactor). MCP was
still using body — a pre-existing inconsistency, not a regression.
* feat: dynamic MCP instructions + tool annotations
The MCP server now generates instructions at startup from actual index
state and injects them into the initialize response. LLMs see collection
names, document counts, content descriptions, and search strategy
guidance in their system prompt — zero tool calls needed for orientation.
Previously, the only guidance was generic static tool descriptions and
a user-invocable "query" prompt that no LLM would discover on its own.
An LLM connecting to QMD had no idea what collections existed, what they
contained, or how to scope searches effectively.
* change default port to 8181
* fix: BM25 score normalization was inverted
The normalization formula `1 / (1 + |bm25|)` is a decreasing function of
match strength. FTS5 BM25 scores are negative where more negative = better
match (e.g., -10 is strong, -0.5 is weak). The formula mapped:
strong match (raw -10) → 1/(1+10) = 9% ← should be highest
weak match (raw -0.5) → 1/(1+0.5) = 67% ← should be lowest
Three downstream effects:
1. `--min-score 0.5` (or MCP minScore: 0.5) filtered OUT strong matches
and kept only weak ones. The MCP instructions recommend this threshold.
2. CLI `formatScore()` color bands never showed green for BM25 results
(best matches scored ~9%, green threshold is 70%).
3. The strong signal optimization in hybridQuery (skip ~2s LLM expansion
when BM25 already has a clear winner) was dead code — strong matches
scored ~0.09, never reaching the 0.85 threshold.
Fix: `|x| / (1 + |x|)` — same (0,1) range, monotonic, no per-query
normalization needed, but now correctly maps strong → high, weak → low.
The normalization was born broken (Math.max(0, x) clamped all
negative BM25 to 0 → every score = 1.0), then PR #76 changed to
Math.abs which made scores vary but inverted the direction. Neither
state was ever correct.
* fix: rerank cache key ignores chunk content
The rerank cache key was (query, file, model) but the actual text sent
to the reranker is a keyword-selected chunk that varies by query terms.
Two different queries hitting the same file can select different chunks,
but the second query gets a stale cached score from the first chunk.
Example:
Query "auth flow" → selects chunk about authentication → score 0.92
Query "auth tokens" → same file, selects chunk about tokens
→ cache HIT on (query, file, model) → returns 0.92 from wrong chunk
Fix: include full chunk text in cache key. getCacheKey() already
SHA-256 hashes its inputs, so this adds no key bloat — just
disambiguation. Old cache entries become natural misses (different key
shape) and re-warm on next query.
* rename MCP tools for clarity, rewrite descriptions for LLM tool selection
Rename MCP tools: vsearch → vector_search, query → deep_search.
LLMs see these names — self-documenting names reduce reliance on
descriptions for tool selection. CLI commands stay unchanged
(qmd vsearch, qmd query) — different namespace, users type those.
Rewrite all search tool descriptions to be action-oriented:
- search: "Search by keyword. Finds documents containing exact
words and phrases in the query."
- vector_search: "Search by meaning. Finds relevant documents even
when they use different words than the query — handles synonyms,
paraphrases, and related concepts."
- deep_search: "Deep search. Auto-expands the query into variations,
searches each by keyword and meaning, and reranks for top hits
across all results."
Rewrite instructions ladder — each tool says what it does, no
"start here" / "escalate as needed" strategy language.
Delete the "query" prompt (registerPrompt) — it restated what
descriptions + instructions already cover. No LLM proactively
calls prompts/get to learn how to use tools.
* supress HTTP server logs during tests
2026-02-11 05:37:33 +08:00
|
|
|
});
|
2026-02-16 04:57:13 +08:00
|
|
|
if (proc.pid) spawnedPids.push(proc.pid);
|
MCP: Streamable HTTP, scoring fixes, tool improvements (#149)
* feat: MCP HTTP transport with daemon lifecycle
Add streaming HTTP transport as an alternative to stdio for the MCP
server. A long-lived HTTP server avoids reloading 3 GGUF models (~2GB)
on every client connection, reducing warm query latency from ~16s (CLI)
to ~10s.
New CLI surface:
qmd mcp --http [--port N] # foreground, default port 3000
qmd mcp --http --daemon # background, PID in ~/.cache/qmd/mcp.pid
qmd mcp stop # stop daemon via PID file
qmd status # now shows MCP daemon liveness
Server implementation (mcp.ts):
- Extract createMcpServer(store) shared by stdio and HTTP transports
- HTTP transport uses WebStandardStreamableHTTPServerTransport with
JSON responses (stateless, no SSE)
- /health endpoint with uptime, /mcp for MCP protocol, 404 otherwise
- Request logging to stderr with timestamps, tool names, query args
Daemon lifecycle (qmd.ts):
- PID file + log file management with stale PID detection
- Absolute paths in Bun.spawn (process.execPath + import.meta.path)
so daemon works regardless of cwd
- mkdirSync for cache dir on fresh installs
- Removes top-level SIGTERM/SIGINT handlers before starting HTTP
server so async cleanup in mcp.ts actually runs
Move hybridQuery() and vectorSearchQuery() into store.ts as standalone
functions that take a Store as first argument. Both CLI and MCP now
call the identical pipeline, eliminating the class of bugs where one
copy drifts from the other.
Shared pipeline (store.ts):
- hybridQuery(): BM25 probe → expand → FTS+vec search → RRF →
chunk → rerank (chunks only) → position-aware blending → dedup
- vectorSearchQuery(): expand → vec search → dedup → sort
- SearchHooks interface for optional progress callbacks
- Constants: STRONG_SIGNAL_MIN_SCORE, STRONG_SIGNAL_MIN_GAP,
RERANK_CANDIDATE_LIMIT (40), addLineNumbers()
Bugs fixed by unification:
- MCP now gets strong-signal short-circuit (was CLI-only)
- Reranker candidate limit unified at 40 (MCP had 30)
- File dedup added to hybrid query (MCP was missing it)
- Collection filter pushed into searchVec DB query
- Filter-then-slice ordering fixed (MCP was slice-then-filter)
* feat: type-routed query expansion — lex→FTS, vec/hyde→vector
expandQuery() now returns typed ExpandedQuery[] instead of string[],
preserving the lex/vec/hyde type info from the LLM's GBNF-structured
output. hybridQuery() and vectorSearchQuery() route searches by type:
lex queries go to FTS only, vec/hyde go to vector only.
Previously, every expanded query ran through BOTH backends — keyword
variants wasted embedding forward passes, semantic paraphrases wasted
BM25 lookups. Type routing eliminates ~4 calls/query with zero quality
loss (cross-backend noise actually hurt RRF fusion).
Cache format changed from newline-separated text to JSON (preserves
types). Old cache entries gracefully re-expand on first access.
CLI expansion tree now shows query types:
├─ original query
├─ lex: keyword variant
├─ vec: semantic meaning
└─ hyde: hypothetical document...
Benchmark (5 queries, 1756-doc index, warm LLM, Apple Silicon):
Metric Old (untyped) New (typed) Delta
Avg backend calls 10.0 6.0 -40%
Total wall time 1278ms 549ms -57%
Avg saved/query — — 146ms
"authentication setup" 12 → 7 calls 511 → 112ms
"database migration strategy" 10 → 6 calls 182 → 106ms
"how to handle errors in API" 10 → 6 calls 216 → 121ms
"meeting notes from last week" 10 → 6 calls 228 → 110ms
"performance optimization" 8 → 5 calls 141 → 100ms
Savings come from skipped embed() calls (~30-80ms each). FTS is
synchronous SQLite (~0ms), so lex→FTS routing is free while
vec/hyde→vector-only avoids wasted embedding passes.
* fix: MCP query snippets now use reranker's best chunk, not full body
extractSnippet() was scanning the entire document body for keyword
matches to build the snippet. But hybridQuery() already identified
the most relevant chunk via cross-attention reranking — rescanning
the full body is redundant and can land on a less relevant section
if the query terms appear elsewhere in the document.
CLI was already using bestChunk (set during the refactor). MCP was
still using body — a pre-existing inconsistency, not a regression.
* feat: dynamic MCP instructions + tool annotations
The MCP server now generates instructions at startup from actual index
state and injects them into the initialize response. LLMs see collection
names, document counts, content descriptions, and search strategy
guidance in their system prompt — zero tool calls needed for orientation.
Previously, the only guidance was generic static tool descriptions and
a user-invocable "query" prompt that no LLM would discover on its own.
An LLM connecting to QMD had no idea what collections existed, what they
contained, or how to scope searches effectively.
* change default port to 8181
* fix: BM25 score normalization was inverted
The normalization formula `1 / (1 + |bm25|)` is a decreasing function of
match strength. FTS5 BM25 scores are negative where more negative = better
match (e.g., -10 is strong, -0.5 is weak). The formula mapped:
strong match (raw -10) → 1/(1+10) = 9% ← should be highest
weak match (raw -0.5) → 1/(1+0.5) = 67% ← should be lowest
Three downstream effects:
1. `--min-score 0.5` (or MCP minScore: 0.5) filtered OUT strong matches
and kept only weak ones. The MCP instructions recommend this threshold.
2. CLI `formatScore()` color bands never showed green for BM25 results
(best matches scored ~9%, green threshold is 70%).
3. The strong signal optimization in hybridQuery (skip ~2s LLM expansion
when BM25 already has a clear winner) was dead code — strong matches
scored ~0.09, never reaching the 0.85 threshold.
Fix: `|x| / (1 + |x|)` — same (0,1) range, monotonic, no per-query
normalization needed, but now correctly maps strong → high, weak → low.
The normalization was born broken (Math.max(0, x) clamped all
negative BM25 to 0 → every score = 1.0), then PR #76 changed to
Math.abs which made scores vary but inverted the direction. Neither
state was ever correct.
* fix: rerank cache key ignores chunk content
The rerank cache key was (query, file, model) but the actual text sent
to the reranker is a keyword-selected chunk that varies by query terms.
Two different queries hitting the same file can select different chunks,
but the second query gets a stale cached score from the first chunk.
Example:
Query "auth flow" → selects chunk about authentication → score 0.92
Query "auth tokens" → same file, selects chunk about tokens
→ cache HIT on (query, file, model) → returns 0.92 from wrong chunk
Fix: include full chunk text in cache key. getCacheKey() already
SHA-256 hashes its inputs, so this adds no key bloat — just
disambiguation. Old cache entries become natural misses (different key
shape) and re-warm on next query.
* rename MCP tools for clarity, rewrite descriptions for LLM tool selection
Rename MCP tools: vsearch → vector_search, query → deep_search.
LLMs see these names — self-documenting names reduce reliance on
descriptions for tool selection. CLI commands stay unchanged
(qmd vsearch, qmd query) — different namespace, users type those.
Rewrite all search tool descriptions to be action-oriented:
- search: "Search by keyword. Finds documents containing exact
words and phrases in the query."
- vector_search: "Search by meaning. Finds relevant documents even
when they use different words than the query — handles synonyms,
paraphrases, and related concepts."
- deep_search: "Deep search. Auto-expands the query into variations,
searches each by keyword and meaning, and reranks for top hits
across all results."
Rewrite instructions ladder — each tool says what it does, no
"start here" / "escalate as needed" strategy language.
Delete the "query" prompt (registerPrompt) — it restated what
descriptions + instructions already cover. No LLM proactively
calls prompts/get to learn how to use tools.
* supress HTTP server logs during tests
2026-02-11 05:37:33 +08:00
|
|
|
return proc;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/** Wait for HTTP server to become ready */
|
|
|
|
|
async function waitForServer(port: number, timeoutMs = 5000): Promise<boolean> {
|
|
|
|
|
const deadline = Date.now() + timeoutMs;
|
|
|
|
|
while (Date.now() < deadline) {
|
|
|
|
|
try {
|
|
|
|
|
const res = await fetch(`http://localhost:${port}/health`);
|
|
|
|
|
if (res.ok) return true;
|
|
|
|
|
} catch { /* not ready yet */ }
|
2026-02-16 04:57:13 +08:00
|
|
|
await sleep(200);
|
MCP: Streamable HTTP, scoring fixes, tool improvements (#149)
* feat: MCP HTTP transport with daemon lifecycle
Add streaming HTTP transport as an alternative to stdio for the MCP
server. A long-lived HTTP server avoids reloading 3 GGUF models (~2GB)
on every client connection, reducing warm query latency from ~16s (CLI)
to ~10s.
New CLI surface:
qmd mcp --http [--port N] # foreground, default port 3000
qmd mcp --http --daemon # background, PID in ~/.cache/qmd/mcp.pid
qmd mcp stop # stop daemon via PID file
qmd status # now shows MCP daemon liveness
Server implementation (mcp.ts):
- Extract createMcpServer(store) shared by stdio and HTTP transports
- HTTP transport uses WebStandardStreamableHTTPServerTransport with
JSON responses (stateless, no SSE)
- /health endpoint with uptime, /mcp for MCP protocol, 404 otherwise
- Request logging to stderr with timestamps, tool names, query args
Daemon lifecycle (qmd.ts):
- PID file + log file management with stale PID detection
- Absolute paths in Bun.spawn (process.execPath + import.meta.path)
so daemon works regardless of cwd
- mkdirSync for cache dir on fresh installs
- Removes top-level SIGTERM/SIGINT handlers before starting HTTP
server so async cleanup in mcp.ts actually runs
Move hybridQuery() and vectorSearchQuery() into store.ts as standalone
functions that take a Store as first argument. Both CLI and MCP now
call the identical pipeline, eliminating the class of bugs where one
copy drifts from the other.
Shared pipeline (store.ts):
- hybridQuery(): BM25 probe → expand → FTS+vec search → RRF →
chunk → rerank (chunks only) → position-aware blending → dedup
- vectorSearchQuery(): expand → vec search → dedup → sort
- SearchHooks interface for optional progress callbacks
- Constants: STRONG_SIGNAL_MIN_SCORE, STRONG_SIGNAL_MIN_GAP,
RERANK_CANDIDATE_LIMIT (40), addLineNumbers()
Bugs fixed by unification:
- MCP now gets strong-signal short-circuit (was CLI-only)
- Reranker candidate limit unified at 40 (MCP had 30)
- File dedup added to hybrid query (MCP was missing it)
- Collection filter pushed into searchVec DB query
- Filter-then-slice ordering fixed (MCP was slice-then-filter)
* feat: type-routed query expansion — lex→FTS, vec/hyde→vector
expandQuery() now returns typed ExpandedQuery[] instead of string[],
preserving the lex/vec/hyde type info from the LLM's GBNF-structured
output. hybridQuery() and vectorSearchQuery() route searches by type:
lex queries go to FTS only, vec/hyde go to vector only.
Previously, every expanded query ran through BOTH backends — keyword
variants wasted embedding forward passes, semantic paraphrases wasted
BM25 lookups. Type routing eliminates ~4 calls/query with zero quality
loss (cross-backend noise actually hurt RRF fusion).
Cache format changed from newline-separated text to JSON (preserves
types). Old cache entries gracefully re-expand on first access.
CLI expansion tree now shows query types:
├─ original query
├─ lex: keyword variant
├─ vec: semantic meaning
└─ hyde: hypothetical document...
Benchmark (5 queries, 1756-doc index, warm LLM, Apple Silicon):
Metric Old (untyped) New (typed) Delta
Avg backend calls 10.0 6.0 -40%
Total wall time 1278ms 549ms -57%
Avg saved/query — — 146ms
"authentication setup" 12 → 7 calls 511 → 112ms
"database migration strategy" 10 → 6 calls 182 → 106ms
"how to handle errors in API" 10 → 6 calls 216 → 121ms
"meeting notes from last week" 10 → 6 calls 228 → 110ms
"performance optimization" 8 → 5 calls 141 → 100ms
Savings come from skipped embed() calls (~30-80ms each). FTS is
synchronous SQLite (~0ms), so lex→FTS routing is free while
vec/hyde→vector-only avoids wasted embedding passes.
* fix: MCP query snippets now use reranker's best chunk, not full body
extractSnippet() was scanning the entire document body for keyword
matches to build the snippet. But hybridQuery() already identified
the most relevant chunk via cross-attention reranking — rescanning
the full body is redundant and can land on a less relevant section
if the query terms appear elsewhere in the document.
CLI was already using bestChunk (set during the refactor). MCP was
still using body — a pre-existing inconsistency, not a regression.
* feat: dynamic MCP instructions + tool annotations
The MCP server now generates instructions at startup from actual index
state and injects them into the initialize response. LLMs see collection
names, document counts, content descriptions, and search strategy
guidance in their system prompt — zero tool calls needed for orientation.
Previously, the only guidance was generic static tool descriptions and
a user-invocable "query" prompt that no LLM would discover on its own.
An LLM connecting to QMD had no idea what collections existed, what they
contained, or how to scope searches effectively.
* change default port to 8181
* fix: BM25 score normalization was inverted
The normalization formula `1 / (1 + |bm25|)` is a decreasing function of
match strength. FTS5 BM25 scores are negative where more negative = better
match (e.g., -10 is strong, -0.5 is weak). The formula mapped:
strong match (raw -10) → 1/(1+10) = 9% ← should be highest
weak match (raw -0.5) → 1/(1+0.5) = 67% ← should be lowest
Three downstream effects:
1. `--min-score 0.5` (or MCP minScore: 0.5) filtered OUT strong matches
and kept only weak ones. The MCP instructions recommend this threshold.
2. CLI `formatScore()` color bands never showed green for BM25 results
(best matches scored ~9%, green threshold is 70%).
3. The strong signal optimization in hybridQuery (skip ~2s LLM expansion
when BM25 already has a clear winner) was dead code — strong matches
scored ~0.09, never reaching the 0.85 threshold.
Fix: `|x| / (1 + |x|)` — same (0,1) range, monotonic, no per-query
normalization needed, but now correctly maps strong → high, weak → low.
The normalization was born broken (Math.max(0, x) clamped all
negative BM25 to 0 → every score = 1.0), then PR #76 changed to
Math.abs which made scores vary but inverted the direction. Neither
state was ever correct.
* fix: rerank cache key ignores chunk content
The rerank cache key was (query, file, model) but the actual text sent
to the reranker is a keyword-selected chunk that varies by query terms.
Two different queries hitting the same file can select different chunks,
but the second query gets a stale cached score from the first chunk.
Example:
Query "auth flow" → selects chunk about authentication → score 0.92
Query "auth tokens" → same file, selects chunk about tokens
→ cache HIT on (query, file, model) → returns 0.92 from wrong chunk
Fix: include full chunk text in cache key. getCacheKey() already
SHA-256 hashes its inputs, so this adds no key bloat — just
disambiguation. Old cache entries become natural misses (different key
shape) and re-warm on next query.
* rename MCP tools for clarity, rewrite descriptions for LLM tool selection
Rename MCP tools: vsearch → vector_search, query → deep_search.
LLMs see these names — self-documenting names reduce reliance on
descriptions for tool selection. CLI commands stay unchanged
(qmd vsearch, qmd query) — different namespace, users type those.
Rewrite all search tool descriptions to be action-oriented:
- search: "Search by keyword. Finds documents containing exact
words and phrases in the query."
- vector_search: "Search by meaning. Finds relevant documents even
when they use different words than the query — handles synonyms,
paraphrases, and related concepts."
- deep_search: "Deep search. Auto-expands the query into variations,
searches each by keyword and meaning, and reranks for top hits
across all results."
Rewrite instructions ladder — each tool says what it does, no
"start here" / "escalate as needed" strategy language.
Delete the "query" prompt (registerPrompt) — it restated what
descriptions + instructions already cover. No LLM proactively
calls prompts/get to learn how to use tools.
* supress HTTP server logs during tests
2026-02-11 05:37:33 +08:00
|
|
|
}
|
|
|
|
|
return false;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/** Pick a random high port unlikely to conflict */
|
|
|
|
|
function randomPort(): number {
|
|
|
|
|
return 10000 + Math.floor(Math.random() * 50000);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
beforeAll(async () => {
|
|
|
|
|
daemonTestDir = await mkdtemp(join(tmpdir(), "qmd-daemon-test-"));
|
|
|
|
|
daemonCacheDir = join(daemonTestDir, "cache");
|
|
|
|
|
daemonDbPath = join(daemonTestDir, "test.sqlite");
|
|
|
|
|
daemonConfigDir = join(daemonTestDir, "config");
|
|
|
|
|
|
|
|
|
|
await mkdir(join(daemonCacheDir, "qmd"), { recursive: true });
|
|
|
|
|
await mkdir(daemonConfigDir, { recursive: true });
|
|
|
|
|
await writeFile(join(daemonConfigDir, "index.yml"), "collections: {}\n");
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
afterAll(async () => {
|
|
|
|
|
// Kill any leftover spawned processes
|
|
|
|
|
for (const pid of spawnedPids) {
|
|
|
|
|
try { process.kill(pid, "SIGTERM"); } catch { /* already dead */ }
|
|
|
|
|
}
|
|
|
|
|
// Also clean up via PID file if present
|
|
|
|
|
try {
|
|
|
|
|
const pf = pidPath();
|
|
|
|
|
if (existsSync(pf)) {
|
|
|
|
|
const pid = parseInt(readFileSync(pf, "utf-8").trim());
|
|
|
|
|
try { process.kill(pid, "SIGTERM"); } catch {}
|
|
|
|
|
unlinkSync(pf);
|
|
|
|
|
}
|
|
|
|
|
} catch {}
|
|
|
|
|
|
|
|
|
|
await rm(daemonTestDir, { recursive: true, force: true });
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
// -------------------------------------------------------------------------
|
|
|
|
|
// Foreground HTTP
|
|
|
|
|
// -------------------------------------------------------------------------
|
|
|
|
|
|
|
|
|
|
test("foreground HTTP server starts and responds to health check", async () => {
|
|
|
|
|
const port = randomPort();
|
|
|
|
|
const proc = spawnHttpServer(port);
|
|
|
|
|
|
|
|
|
|
try {
|
|
|
|
|
const ready = await waitForServer(port);
|
|
|
|
|
expect(ready).toBe(true);
|
|
|
|
|
|
|
|
|
|
const res = await fetch(`http://localhost:${port}/health`);
|
|
|
|
|
expect(res.status).toBe(200);
|
|
|
|
|
const body = await res.json();
|
|
|
|
|
expect(body.status).toBe("ok");
|
|
|
|
|
} finally {
|
|
|
|
|
proc.kill("SIGTERM");
|
2026-02-16 04:57:13 +08:00
|
|
|
await new Promise(r => proc.on("close", r));
|
MCP: Streamable HTTP, scoring fixes, tool improvements (#149)
* feat: MCP HTTP transport with daemon lifecycle
Add streaming HTTP transport as an alternative to stdio for the MCP
server. A long-lived HTTP server avoids reloading 3 GGUF models (~2GB)
on every client connection, reducing warm query latency from ~16s (CLI)
to ~10s.
New CLI surface:
qmd mcp --http [--port N] # foreground, default port 3000
qmd mcp --http --daemon # background, PID in ~/.cache/qmd/mcp.pid
qmd mcp stop # stop daemon via PID file
qmd status # now shows MCP daemon liveness
Server implementation (mcp.ts):
- Extract createMcpServer(store) shared by stdio and HTTP transports
- HTTP transport uses WebStandardStreamableHTTPServerTransport with
JSON responses (stateless, no SSE)
- /health endpoint with uptime, /mcp for MCP protocol, 404 otherwise
- Request logging to stderr with timestamps, tool names, query args
Daemon lifecycle (qmd.ts):
- PID file + log file management with stale PID detection
- Absolute paths in Bun.spawn (process.execPath + import.meta.path)
so daemon works regardless of cwd
- mkdirSync for cache dir on fresh installs
- Removes top-level SIGTERM/SIGINT handlers before starting HTTP
server so async cleanup in mcp.ts actually runs
Move hybridQuery() and vectorSearchQuery() into store.ts as standalone
functions that take a Store as first argument. Both CLI and MCP now
call the identical pipeline, eliminating the class of bugs where one
copy drifts from the other.
Shared pipeline (store.ts):
- hybridQuery(): BM25 probe → expand → FTS+vec search → RRF →
chunk → rerank (chunks only) → position-aware blending → dedup
- vectorSearchQuery(): expand → vec search → dedup → sort
- SearchHooks interface for optional progress callbacks
- Constants: STRONG_SIGNAL_MIN_SCORE, STRONG_SIGNAL_MIN_GAP,
RERANK_CANDIDATE_LIMIT (40), addLineNumbers()
Bugs fixed by unification:
- MCP now gets strong-signal short-circuit (was CLI-only)
- Reranker candidate limit unified at 40 (MCP had 30)
- File dedup added to hybrid query (MCP was missing it)
- Collection filter pushed into searchVec DB query
- Filter-then-slice ordering fixed (MCP was slice-then-filter)
* feat: type-routed query expansion — lex→FTS, vec/hyde→vector
expandQuery() now returns typed ExpandedQuery[] instead of string[],
preserving the lex/vec/hyde type info from the LLM's GBNF-structured
output. hybridQuery() and vectorSearchQuery() route searches by type:
lex queries go to FTS only, vec/hyde go to vector only.
Previously, every expanded query ran through BOTH backends — keyword
variants wasted embedding forward passes, semantic paraphrases wasted
BM25 lookups. Type routing eliminates ~4 calls/query with zero quality
loss (cross-backend noise actually hurt RRF fusion).
Cache format changed from newline-separated text to JSON (preserves
types). Old cache entries gracefully re-expand on first access.
CLI expansion tree now shows query types:
├─ original query
├─ lex: keyword variant
├─ vec: semantic meaning
└─ hyde: hypothetical document...
Benchmark (5 queries, 1756-doc index, warm LLM, Apple Silicon):
Metric Old (untyped) New (typed) Delta
Avg backend calls 10.0 6.0 -40%
Total wall time 1278ms 549ms -57%
Avg saved/query — — 146ms
"authentication setup" 12 → 7 calls 511 → 112ms
"database migration strategy" 10 → 6 calls 182 → 106ms
"how to handle errors in API" 10 → 6 calls 216 → 121ms
"meeting notes from last week" 10 → 6 calls 228 → 110ms
"performance optimization" 8 → 5 calls 141 → 100ms
Savings come from skipped embed() calls (~30-80ms each). FTS is
synchronous SQLite (~0ms), so lex→FTS routing is free while
vec/hyde→vector-only avoids wasted embedding passes.
* fix: MCP query snippets now use reranker's best chunk, not full body
extractSnippet() was scanning the entire document body for keyword
matches to build the snippet. But hybridQuery() already identified
the most relevant chunk via cross-attention reranking — rescanning
the full body is redundant and can land on a less relevant section
if the query terms appear elsewhere in the document.
CLI was already using bestChunk (set during the refactor). MCP was
still using body — a pre-existing inconsistency, not a regression.
* feat: dynamic MCP instructions + tool annotations
The MCP server now generates instructions at startup from actual index
state and injects them into the initialize response. LLMs see collection
names, document counts, content descriptions, and search strategy
guidance in their system prompt — zero tool calls needed for orientation.
Previously, the only guidance was generic static tool descriptions and
a user-invocable "query" prompt that no LLM would discover on its own.
An LLM connecting to QMD had no idea what collections existed, what they
contained, or how to scope searches effectively.
* change default port to 8181
* fix: BM25 score normalization was inverted
The normalization formula `1 / (1 + |bm25|)` is a decreasing function of
match strength. FTS5 BM25 scores are negative where more negative = better
match (e.g., -10 is strong, -0.5 is weak). The formula mapped:
strong match (raw -10) → 1/(1+10) = 9% ← should be highest
weak match (raw -0.5) → 1/(1+0.5) = 67% ← should be lowest
Three downstream effects:
1. `--min-score 0.5` (or MCP minScore: 0.5) filtered OUT strong matches
and kept only weak ones. The MCP instructions recommend this threshold.
2. CLI `formatScore()` color bands never showed green for BM25 results
(best matches scored ~9%, green threshold is 70%).
3. The strong signal optimization in hybridQuery (skip ~2s LLM expansion
when BM25 already has a clear winner) was dead code — strong matches
scored ~0.09, never reaching the 0.85 threshold.
Fix: `|x| / (1 + |x|)` — same (0,1) range, monotonic, no per-query
normalization needed, but now correctly maps strong → high, weak → low.
The normalization was born broken (Math.max(0, x) clamped all
negative BM25 to 0 → every score = 1.0), then PR #76 changed to
Math.abs which made scores vary but inverted the direction. Neither
state was ever correct.
* fix: rerank cache key ignores chunk content
The rerank cache key was (query, file, model) but the actual text sent
to the reranker is a keyword-selected chunk that varies by query terms.
Two different queries hitting the same file can select different chunks,
but the second query gets a stale cached score from the first chunk.
Example:
Query "auth flow" → selects chunk about authentication → score 0.92
Query "auth tokens" → same file, selects chunk about tokens
→ cache HIT on (query, file, model) → returns 0.92 from wrong chunk
Fix: include full chunk text in cache key. getCacheKey() already
SHA-256 hashes its inputs, so this adds no key bloat — just
disambiguation. Old cache entries become natural misses (different key
shape) and re-warm on next query.
* rename MCP tools for clarity, rewrite descriptions for LLM tool selection
Rename MCP tools: vsearch → vector_search, query → deep_search.
LLMs see these names — self-documenting names reduce reliance on
descriptions for tool selection. CLI commands stay unchanged
(qmd vsearch, qmd query) — different namespace, users type those.
Rewrite all search tool descriptions to be action-oriented:
- search: "Search by keyword. Finds documents containing exact
words and phrases in the query."
- vector_search: "Search by meaning. Finds relevant documents even
when they use different words than the query — handles synonyms,
paraphrases, and related concepts."
- deep_search: "Deep search. Auto-expands the query into variations,
searches each by keyword and meaning, and reranks for top hits
across all results."
Rewrite instructions ladder — each tool says what it does, no
"start here" / "escalate as needed" strategy language.
Delete the "query" prompt (registerPrompt) — it restated what
descriptions + instructions already cover. No LLM proactively
calls prompts/get to learn how to use tools.
* supress HTTP server logs during tests
2026-02-11 05:37:33 +08:00
|
|
|
}
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
// -------------------------------------------------------------------------
|
|
|
|
|
// Daemon lifecycle
|
|
|
|
|
// -------------------------------------------------------------------------
|
|
|
|
|
|
|
|
|
|
test("--daemon writes PID file and starts server", async () => {
|
|
|
|
|
const port = randomPort();
|
|
|
|
|
const { stdout, exitCode } = await runDaemonQmd([
|
|
|
|
|
"mcp", "--http", "--daemon", "--port", String(port),
|
|
|
|
|
]);
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain(`http://localhost:${port}/mcp`);
|
|
|
|
|
|
|
|
|
|
// PID file should exist
|
|
|
|
|
expect(existsSync(pidPath())).toBe(true);
|
|
|
|
|
|
|
|
|
|
const pid = parseInt(readFileSync(pidPath(), "utf-8").trim());
|
|
|
|
|
spawnedPids.push(pid);
|
|
|
|
|
|
|
|
|
|
// Server should be reachable
|
|
|
|
|
const ready = await waitForServer(port);
|
|
|
|
|
expect(ready).toBe(true);
|
|
|
|
|
|
|
|
|
|
// Clean up
|
|
|
|
|
process.kill(pid, "SIGTERM");
|
2026-02-16 04:57:13 +08:00
|
|
|
await sleep(500);
|
2026-02-21 09:52:53 +08:00
|
|
|
try { unlinkSync(pidPath()); } catch {}
|
MCP: Streamable HTTP, scoring fixes, tool improvements (#149)
* feat: MCP HTTP transport with daemon lifecycle
Add streaming HTTP transport as an alternative to stdio for the MCP
server. A long-lived HTTP server avoids reloading 3 GGUF models (~2GB)
on every client connection, reducing warm query latency from ~16s (CLI)
to ~10s.
New CLI surface:
qmd mcp --http [--port N] # foreground, default port 3000
qmd mcp --http --daemon # background, PID in ~/.cache/qmd/mcp.pid
qmd mcp stop # stop daemon via PID file
qmd status # now shows MCP daemon liveness
Server implementation (mcp.ts):
- Extract createMcpServer(store) shared by stdio and HTTP transports
- HTTP transport uses WebStandardStreamableHTTPServerTransport with
JSON responses (stateless, no SSE)
- /health endpoint with uptime, /mcp for MCP protocol, 404 otherwise
- Request logging to stderr with timestamps, tool names, query args
Daemon lifecycle (qmd.ts):
- PID file + log file management with stale PID detection
- Absolute paths in Bun.spawn (process.execPath + import.meta.path)
so daemon works regardless of cwd
- mkdirSync for cache dir on fresh installs
- Removes top-level SIGTERM/SIGINT handlers before starting HTTP
server so async cleanup in mcp.ts actually runs
Move hybridQuery() and vectorSearchQuery() into store.ts as standalone
functions that take a Store as first argument. Both CLI and MCP now
call the identical pipeline, eliminating the class of bugs where one
copy drifts from the other.
Shared pipeline (store.ts):
- hybridQuery(): BM25 probe → expand → FTS+vec search → RRF →
chunk → rerank (chunks only) → position-aware blending → dedup
- vectorSearchQuery(): expand → vec search → dedup → sort
- SearchHooks interface for optional progress callbacks
- Constants: STRONG_SIGNAL_MIN_SCORE, STRONG_SIGNAL_MIN_GAP,
RERANK_CANDIDATE_LIMIT (40), addLineNumbers()
Bugs fixed by unification:
- MCP now gets strong-signal short-circuit (was CLI-only)
- Reranker candidate limit unified at 40 (MCP had 30)
- File dedup added to hybrid query (MCP was missing it)
- Collection filter pushed into searchVec DB query
- Filter-then-slice ordering fixed (MCP was slice-then-filter)
* feat: type-routed query expansion — lex→FTS, vec/hyde→vector
expandQuery() now returns typed ExpandedQuery[] instead of string[],
preserving the lex/vec/hyde type info from the LLM's GBNF-structured
output. hybridQuery() and vectorSearchQuery() route searches by type:
lex queries go to FTS only, vec/hyde go to vector only.
Previously, every expanded query ran through BOTH backends — keyword
variants wasted embedding forward passes, semantic paraphrases wasted
BM25 lookups. Type routing eliminates ~4 calls/query with zero quality
loss (cross-backend noise actually hurt RRF fusion).
Cache format changed from newline-separated text to JSON (preserves
types). Old cache entries gracefully re-expand on first access.
CLI expansion tree now shows query types:
├─ original query
├─ lex: keyword variant
├─ vec: semantic meaning
└─ hyde: hypothetical document...
Benchmark (5 queries, 1756-doc index, warm LLM, Apple Silicon):
Metric Old (untyped) New (typed) Delta
Avg backend calls 10.0 6.0 -40%
Total wall time 1278ms 549ms -57%
Avg saved/query — — 146ms
"authentication setup" 12 → 7 calls 511 → 112ms
"database migration strategy" 10 → 6 calls 182 → 106ms
"how to handle errors in API" 10 → 6 calls 216 → 121ms
"meeting notes from last week" 10 → 6 calls 228 → 110ms
"performance optimization" 8 → 5 calls 141 → 100ms
Savings come from skipped embed() calls (~30-80ms each). FTS is
synchronous SQLite (~0ms), so lex→FTS routing is free while
vec/hyde→vector-only avoids wasted embedding passes.
* fix: MCP query snippets now use reranker's best chunk, not full body
extractSnippet() was scanning the entire document body for keyword
matches to build the snippet. But hybridQuery() already identified
the most relevant chunk via cross-attention reranking — rescanning
the full body is redundant and can land on a less relevant section
if the query terms appear elsewhere in the document.
CLI was already using bestChunk (set during the refactor). MCP was
still using body — a pre-existing inconsistency, not a regression.
* feat: dynamic MCP instructions + tool annotations
The MCP server now generates instructions at startup from actual index
state and injects them into the initialize response. LLMs see collection
names, document counts, content descriptions, and search strategy
guidance in their system prompt — zero tool calls needed for orientation.
Previously, the only guidance was generic static tool descriptions and
a user-invocable "query" prompt that no LLM would discover on its own.
An LLM connecting to QMD had no idea what collections existed, what they
contained, or how to scope searches effectively.
* change default port to 8181
* fix: BM25 score normalization was inverted
The normalization formula `1 / (1 + |bm25|)` is a decreasing function of
match strength. FTS5 BM25 scores are negative where more negative = better
match (e.g., -10 is strong, -0.5 is weak). The formula mapped:
strong match (raw -10) → 1/(1+10) = 9% ← should be highest
weak match (raw -0.5) → 1/(1+0.5) = 67% ← should be lowest
Three downstream effects:
1. `--min-score 0.5` (or MCP minScore: 0.5) filtered OUT strong matches
and kept only weak ones. The MCP instructions recommend this threshold.
2. CLI `formatScore()` color bands never showed green for BM25 results
(best matches scored ~9%, green threshold is 70%).
3. The strong signal optimization in hybridQuery (skip ~2s LLM expansion
when BM25 already has a clear winner) was dead code — strong matches
scored ~0.09, never reaching the 0.85 threshold.
Fix: `|x| / (1 + |x|)` — same (0,1) range, monotonic, no per-query
normalization needed, but now correctly maps strong → high, weak → low.
The normalization was born broken (Math.max(0, x) clamped all
negative BM25 to 0 → every score = 1.0), then PR #76 changed to
Math.abs which made scores vary but inverted the direction. Neither
state was ever correct.
* fix: rerank cache key ignores chunk content
The rerank cache key was (query, file, model) but the actual text sent
to the reranker is a keyword-selected chunk that varies by query terms.
Two different queries hitting the same file can select different chunks,
but the second query gets a stale cached score from the first chunk.
Example:
Query "auth flow" → selects chunk about authentication → score 0.92
Query "auth tokens" → same file, selects chunk about tokens
→ cache HIT on (query, file, model) → returns 0.92 from wrong chunk
Fix: include full chunk text in cache key. getCacheKey() already
SHA-256 hashes its inputs, so this adds no key bloat — just
disambiguation. Old cache entries become natural misses (different key
shape) and re-warm on next query.
* rename MCP tools for clarity, rewrite descriptions for LLM tool selection
Rename MCP tools: vsearch → vector_search, query → deep_search.
LLMs see these names — self-documenting names reduce reliance on
descriptions for tool selection. CLI commands stay unchanged
(qmd vsearch, qmd query) — different namespace, users type those.
Rewrite all search tool descriptions to be action-oriented:
- search: "Search by keyword. Finds documents containing exact
words and phrases in the query."
- vector_search: "Search by meaning. Finds relevant documents even
when they use different words than the query — handles synonyms,
paraphrases, and related concepts."
- deep_search: "Deep search. Auto-expands the query into variations,
searches each by keyword and meaning, and reranks for top hits
across all results."
Rewrite instructions ladder — each tool says what it does, no
"start here" / "escalate as needed" strategy language.
Delete the "query" prompt (registerPrompt) — it restated what
descriptions + instructions already cover. No LLM proactively
calls prompts/get to learn how to use tools.
* supress HTTP server logs during tests
2026-02-11 05:37:33 +08:00
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("stop kills daemon and removes PID file", async () => {
|
|
|
|
|
const port = randomPort();
|
|
|
|
|
// Start daemon
|
|
|
|
|
const { exitCode: startCode } = await runDaemonQmd([
|
|
|
|
|
"mcp", "--http", "--daemon", "--port", String(port),
|
|
|
|
|
]);
|
|
|
|
|
expect(startCode).toBe(0);
|
|
|
|
|
|
|
|
|
|
const pid = parseInt(readFileSync(pidPath(), "utf-8").trim());
|
|
|
|
|
spawnedPids.push(pid);
|
|
|
|
|
|
|
|
|
|
await waitForServer(port);
|
|
|
|
|
|
|
|
|
|
// Stop it
|
|
|
|
|
const { stdout: stopOut, exitCode: stopCode } = await runDaemonQmd(["mcp", "stop"]);
|
|
|
|
|
expect(stopCode).toBe(0);
|
|
|
|
|
expect(stopOut).toContain("Stopped");
|
|
|
|
|
|
|
|
|
|
// PID file should be gone
|
2026-02-21 09:52:53 +08:00
|
|
|
expect(existsSync(pidPath())).toBe(false);
|
MCP: Streamable HTTP, scoring fixes, tool improvements (#149)
* feat: MCP HTTP transport with daemon lifecycle
Add streaming HTTP transport as an alternative to stdio for the MCP
server. A long-lived HTTP server avoids reloading 3 GGUF models (~2GB)
on every client connection, reducing warm query latency from ~16s (CLI)
to ~10s.
New CLI surface:
qmd mcp --http [--port N] # foreground, default port 3000
qmd mcp --http --daemon # background, PID in ~/.cache/qmd/mcp.pid
qmd mcp stop # stop daemon via PID file
qmd status # now shows MCP daemon liveness
Server implementation (mcp.ts):
- Extract createMcpServer(store) shared by stdio and HTTP transports
- HTTP transport uses WebStandardStreamableHTTPServerTransport with
JSON responses (stateless, no SSE)
- /health endpoint with uptime, /mcp for MCP protocol, 404 otherwise
- Request logging to stderr with timestamps, tool names, query args
Daemon lifecycle (qmd.ts):
- PID file + log file management with stale PID detection
- Absolute paths in Bun.spawn (process.execPath + import.meta.path)
so daemon works regardless of cwd
- mkdirSync for cache dir on fresh installs
- Removes top-level SIGTERM/SIGINT handlers before starting HTTP
server so async cleanup in mcp.ts actually runs
Move hybridQuery() and vectorSearchQuery() into store.ts as standalone
functions that take a Store as first argument. Both CLI and MCP now
call the identical pipeline, eliminating the class of bugs where one
copy drifts from the other.
Shared pipeline (store.ts):
- hybridQuery(): BM25 probe → expand → FTS+vec search → RRF →
chunk → rerank (chunks only) → position-aware blending → dedup
- vectorSearchQuery(): expand → vec search → dedup → sort
- SearchHooks interface for optional progress callbacks
- Constants: STRONG_SIGNAL_MIN_SCORE, STRONG_SIGNAL_MIN_GAP,
RERANK_CANDIDATE_LIMIT (40), addLineNumbers()
Bugs fixed by unification:
- MCP now gets strong-signal short-circuit (was CLI-only)
- Reranker candidate limit unified at 40 (MCP had 30)
- File dedup added to hybrid query (MCP was missing it)
- Collection filter pushed into searchVec DB query
- Filter-then-slice ordering fixed (MCP was slice-then-filter)
* feat: type-routed query expansion — lex→FTS, vec/hyde→vector
expandQuery() now returns typed ExpandedQuery[] instead of string[],
preserving the lex/vec/hyde type info from the LLM's GBNF-structured
output. hybridQuery() and vectorSearchQuery() route searches by type:
lex queries go to FTS only, vec/hyde go to vector only.
Previously, every expanded query ran through BOTH backends — keyword
variants wasted embedding forward passes, semantic paraphrases wasted
BM25 lookups. Type routing eliminates ~4 calls/query with zero quality
loss (cross-backend noise actually hurt RRF fusion).
Cache format changed from newline-separated text to JSON (preserves
types). Old cache entries gracefully re-expand on first access.
CLI expansion tree now shows query types:
├─ original query
├─ lex: keyword variant
├─ vec: semantic meaning
└─ hyde: hypothetical document...
Benchmark (5 queries, 1756-doc index, warm LLM, Apple Silicon):
Metric Old (untyped) New (typed) Delta
Avg backend calls 10.0 6.0 -40%
Total wall time 1278ms 549ms -57%
Avg saved/query — — 146ms
"authentication setup" 12 → 7 calls 511 → 112ms
"database migration strategy" 10 → 6 calls 182 → 106ms
"how to handle errors in API" 10 → 6 calls 216 → 121ms
"meeting notes from last week" 10 → 6 calls 228 → 110ms
"performance optimization" 8 → 5 calls 141 → 100ms
Savings come from skipped embed() calls (~30-80ms each). FTS is
synchronous SQLite (~0ms), so lex→FTS routing is free while
vec/hyde→vector-only avoids wasted embedding passes.
* fix: MCP query snippets now use reranker's best chunk, not full body
extractSnippet() was scanning the entire document body for keyword
matches to build the snippet. But hybridQuery() already identified
the most relevant chunk via cross-attention reranking — rescanning
the full body is redundant and can land on a less relevant section
if the query terms appear elsewhere in the document.
CLI was already using bestChunk (set during the refactor). MCP was
still using body — a pre-existing inconsistency, not a regression.
* feat: dynamic MCP instructions + tool annotations
The MCP server now generates instructions at startup from actual index
state and injects them into the initialize response. LLMs see collection
names, document counts, content descriptions, and search strategy
guidance in their system prompt — zero tool calls needed for orientation.
Previously, the only guidance was generic static tool descriptions and
a user-invocable "query" prompt that no LLM would discover on its own.
An LLM connecting to QMD had no idea what collections existed, what they
contained, or how to scope searches effectively.
* change default port to 8181
* fix: BM25 score normalization was inverted
The normalization formula `1 / (1 + |bm25|)` is a decreasing function of
match strength. FTS5 BM25 scores are negative where more negative = better
match (e.g., -10 is strong, -0.5 is weak). The formula mapped:
strong match (raw -10) → 1/(1+10) = 9% ← should be highest
weak match (raw -0.5) → 1/(1+0.5) = 67% ← should be lowest
Three downstream effects:
1. `--min-score 0.5` (or MCP minScore: 0.5) filtered OUT strong matches
and kept only weak ones. The MCP instructions recommend this threshold.
2. CLI `formatScore()` color bands never showed green for BM25 results
(best matches scored ~9%, green threshold is 70%).
3. The strong signal optimization in hybridQuery (skip ~2s LLM expansion
when BM25 already has a clear winner) was dead code — strong matches
scored ~0.09, never reaching the 0.85 threshold.
Fix: `|x| / (1 + |x|)` — same (0,1) range, monotonic, no per-query
normalization needed, but now correctly maps strong → high, weak → low.
The normalization was born broken (Math.max(0, x) clamped all
negative BM25 to 0 → every score = 1.0), then PR #76 changed to
Math.abs which made scores vary but inverted the direction. Neither
state was ever correct.
* fix: rerank cache key ignores chunk content
The rerank cache key was (query, file, model) but the actual text sent
to the reranker is a keyword-selected chunk that varies by query terms.
Two different queries hitting the same file can select different chunks,
but the second query gets a stale cached score from the first chunk.
Example:
Query "auth flow" → selects chunk about authentication → score 0.92
Query "auth tokens" → same file, selects chunk about tokens
→ cache HIT on (query, file, model) → returns 0.92 from wrong chunk
Fix: include full chunk text in cache key. getCacheKey() already
SHA-256 hashes its inputs, so this adds no key bloat — just
disambiguation. Old cache entries become natural misses (different key
shape) and re-warm on next query.
* rename MCP tools for clarity, rewrite descriptions for LLM tool selection
Rename MCP tools: vsearch → vector_search, query → deep_search.
LLMs see these names — self-documenting names reduce reliance on
descriptions for tool selection. CLI commands stay unchanged
(qmd vsearch, qmd query) — different namespace, users type those.
Rewrite all search tool descriptions to be action-oriented:
- search: "Search by keyword. Finds documents containing exact
words and phrases in the query."
- vector_search: "Search by meaning. Finds relevant documents even
when they use different words than the query — handles synonyms,
paraphrases, and related concepts."
- deep_search: "Deep search. Auto-expands the query into variations,
searches each by keyword and meaning, and reranks for top hits
across all results."
Rewrite instructions ladder — each tool says what it does, no
"start here" / "escalate as needed" strategy language.
Delete the "query" prompt (registerPrompt) — it restated what
descriptions + instructions already cover. No LLM proactively
calls prompts/get to learn how to use tools.
* supress HTTP server logs during tests
2026-02-11 05:37:33 +08:00
|
|
|
|
|
|
|
|
// Process should be dead
|
2026-02-16 04:57:13 +08:00
|
|
|
await sleep(500);
|
MCP: Streamable HTTP, scoring fixes, tool improvements (#149)
* feat: MCP HTTP transport with daemon lifecycle
Add streaming HTTP transport as an alternative to stdio for the MCP
server. A long-lived HTTP server avoids reloading 3 GGUF models (~2GB)
on every client connection, reducing warm query latency from ~16s (CLI)
to ~10s.
New CLI surface:
qmd mcp --http [--port N] # foreground, default port 3000
qmd mcp --http --daemon # background, PID in ~/.cache/qmd/mcp.pid
qmd mcp stop # stop daemon via PID file
qmd status # now shows MCP daemon liveness
Server implementation (mcp.ts):
- Extract createMcpServer(store) shared by stdio and HTTP transports
- HTTP transport uses WebStandardStreamableHTTPServerTransport with
JSON responses (stateless, no SSE)
- /health endpoint with uptime, /mcp for MCP protocol, 404 otherwise
- Request logging to stderr with timestamps, tool names, query args
Daemon lifecycle (qmd.ts):
- PID file + log file management with stale PID detection
- Absolute paths in Bun.spawn (process.execPath + import.meta.path)
so daemon works regardless of cwd
- mkdirSync for cache dir on fresh installs
- Removes top-level SIGTERM/SIGINT handlers before starting HTTP
server so async cleanup in mcp.ts actually runs
Move hybridQuery() and vectorSearchQuery() into store.ts as standalone
functions that take a Store as first argument. Both CLI and MCP now
call the identical pipeline, eliminating the class of bugs where one
copy drifts from the other.
Shared pipeline (store.ts):
- hybridQuery(): BM25 probe → expand → FTS+vec search → RRF →
chunk → rerank (chunks only) → position-aware blending → dedup
- vectorSearchQuery(): expand → vec search → dedup → sort
- SearchHooks interface for optional progress callbacks
- Constants: STRONG_SIGNAL_MIN_SCORE, STRONG_SIGNAL_MIN_GAP,
RERANK_CANDIDATE_LIMIT (40), addLineNumbers()
Bugs fixed by unification:
- MCP now gets strong-signal short-circuit (was CLI-only)
- Reranker candidate limit unified at 40 (MCP had 30)
- File dedup added to hybrid query (MCP was missing it)
- Collection filter pushed into searchVec DB query
- Filter-then-slice ordering fixed (MCP was slice-then-filter)
* feat: type-routed query expansion — lex→FTS, vec/hyde→vector
expandQuery() now returns typed ExpandedQuery[] instead of string[],
preserving the lex/vec/hyde type info from the LLM's GBNF-structured
output. hybridQuery() and vectorSearchQuery() route searches by type:
lex queries go to FTS only, vec/hyde go to vector only.
Previously, every expanded query ran through BOTH backends — keyword
variants wasted embedding forward passes, semantic paraphrases wasted
BM25 lookups. Type routing eliminates ~4 calls/query with zero quality
loss (cross-backend noise actually hurt RRF fusion).
Cache format changed from newline-separated text to JSON (preserves
types). Old cache entries gracefully re-expand on first access.
CLI expansion tree now shows query types:
├─ original query
├─ lex: keyword variant
├─ vec: semantic meaning
└─ hyde: hypothetical document...
Benchmark (5 queries, 1756-doc index, warm LLM, Apple Silicon):
Metric Old (untyped) New (typed) Delta
Avg backend calls 10.0 6.0 -40%
Total wall time 1278ms 549ms -57%
Avg saved/query — — 146ms
"authentication setup" 12 → 7 calls 511 → 112ms
"database migration strategy" 10 → 6 calls 182 → 106ms
"how to handle errors in API" 10 → 6 calls 216 → 121ms
"meeting notes from last week" 10 → 6 calls 228 → 110ms
"performance optimization" 8 → 5 calls 141 → 100ms
Savings come from skipped embed() calls (~30-80ms each). FTS is
synchronous SQLite (~0ms), so lex→FTS routing is free while
vec/hyde→vector-only avoids wasted embedding passes.
* fix: MCP query snippets now use reranker's best chunk, not full body
extractSnippet() was scanning the entire document body for keyword
matches to build the snippet. But hybridQuery() already identified
the most relevant chunk via cross-attention reranking — rescanning
the full body is redundant and can land on a less relevant section
if the query terms appear elsewhere in the document.
CLI was already using bestChunk (set during the refactor). MCP was
still using body — a pre-existing inconsistency, not a regression.
* feat: dynamic MCP instructions + tool annotations
The MCP server now generates instructions at startup from actual index
state and injects them into the initialize response. LLMs see collection
names, document counts, content descriptions, and search strategy
guidance in their system prompt — zero tool calls needed for orientation.
Previously, the only guidance was generic static tool descriptions and
a user-invocable "query" prompt that no LLM would discover on its own.
An LLM connecting to QMD had no idea what collections existed, what they
contained, or how to scope searches effectively.
* change default port to 8181
* fix: BM25 score normalization was inverted
The normalization formula `1 / (1 + |bm25|)` is a decreasing function of
match strength. FTS5 BM25 scores are negative where more negative = better
match (e.g., -10 is strong, -0.5 is weak). The formula mapped:
strong match (raw -10) → 1/(1+10) = 9% ← should be highest
weak match (raw -0.5) → 1/(1+0.5) = 67% ← should be lowest
Three downstream effects:
1. `--min-score 0.5` (or MCP minScore: 0.5) filtered OUT strong matches
and kept only weak ones. The MCP instructions recommend this threshold.
2. CLI `formatScore()` color bands never showed green for BM25 results
(best matches scored ~9%, green threshold is 70%).
3. The strong signal optimization in hybridQuery (skip ~2s LLM expansion
when BM25 already has a clear winner) was dead code — strong matches
scored ~0.09, never reaching the 0.85 threshold.
Fix: `|x| / (1 + |x|)` — same (0,1) range, monotonic, no per-query
normalization needed, but now correctly maps strong → high, weak → low.
The normalization was born broken (Math.max(0, x) clamped all
negative BM25 to 0 → every score = 1.0), then PR #76 changed to
Math.abs which made scores vary but inverted the direction. Neither
state was ever correct.
* fix: rerank cache key ignores chunk content
The rerank cache key was (query, file, model) but the actual text sent
to the reranker is a keyword-selected chunk that varies by query terms.
Two different queries hitting the same file can select different chunks,
but the second query gets a stale cached score from the first chunk.
Example:
Query "auth flow" → selects chunk about authentication → score 0.92
Query "auth tokens" → same file, selects chunk about tokens
→ cache HIT on (query, file, model) → returns 0.92 from wrong chunk
Fix: include full chunk text in cache key. getCacheKey() already
SHA-256 hashes its inputs, so this adds no key bloat — just
disambiguation. Old cache entries become natural misses (different key
shape) and re-warm on next query.
* rename MCP tools for clarity, rewrite descriptions for LLM tool selection
Rename MCP tools: vsearch → vector_search, query → deep_search.
LLMs see these names — self-documenting names reduce reliance on
descriptions for tool selection. CLI commands stay unchanged
(qmd vsearch, qmd query) — different namespace, users type those.
Rewrite all search tool descriptions to be action-oriented:
- search: "Search by keyword. Finds documents containing exact
words and phrases in the query."
- vector_search: "Search by meaning. Finds relevant documents even
when they use different words than the query — handles synonyms,
paraphrases, and related concepts."
- deep_search: "Deep search. Auto-expands the query into variations,
searches each by keyword and meaning, and reranks for top hits
across all results."
Rewrite instructions ladder — each tool says what it does, no
"start here" / "escalate as needed" strategy language.
Delete the "query" prompt (registerPrompt) — it restated what
descriptions + instructions already cover. No LLM proactively
calls prompts/get to learn how to use tools.
* supress HTTP server logs during tests
2026-02-11 05:37:33 +08:00
|
|
|
expect(() => process.kill(pid, 0)).toThrow();
|
|
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("stop handles dead PID gracefully (cleans stale file)", async () => {
|
|
|
|
|
// Write a PID file pointing to a dead process
|
|
|
|
|
writeFileSync(pidPath(), "999999999");
|
|
|
|
|
|
|
|
|
|
const { stdout, exitCode } = await runDaemonQmd(["mcp", "stop"]);
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain("stale");
|
|
|
|
|
|
|
|
|
|
// PID file should be cleaned up
|
2026-02-21 09:52:53 +08:00
|
|
|
expect(existsSync(pidPath())).toBe(false);
|
MCP: Streamable HTTP, scoring fixes, tool improvements (#149)
* feat: MCP HTTP transport with daemon lifecycle
Add streaming HTTP transport as an alternative to stdio for the MCP
server. A long-lived HTTP server avoids reloading 3 GGUF models (~2GB)
on every client connection, reducing warm query latency from ~16s (CLI)
to ~10s.
New CLI surface:
qmd mcp --http [--port N] # foreground, default port 3000
qmd mcp --http --daemon # background, PID in ~/.cache/qmd/mcp.pid
qmd mcp stop # stop daemon via PID file
qmd status # now shows MCP daemon liveness
Server implementation (mcp.ts):
- Extract createMcpServer(store) shared by stdio and HTTP transports
- HTTP transport uses WebStandardStreamableHTTPServerTransport with
JSON responses (stateless, no SSE)
- /health endpoint with uptime, /mcp for MCP protocol, 404 otherwise
- Request logging to stderr with timestamps, tool names, query args
Daemon lifecycle (qmd.ts):
- PID file + log file management with stale PID detection
- Absolute paths in Bun.spawn (process.execPath + import.meta.path)
so daemon works regardless of cwd
- mkdirSync for cache dir on fresh installs
- Removes top-level SIGTERM/SIGINT handlers before starting HTTP
server so async cleanup in mcp.ts actually runs
Move hybridQuery() and vectorSearchQuery() into store.ts as standalone
functions that take a Store as first argument. Both CLI and MCP now
call the identical pipeline, eliminating the class of bugs where one
copy drifts from the other.
Shared pipeline (store.ts):
- hybridQuery(): BM25 probe → expand → FTS+vec search → RRF →
chunk → rerank (chunks only) → position-aware blending → dedup
- vectorSearchQuery(): expand → vec search → dedup → sort
- SearchHooks interface for optional progress callbacks
- Constants: STRONG_SIGNAL_MIN_SCORE, STRONG_SIGNAL_MIN_GAP,
RERANK_CANDIDATE_LIMIT (40), addLineNumbers()
Bugs fixed by unification:
- MCP now gets strong-signal short-circuit (was CLI-only)
- Reranker candidate limit unified at 40 (MCP had 30)
- File dedup added to hybrid query (MCP was missing it)
- Collection filter pushed into searchVec DB query
- Filter-then-slice ordering fixed (MCP was slice-then-filter)
* feat: type-routed query expansion — lex→FTS, vec/hyde→vector
expandQuery() now returns typed ExpandedQuery[] instead of string[],
preserving the lex/vec/hyde type info from the LLM's GBNF-structured
output. hybridQuery() and vectorSearchQuery() route searches by type:
lex queries go to FTS only, vec/hyde go to vector only.
Previously, every expanded query ran through BOTH backends — keyword
variants wasted embedding forward passes, semantic paraphrases wasted
BM25 lookups. Type routing eliminates ~4 calls/query with zero quality
loss (cross-backend noise actually hurt RRF fusion).
Cache format changed from newline-separated text to JSON (preserves
types). Old cache entries gracefully re-expand on first access.
CLI expansion tree now shows query types:
├─ original query
├─ lex: keyword variant
├─ vec: semantic meaning
└─ hyde: hypothetical document...
Benchmark (5 queries, 1756-doc index, warm LLM, Apple Silicon):
Metric Old (untyped) New (typed) Delta
Avg backend calls 10.0 6.0 -40%
Total wall time 1278ms 549ms -57%
Avg saved/query — — 146ms
"authentication setup" 12 → 7 calls 511 → 112ms
"database migration strategy" 10 → 6 calls 182 → 106ms
"how to handle errors in API" 10 → 6 calls 216 → 121ms
"meeting notes from last week" 10 → 6 calls 228 → 110ms
"performance optimization" 8 → 5 calls 141 → 100ms
Savings come from skipped embed() calls (~30-80ms each). FTS is
synchronous SQLite (~0ms), so lex→FTS routing is free while
vec/hyde→vector-only avoids wasted embedding passes.
* fix: MCP query snippets now use reranker's best chunk, not full body
extractSnippet() was scanning the entire document body for keyword
matches to build the snippet. But hybridQuery() already identified
the most relevant chunk via cross-attention reranking — rescanning
the full body is redundant and can land on a less relevant section
if the query terms appear elsewhere in the document.
CLI was already using bestChunk (set during the refactor). MCP was
still using body — a pre-existing inconsistency, not a regression.
* feat: dynamic MCP instructions + tool annotations
The MCP server now generates instructions at startup from actual index
state and injects them into the initialize response. LLMs see collection
names, document counts, content descriptions, and search strategy
guidance in their system prompt — zero tool calls needed for orientation.
Previously, the only guidance was generic static tool descriptions and
a user-invocable "query" prompt that no LLM would discover on its own.
An LLM connecting to QMD had no idea what collections existed, what they
contained, or how to scope searches effectively.
* change default port to 8181
* fix: BM25 score normalization was inverted
The normalization formula `1 / (1 + |bm25|)` is a decreasing function of
match strength. FTS5 BM25 scores are negative where more negative = better
match (e.g., -10 is strong, -0.5 is weak). The formula mapped:
strong match (raw -10) → 1/(1+10) = 9% ← should be highest
weak match (raw -0.5) → 1/(1+0.5) = 67% ← should be lowest
Three downstream effects:
1. `--min-score 0.5` (or MCP minScore: 0.5) filtered OUT strong matches
and kept only weak ones. The MCP instructions recommend this threshold.
2. CLI `formatScore()` color bands never showed green for BM25 results
(best matches scored ~9%, green threshold is 70%).
3. The strong signal optimization in hybridQuery (skip ~2s LLM expansion
when BM25 already has a clear winner) was dead code — strong matches
scored ~0.09, never reaching the 0.85 threshold.
Fix: `|x| / (1 + |x|)` — same (0,1) range, monotonic, no per-query
normalization needed, but now correctly maps strong → high, weak → low.
The normalization was born broken (Math.max(0, x) clamped all
negative BM25 to 0 → every score = 1.0), then PR #76 changed to
Math.abs which made scores vary but inverted the direction. Neither
state was ever correct.
* fix: rerank cache key ignores chunk content
The rerank cache key was (query, file, model) but the actual text sent
to the reranker is a keyword-selected chunk that varies by query terms.
Two different queries hitting the same file can select different chunks,
but the second query gets a stale cached score from the first chunk.
Example:
Query "auth flow" → selects chunk about authentication → score 0.92
Query "auth tokens" → same file, selects chunk about tokens
→ cache HIT on (query, file, model) → returns 0.92 from wrong chunk
Fix: include full chunk text in cache key. getCacheKey() already
SHA-256 hashes its inputs, so this adds no key bloat — just
disambiguation. Old cache entries become natural misses (different key
shape) and re-warm on next query.
* rename MCP tools for clarity, rewrite descriptions for LLM tool selection
Rename MCP tools: vsearch → vector_search, query → deep_search.
LLMs see these names — self-documenting names reduce reliance on
descriptions for tool selection. CLI commands stay unchanged
(qmd vsearch, qmd query) — different namespace, users type those.
Rewrite all search tool descriptions to be action-oriented:
- search: "Search by keyword. Finds documents containing exact
words and phrases in the query."
- vector_search: "Search by meaning. Finds relevant documents even
when they use different words than the query — handles synonyms,
paraphrases, and related concepts."
- deep_search: "Deep search. Auto-expands the query into variations,
searches each by keyword and meaning, and reranks for top hits
across all results."
Rewrite instructions ladder — each tool says what it does, no
"start here" / "escalate as needed" strategy language.
Delete the "query" prompt (registerPrompt) — it restated what
descriptions + instructions already cover. No LLM proactively
calls prompts/get to learn how to use tools.
* supress HTTP server logs during tests
2026-02-11 05:37:33 +08:00
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("--daemon rejects if already running", async () => {
|
|
|
|
|
const port = randomPort();
|
|
|
|
|
// Start first daemon
|
|
|
|
|
const { exitCode: firstCode } = await runDaemonQmd([
|
|
|
|
|
"mcp", "--http", "--daemon", "--port", String(port),
|
|
|
|
|
]);
|
|
|
|
|
expect(firstCode).toBe(0);
|
|
|
|
|
|
|
|
|
|
const pid = parseInt(readFileSync(pidPath(), "utf-8").trim());
|
|
|
|
|
spawnedPids.push(pid);
|
|
|
|
|
|
|
|
|
|
await waitForServer(port);
|
|
|
|
|
|
|
|
|
|
// Try to start second daemon — should fail
|
|
|
|
|
const { stderr, exitCode } = await runDaemonQmd([
|
|
|
|
|
"mcp", "--http", "--daemon", "--port", String(port + 1),
|
|
|
|
|
]);
|
|
|
|
|
expect(exitCode).toBe(1);
|
|
|
|
|
expect(stderr).toContain("Already running");
|
|
|
|
|
|
|
|
|
|
// Clean up first daemon
|
|
|
|
|
process.kill(pid, "SIGTERM");
|
2026-02-16 04:57:13 +08:00
|
|
|
await sleep(500);
|
2026-02-21 09:52:53 +08:00
|
|
|
try { unlinkSync(pidPath()); } catch {}
|
MCP: Streamable HTTP, scoring fixes, tool improvements (#149)
* feat: MCP HTTP transport with daemon lifecycle
Add streaming HTTP transport as an alternative to stdio for the MCP
server. A long-lived HTTP server avoids reloading 3 GGUF models (~2GB)
on every client connection, reducing warm query latency from ~16s (CLI)
to ~10s.
New CLI surface:
qmd mcp --http [--port N] # foreground, default port 3000
qmd mcp --http --daemon # background, PID in ~/.cache/qmd/mcp.pid
qmd mcp stop # stop daemon via PID file
qmd status # now shows MCP daemon liveness
Server implementation (mcp.ts):
- Extract createMcpServer(store) shared by stdio and HTTP transports
- HTTP transport uses WebStandardStreamableHTTPServerTransport with
JSON responses (stateless, no SSE)
- /health endpoint with uptime, /mcp for MCP protocol, 404 otherwise
- Request logging to stderr with timestamps, tool names, query args
Daemon lifecycle (qmd.ts):
- PID file + log file management with stale PID detection
- Absolute paths in Bun.spawn (process.execPath + import.meta.path)
so daemon works regardless of cwd
- mkdirSync for cache dir on fresh installs
- Removes top-level SIGTERM/SIGINT handlers before starting HTTP
server so async cleanup in mcp.ts actually runs
Move hybridQuery() and vectorSearchQuery() into store.ts as standalone
functions that take a Store as first argument. Both CLI and MCP now
call the identical pipeline, eliminating the class of bugs where one
copy drifts from the other.
Shared pipeline (store.ts):
- hybridQuery(): BM25 probe → expand → FTS+vec search → RRF →
chunk → rerank (chunks only) → position-aware blending → dedup
- vectorSearchQuery(): expand → vec search → dedup → sort
- SearchHooks interface for optional progress callbacks
- Constants: STRONG_SIGNAL_MIN_SCORE, STRONG_SIGNAL_MIN_GAP,
RERANK_CANDIDATE_LIMIT (40), addLineNumbers()
Bugs fixed by unification:
- MCP now gets strong-signal short-circuit (was CLI-only)
- Reranker candidate limit unified at 40 (MCP had 30)
- File dedup added to hybrid query (MCP was missing it)
- Collection filter pushed into searchVec DB query
- Filter-then-slice ordering fixed (MCP was slice-then-filter)
* feat: type-routed query expansion — lex→FTS, vec/hyde→vector
expandQuery() now returns typed ExpandedQuery[] instead of string[],
preserving the lex/vec/hyde type info from the LLM's GBNF-structured
output. hybridQuery() and vectorSearchQuery() route searches by type:
lex queries go to FTS only, vec/hyde go to vector only.
Previously, every expanded query ran through BOTH backends — keyword
variants wasted embedding forward passes, semantic paraphrases wasted
BM25 lookups. Type routing eliminates ~4 calls/query with zero quality
loss (cross-backend noise actually hurt RRF fusion).
Cache format changed from newline-separated text to JSON (preserves
types). Old cache entries gracefully re-expand on first access.
CLI expansion tree now shows query types:
├─ original query
├─ lex: keyword variant
├─ vec: semantic meaning
└─ hyde: hypothetical document...
Benchmark (5 queries, 1756-doc index, warm LLM, Apple Silicon):
Metric Old (untyped) New (typed) Delta
Avg backend calls 10.0 6.0 -40%
Total wall time 1278ms 549ms -57%
Avg saved/query — — 146ms
"authentication setup" 12 → 7 calls 511 → 112ms
"database migration strategy" 10 → 6 calls 182 → 106ms
"how to handle errors in API" 10 → 6 calls 216 → 121ms
"meeting notes from last week" 10 → 6 calls 228 → 110ms
"performance optimization" 8 → 5 calls 141 → 100ms
Savings come from skipped embed() calls (~30-80ms each). FTS is
synchronous SQLite (~0ms), so lex→FTS routing is free while
vec/hyde→vector-only avoids wasted embedding passes.
* fix: MCP query snippets now use reranker's best chunk, not full body
extractSnippet() was scanning the entire document body for keyword
matches to build the snippet. But hybridQuery() already identified
the most relevant chunk via cross-attention reranking — rescanning
the full body is redundant and can land on a less relevant section
if the query terms appear elsewhere in the document.
CLI was already using bestChunk (set during the refactor). MCP was
still using body — a pre-existing inconsistency, not a regression.
* feat: dynamic MCP instructions + tool annotations
The MCP server now generates instructions at startup from actual index
state and injects them into the initialize response. LLMs see collection
names, document counts, content descriptions, and search strategy
guidance in their system prompt — zero tool calls needed for orientation.
Previously, the only guidance was generic static tool descriptions and
a user-invocable "query" prompt that no LLM would discover on its own.
An LLM connecting to QMD had no idea what collections existed, what they
contained, or how to scope searches effectively.
* change default port to 8181
* fix: BM25 score normalization was inverted
The normalization formula `1 / (1 + |bm25|)` is a decreasing function of
match strength. FTS5 BM25 scores are negative where more negative = better
match (e.g., -10 is strong, -0.5 is weak). The formula mapped:
strong match (raw -10) → 1/(1+10) = 9% ← should be highest
weak match (raw -0.5) → 1/(1+0.5) = 67% ← should be lowest
Three downstream effects:
1. `--min-score 0.5` (or MCP minScore: 0.5) filtered OUT strong matches
and kept only weak ones. The MCP instructions recommend this threshold.
2. CLI `formatScore()` color bands never showed green for BM25 results
(best matches scored ~9%, green threshold is 70%).
3. The strong signal optimization in hybridQuery (skip ~2s LLM expansion
when BM25 already has a clear winner) was dead code — strong matches
scored ~0.09, never reaching the 0.85 threshold.
Fix: `|x| / (1 + |x|)` — same (0,1) range, monotonic, no per-query
normalization needed, but now correctly maps strong → high, weak → low.
The normalization was born broken (Math.max(0, x) clamped all
negative BM25 to 0 → every score = 1.0), then PR #76 changed to
Math.abs which made scores vary but inverted the direction. Neither
state was ever correct.
* fix: rerank cache key ignores chunk content
The rerank cache key was (query, file, model) but the actual text sent
to the reranker is a keyword-selected chunk that varies by query terms.
Two different queries hitting the same file can select different chunks,
but the second query gets a stale cached score from the first chunk.
Example:
Query "auth flow" → selects chunk about authentication → score 0.92
Query "auth tokens" → same file, selects chunk about tokens
→ cache HIT on (query, file, model) → returns 0.92 from wrong chunk
Fix: include full chunk text in cache key. getCacheKey() already
SHA-256 hashes its inputs, so this adds no key bloat — just
disambiguation. Old cache entries become natural misses (different key
shape) and re-warm on next query.
* rename MCP tools for clarity, rewrite descriptions for LLM tool selection
Rename MCP tools: vsearch → vector_search, query → deep_search.
LLMs see these names — self-documenting names reduce reliance on
descriptions for tool selection. CLI commands stay unchanged
(qmd vsearch, qmd query) — different namespace, users type those.
Rewrite all search tool descriptions to be action-oriented:
- search: "Search by keyword. Finds documents containing exact
words and phrases in the query."
- vector_search: "Search by meaning. Finds relevant documents even
when they use different words than the query — handles synonyms,
paraphrases, and related concepts."
- deep_search: "Deep search. Auto-expands the query into variations,
searches each by keyword and meaning, and reranks for top hits
across all results."
Rewrite instructions ladder — each tool says what it does, no
"start here" / "escalate as needed" strategy language.
Delete the "query" prompt (registerPrompt) — it restated what
descriptions + instructions already cover. No LLM proactively
calls prompts/get to learn how to use tools.
* supress HTTP server logs during tests
2026-02-11 05:37:33 +08:00
|
|
|
});
|
|
|
|
|
|
|
|
|
|
test("--daemon cleans stale PID file and starts fresh", async () => {
|
|
|
|
|
// Write a stale PID file
|
|
|
|
|
writeFileSync(pidPath(), "999999999");
|
|
|
|
|
|
|
|
|
|
const port = randomPort();
|
|
|
|
|
const { exitCode, stdout } = await runDaemonQmd([
|
|
|
|
|
"mcp", "--http", "--daemon", "--port", String(port),
|
|
|
|
|
]);
|
|
|
|
|
expect(exitCode).toBe(0);
|
|
|
|
|
expect(stdout).toContain(`http://localhost:${port}/mcp`);
|
|
|
|
|
|
|
|
|
|
const pid = parseInt(readFileSync(pidPath(), "utf-8").trim());
|
|
|
|
|
spawnedPids.push(pid);
|
|
|
|
|
expect(pid).not.toBe(999999999);
|
|
|
|
|
|
|
|
|
|
// Clean up
|
|
|
|
|
const ready = await waitForServer(port);
|
|
|
|
|
expect(ready).toBe(true);
|
|
|
|
|
process.kill(pid, "SIGTERM");
|
2026-02-16 04:57:13 +08:00
|
|
|
await sleep(500);
|
2026-02-21 09:52:53 +08:00
|
|
|
try { unlinkSync(pidPath()); } catch {}
|
MCP: Streamable HTTP, scoring fixes, tool improvements (#149)
* feat: MCP HTTP transport with daemon lifecycle
Add streaming HTTP transport as an alternative to stdio for the MCP
server. A long-lived HTTP server avoids reloading 3 GGUF models (~2GB)
on every client connection, reducing warm query latency from ~16s (CLI)
to ~10s.
New CLI surface:
qmd mcp --http [--port N] # foreground, default port 3000
qmd mcp --http --daemon # background, PID in ~/.cache/qmd/mcp.pid
qmd mcp stop # stop daemon via PID file
qmd status # now shows MCP daemon liveness
Server implementation (mcp.ts):
- Extract createMcpServer(store) shared by stdio and HTTP transports
- HTTP transport uses WebStandardStreamableHTTPServerTransport with
JSON responses (stateless, no SSE)
- /health endpoint with uptime, /mcp for MCP protocol, 404 otherwise
- Request logging to stderr with timestamps, tool names, query args
Daemon lifecycle (qmd.ts):
- PID file + log file management with stale PID detection
- Absolute paths in Bun.spawn (process.execPath + import.meta.path)
so daemon works regardless of cwd
- mkdirSync for cache dir on fresh installs
- Removes top-level SIGTERM/SIGINT handlers before starting HTTP
server so async cleanup in mcp.ts actually runs
Move hybridQuery() and vectorSearchQuery() into store.ts as standalone
functions that take a Store as first argument. Both CLI and MCP now
call the identical pipeline, eliminating the class of bugs where one
copy drifts from the other.
Shared pipeline (store.ts):
- hybridQuery(): BM25 probe → expand → FTS+vec search → RRF →
chunk → rerank (chunks only) → position-aware blending → dedup
- vectorSearchQuery(): expand → vec search → dedup → sort
- SearchHooks interface for optional progress callbacks
- Constants: STRONG_SIGNAL_MIN_SCORE, STRONG_SIGNAL_MIN_GAP,
RERANK_CANDIDATE_LIMIT (40), addLineNumbers()
Bugs fixed by unification:
- MCP now gets strong-signal short-circuit (was CLI-only)
- Reranker candidate limit unified at 40 (MCP had 30)
- File dedup added to hybrid query (MCP was missing it)
- Collection filter pushed into searchVec DB query
- Filter-then-slice ordering fixed (MCP was slice-then-filter)
* feat: type-routed query expansion — lex→FTS, vec/hyde→vector
expandQuery() now returns typed ExpandedQuery[] instead of string[],
preserving the lex/vec/hyde type info from the LLM's GBNF-structured
output. hybridQuery() and vectorSearchQuery() route searches by type:
lex queries go to FTS only, vec/hyde go to vector only.
Previously, every expanded query ran through BOTH backends — keyword
variants wasted embedding forward passes, semantic paraphrases wasted
BM25 lookups. Type routing eliminates ~4 calls/query with zero quality
loss (cross-backend noise actually hurt RRF fusion).
Cache format changed from newline-separated text to JSON (preserves
types). Old cache entries gracefully re-expand on first access.
CLI expansion tree now shows query types:
├─ original query
├─ lex: keyword variant
├─ vec: semantic meaning
└─ hyde: hypothetical document...
Benchmark (5 queries, 1756-doc index, warm LLM, Apple Silicon):
Metric Old (untyped) New (typed) Delta
Avg backend calls 10.0 6.0 -40%
Total wall time 1278ms 549ms -57%
Avg saved/query — — 146ms
"authentication setup" 12 → 7 calls 511 → 112ms
"database migration strategy" 10 → 6 calls 182 → 106ms
"how to handle errors in API" 10 → 6 calls 216 → 121ms
"meeting notes from last week" 10 → 6 calls 228 → 110ms
"performance optimization" 8 → 5 calls 141 → 100ms
Savings come from skipped embed() calls (~30-80ms each). FTS is
synchronous SQLite (~0ms), so lex→FTS routing is free while
vec/hyde→vector-only avoids wasted embedding passes.
* fix: MCP query snippets now use reranker's best chunk, not full body
extractSnippet() was scanning the entire document body for keyword
matches to build the snippet. But hybridQuery() already identified
the most relevant chunk via cross-attention reranking — rescanning
the full body is redundant and can land on a less relevant section
if the query terms appear elsewhere in the document.
CLI was already using bestChunk (set during the refactor). MCP was
still using body — a pre-existing inconsistency, not a regression.
* feat: dynamic MCP instructions + tool annotations
The MCP server now generates instructions at startup from actual index
state and injects them into the initialize response. LLMs see collection
names, document counts, content descriptions, and search strategy
guidance in their system prompt — zero tool calls needed for orientation.
Previously, the only guidance was generic static tool descriptions and
a user-invocable "query" prompt that no LLM would discover on its own.
An LLM connecting to QMD had no idea what collections existed, what they
contained, or how to scope searches effectively.
* change default port to 8181
* fix: BM25 score normalization was inverted
The normalization formula `1 / (1 + |bm25|)` is a decreasing function of
match strength. FTS5 BM25 scores are negative where more negative = better
match (e.g., -10 is strong, -0.5 is weak). The formula mapped:
strong match (raw -10) → 1/(1+10) = 9% ← should be highest
weak match (raw -0.5) → 1/(1+0.5) = 67% ← should be lowest
Three downstream effects:
1. `--min-score 0.5` (or MCP minScore: 0.5) filtered OUT strong matches
and kept only weak ones. The MCP instructions recommend this threshold.
2. CLI `formatScore()` color bands never showed green for BM25 results
(best matches scored ~9%, green threshold is 70%).
3. The strong signal optimization in hybridQuery (skip ~2s LLM expansion
when BM25 already has a clear winner) was dead code — strong matches
scored ~0.09, never reaching the 0.85 threshold.
Fix: `|x| / (1 + |x|)` — same (0,1) range, monotonic, no per-query
normalization needed, but now correctly maps strong → high, weak → low.
The normalization was born broken (Math.max(0, x) clamped all
negative BM25 to 0 → every score = 1.0), then PR #76 changed to
Math.abs which made scores vary but inverted the direction. Neither
state was ever correct.
* fix: rerank cache key ignores chunk content
The rerank cache key was (query, file, model) but the actual text sent
to the reranker is a keyword-selected chunk that varies by query terms.
Two different queries hitting the same file can select different chunks,
but the second query gets a stale cached score from the first chunk.
Example:
Query "auth flow" → selects chunk about authentication → score 0.92
Query "auth tokens" → same file, selects chunk about tokens
→ cache HIT on (query, file, model) → returns 0.92 from wrong chunk
Fix: include full chunk text in cache key. getCacheKey() already
SHA-256 hashes its inputs, so this adds no key bloat — just
disambiguation. Old cache entries become natural misses (different key
shape) and re-warm on next query.
* rename MCP tools for clarity, rewrite descriptions for LLM tool selection
Rename MCP tools: vsearch → vector_search, query → deep_search.
LLMs see these names — self-documenting names reduce reliance on
descriptions for tool selection. CLI commands stay unchanged
(qmd vsearch, qmd query) — different namespace, users type those.
Rewrite all search tool descriptions to be action-oriented:
- search: "Search by keyword. Finds documents containing exact
words and phrases in the query."
- vector_search: "Search by meaning. Finds relevant documents even
when they use different words than the query — handles synonyms,
paraphrases, and related concepts."
- deep_search: "Deep search. Auto-expands the query into variations,
searches each by keyword and meaning, and reranks for top hits
across all results."
Rewrite instructions ladder — each tool says what it does, no
"start here" / "escalate as needed" strategy language.
Delete the "query" prompt (registerPrompt) — it restated what
descriptions + instructions already cover. No LLM proactively
calls prompts/get to learn how to use tools.
* supress HTTP server logs during tests
2026-02-11 05:37:33 +08:00
|
|
|
});
|
|
|
|
|
});
|