CodeIssuesDiscussionsWikiPull RequestsProjectsCommitsActionsReleasesContributorsPulse● GatesSecuritySettingsDeploymentsPipelineInsightsAgents✨ Explain✨ Ask AI✨ Workspace✨ Spec✨ Tests▓ Debt Map✨ NL Search🏛 Archaeology
1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 | /**
* Block D9 — Copilot-style code completion engine.
*
* Exposes a single async function `completeCode` that turns a prefix (+ optional
* suffix) into the characters that should be inserted at the cursor. Used by
* the `/api/copilot/completions` endpoint, which IDE plugins (VS Code, Neovim,
* JetBrains) call on every keystroke.
*
* Design notes:
* - Uses Sonnet 4.6 for quality inline suggestions.
* - Input is clipped aggressively (8k chars before, 2k chars after) so a huge
* file doesn't blow the token budget.
* - Never throws. On any error (bad key, timeout, rate limit) we return an
* empty completion so the editor just stays silent rather than popping a
* scary modal at the user.
* - Inline LRU keeps identical requests (the editor firing the same
* completion twice in rapid succession) from each burning an API call.
* - We log prefix.length only — never the content itself, which may contain
* API keys, private tokens, or proprietary source.
*/
import {
getAnthropic,
MODEL_HAIKU,
extractText,
isAiAvailable,
} from "./ai-client";
import { createHash } from "crypto";
export interface CompleteArgs {
prefix: string;
suffix?: string;
language?: string;
maxTokens?: number;
repoHint?: string;
}
export interface CompleteResult {
completion: string;
model: string;
cached: boolean;
}
// ---------- Inline LRU (size cap 200, TTL ~5 min) ----------
// Intentionally standalone rather than reusing src/lib/cache.ts: completion
// cache entries have a very different shape + access pattern (write-heavy,
// short-lived, never invalidated by repo events) and keeping the logic local
// makes the file easier to reason about and test.
interface CacheEntry {
value: string;
expiresAt: number;
}
const CACHE_MAX = 200;
const CACHE_TTL_MS = 5 * 60 * 1000;
const cacheStore = new Map<string, CacheEntry>();
function cacheKey(prefix: string, suffix: string, language: string): string {
return createHash("sha256")
.update(prefix)
.update("\0")
.update(suffix)
.update("\0")
.update(language)
.digest("hex");
}
function cacheGet(key: string): string | undefined {
const entry = cacheStore.get(key);
if (!entry) return undefined;
if (Date.now() > entry.expiresAt) {
cacheStore.delete(key);
return undefined;
}
// Move to end (MRU).
cacheStore.delete(key);
cacheStore.set(key, entry);
return entry.value;
}
function cacheSet(key: string, value: string): void {
cacheStore.delete(key);
if (cacheStore.size >= CACHE_MAX) {
const oldest = cacheStore.keys().next().value;
if (oldest !== undefined) cacheStore.delete(oldest);
}
cacheStore.set(key, { value, expiresAt: Date.now() + CACHE_TTL_MS });
}
/**
* Strip leading/trailing markdown code fences that Claude sometimes emits
* despite the system prompt forbidding them. Handles:
* ```lang\n...\n```
* ```\n...\n```
* Leaves un-fenced content untouched.
*/
function stripCodeFences(text: string): string {
let out = text;
// Leading fence (optionally with language label)
out = out.replace(/^\s*```[A-Za-z0-9_+-]*\s*\n?/, "");
// Trailing fence
out = out.replace(/\n?\s*```\s*$/, "");
return out;
}
export async function completeCode(
args: CompleteArgs
): Promise<CompleteResult> {
const prefix = (args.prefix || "").slice(-8000);
const suffix = (args.suffix || "").slice(0, 2000);
const language = args.language || "unknown";
const repoHint = args.repoHint || "unknown";
const maxTokens = args.maxTokens ?? 256;
if (!isAiAvailable()) {
return { completion: "", model: "fallback", cached: false };
}
const key = cacheKey(prefix, suffix, language);
const hit = cacheGet(key);
if (hit !== undefined) {
return { completion: hit, model: MODEL_SONNET, cached: true };
}
try {
const client = getAnthropic();
const response = await client.messages.create({
model: MODEL_SONNET,
max_tokens: maxTokens,
system:
"You are a code completion engine. Given a prefix and optional suffix, output ONLY the characters that should be inserted at the cursor. No explanations. No markdown fences. No commentary.",
messages: [
{
role: "user",
content:
`Language: ${language}\n` +
`Repo: ${repoHint}\n\n` +
`PREFIX:\n${prefix}\n\n` +
`SUFFIX:\n${suffix}`,
},
],
});
const raw = extractText(response);
const completion = stripCodeFences(raw);
cacheSet(key, completion);
return { completion, model: MODEL_SONNET, cached: false };
} catch (err) {
// Never throw — the editor should degrade silently. Log length only, not
// the prefix itself, which can contain secrets.
console.error(
"[ai-completion] completeCode failed (prefix.length=" +
prefix.length +
"):",
(err as Error)?.message || err
);
return { completion: "", model: "error", cached: false };
}
}
/**
* Test-only helpers. Not part of the public API — tests use these to seed
* entries so they can exercise the cache without hitting the real Anthropic
* endpoint, and to reset state between runs.
*/
export const __test = {
cacheKey,
cacheGet,
cacheSet,
clear() {
cacheStore.clear();
},
size() {
return cacheStore.size;
},
stripCodeFences,
};
|