Pre-launch — Gluecron is in final validation. Public signups and git hosting for non-owner users open after launch review.
CodeIssuesDiscussionsWikiPull RequestsProjectsCommitsActionsReleasesContributorsPulse● GatesSecuritySettingsDeploymentsPipelineInsightsAgents✨ Explain✨ Ask AI✨ Workspace✨ Spec✨ Tests▓ Debt Map✨ NL Search🏛 Archaeology
claude/adoring-hopper-5x74bqclaude/affectionate-feynman-ykrf1hclaude/architecture-audit-design-wxprenclaude/build-status-update-3MXsfclaude/charming-meitner-mllb5rclaude/compare-gate-gluecron-s4mFQclaude/confident-faraday-tikcwbclaude/continue-work-XMTlIclaude/crontech-gluecron-deploy-7MIECclaude/crontech-platform-setup-SeKfwclaude/design-2026claude/ecstatic-ptolemy-jMdigclaude/enhance-github-integration-QNHdGclaude/fix-aa-loop-issue-PonMQclaude/fix-actions-and-processclaude/fix-desktop-errors-XqoW8claude/fix-red-workflowsclaude/fix-website-access-6FKJNclaude/gatetest-integration-hardeningclaude/github-audit-improvements-bDFr9claude/gluecron-launch-status-FoMRlclaude/hopeful-lamport-olfCTclaude/issue-to-pr-and-protectionsclaude/jolly-heisenberg-2sg1Qclaude/launch-preparation-QmTb6claude/new-session-xk1l7claude/plan-platform-architecture-kkN4yclaude/platform-analysis-roadmap-1nUGLclaude/platform-launch-assessment-8dWV8claude/polish-platform-release-AeDrUclaude/resume-previous-work-KzyLwclaude/review-crontech-handoff-qYEVqclaude/review-project-completeness-lHhS2claude/review-readme-docs-ulqPKclaude/serene-edison-rj87weclaude/setup-multi-repo-dev-BCwNQclaude/ship-fixes-and-tests-Jvz1cclaude/site-audit-competitive-pctlwgclaude/site-migration-vercel-XstpKclaude/standalone-product-repos-XHFTDcopilot/feat-smart-empty-states-keyboard-first-enhancementcopilot/feat-smart-morning-digest-review-context-restorecopilot/fix-and-process-workflowscopilot/update-ai-powered-code-reviewfeat/debt-mapfeat/push-policy-codeowners-hardeningfeat/smart-digest-contextfeat/stage-impactfeat/t1-secret-migrationfeat/u-polishfeat/w-self-hostfeat/w2-claude-configfix/agent-journey-orphan-sweepgatetest/auto-fix-1776586424172gatetest/auto-fix-1776586534814gatetest/auto-fix-1776590685143gatetest/auto-fix-1776590808199mainops/redeploy-retriggerstyle/dxt-cta-themeworktree-agent-a3377aad30d55da26worktree-agent-a7ef607b7ee1d6c74
ai-completion.ts6.1 KB · 202 lines
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
/**
 * Block D9 — Copilot-style code completion engine.
 *
 * Exposes a single async function `completeCode` that turns a prefix (+ optional
 * suffix) into the characters that should be inserted at the cursor. Used by
 * the `/api/copilot/completions` endpoint, which IDE plugins (VS Code, Neovim,
 * JetBrains) call on every keystroke.
 *
 * Design notes:
 *   - Uses Haiku because latency matters more than depth for inline suggestions.
 *   - Input is clipped aggressively (8k chars before, 2k chars after) so a huge
 *     file doesn't blow the token budget.
 *   - Never throws. On any error (bad key, timeout, rate limit) we return an
 *     empty completion so the editor just stays silent rather than popping a
 *     scary modal at the user.
 *   - Inline LRU keeps identical requests (the editor firing the same
 *     completion twice in rapid succession) from each burning an API call.
 *   - We log prefix.length only — never the content itself, which may contain
 *     API keys, private tokens, or proprietary source.
 */

import {
  getAnthropic,
  MODEL_HAIKU,
  extractText,
  isAiAvailable,
} from "./ai-client";
import { createHash } from "crypto";

export interface CompleteArgs {
  prefix: string;
  suffix?: string;
  language?: string;
  maxTokens?: number;
  repoHint?: string;
}

export interface CompleteResult {
  completion: string;
  model: string;
  cached: boolean;
}

// ---------- Inline LRU (size cap 200, TTL ~5 min) ----------
// Intentionally standalone rather than reusing src/lib/cache.ts: completion
// cache entries have a very different shape + access pattern (write-heavy,
// short-lived, never invalidated by repo events) and keeping the logic local
// makes the file easier to reason about and test.

interface CacheEntry {
  value: string;
  expiresAt: number;
}

const CACHE_MAX = 200;
const CACHE_TTL_MS = 5 * 60 * 1000;
const cacheStore = new Map<string, CacheEntry>();

function cacheKey(prefix: string, suffix: string, language: string): string {
  return createHash("sha256")
    .update(prefix)
    .update("\0")
    .update(suffix)
    .update("\0")
    .update(language)
    .digest("hex");
}

function cacheGet(key: string): string | undefined {
  const entry = cacheStore.get(key);
  if (!entry) return undefined;
  if (Date.now() > entry.expiresAt) {
    cacheStore.delete(key);
    return undefined;
  }
  // Move to end (MRU).
  cacheStore.delete(key);
  cacheStore.set(key, entry);
  return entry.value;
}

function cacheSet(key: string, value: string): void {
  cacheStore.delete(key);
  if (cacheStore.size >= CACHE_MAX) {
    const oldest = cacheStore.keys().next().value;
    if (oldest !== undefined) cacheStore.delete(oldest);
  }
  cacheStore.set(key, { value, expiresAt: Date.now() + CACHE_TTL_MS });
}

/**
 * Strip leading/trailing markdown code fences that Claude sometimes emits
 * despite the system prompt forbidding them. Handles:
 *   ```lang\n...\n```
 *   ```\n...\n```
 * Leaves un-fenced content untouched.
 */
function stripCodeFences(text: string): string {
  let out = text;
  // Leading fence (optionally with language label)
  out = out.replace(/^\s*```[A-Za-z0-9_+-]*\s*\n?/, "");
  // Trailing fence
  out = out.replace(/\n?\s*```\s*$/, "");
  return out;
}

export async function completeCode(
  args: CompleteArgs
): Promise<CompleteResult> {
  const prefix = (args.prefix || "").slice(-8000);
  const suffix = (args.suffix || "").slice(0, 2000);
  const language = args.language || "unknown";
  const repoHint = args.repoHint || "unknown";
  const maxTokens = args.maxTokens ?? 256;

  if (!isAiAvailable()) {
    return { completion: "", model: "fallback", cached: false };
  }

  const key = cacheKey(prefix, suffix, language);
  const hit = cacheGet(key);
  if (hit !== undefined) {
    // Log the cache hit too — useful for cost-tracking and rate insight.
    try {
      const { logAiEvent } = await import("./ai-flywheel");
      logAiEvent({
        actionType: "completion",
        model: MODEL_HAIKU,
        summary: `cached completion (${language})`,
        latencyMs: 0,
        success: true,
        metadata: { cached: true, language, repoHint },
      });
    } catch {
      /* telemetry must not break completion */
    }
    return { completion: hit, model: MODEL_HAIKU, cached: true };
  }

  try {
    const { recordAi } = await import("./ai-flywheel");
    const client = getAnthropic();
    const response = await recordAi(
      {
        actionType: "completion",
        model: MODEL_HAIKU,
        summary: `code completion (${language})`,
        metadata: { language, repoHint, prefixLen: prefix.length },
      },
      () =>
        client.messages.create({
          model: MODEL_HAIKU,
          max_tokens: maxTokens,
          system:
            "You are a code completion engine. Given a prefix and optional suffix, output ONLY the characters that should be inserted at the cursor. No explanations. No markdown fences. No commentary.",
          messages: [
            {
              role: "user",
              content:
                `Language: ${language}\n` +
                `Repo: ${repoHint}\n\n` +
                `PREFIX:\n${prefix}\n\n` +
                `SUFFIX:\n${suffix}`,
            },
          ],
        })
    );

    const raw = extractText(response);
    const completion = stripCodeFences(raw);
    cacheSet(key, completion);
    return { completion, model: MODEL_HAIKU, cached: false };
  } catch (err) {
    // Never throw — the editor should degrade silently. Log length only, not
    // the prefix itself, which can contain secrets.
    console.error(
      "[ai-completion] completeCode failed (prefix.length=" +
        prefix.length +
        "):",
      (err as Error)?.message || err
    );
    return { completion: "", model: "error", cached: false };
  }
}

/**
 * Test-only helpers. Not part of the public API — tests use these to seed
 * entries so they can exercise the cache without hitting the real Anthropic
 * endpoint, and to reset state between runs.
 */
export const __test = {
  cacheKey,
  cacheGet,
  cacheSet,
  clear() {
    cacheStore.clear();
  },
  size() {
    return cacheStore.size;
  },
  stripCodeFences,
};