Repository navigation
Expand file tree
/
Copy pathinstinct.html
More file actions
57 lines (57 loc) · 41.6 KB
/
Copy pathinstinct.html
File metadata and controls
57 lines (57 loc) · 41.6 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
<!DOCTYPE html><!--UZagGqlyGLrSk_oeqt4G9--><html lang="en"><head><meta charSet="utf-8"/><meta name="viewport" content="width=device-width, initial-scale=1"/><link rel="preload" href="/_next/static/media/03fc1b4a8d284b5e-s.p.af4fcd24.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="preload" href="/_next/static/media/99e609270109b47d-s.p.64b9304e.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="preload" as="image" href="/images/blog/instinct-hero.png"/><link rel="stylesheet" href="/_next/static/chunks/4723c630d603d667.css" data-precedence="next"/><link rel="stylesheet" href="/_next/static/chunks/d7fbbb6524d78f08.css" data-precedence="next"/><link rel="preload" as="script" fetchPriority="low" href="/_next/static/chunks/56c576438cbfc236.js"/><script src="/_next/static/chunks/7ebaf881f5fa7446.js" async=""></script><script src="/_next/static/chunks/03666d851f44ca00.js" async=""></script><script src="/_next/static/chunks/4c788cda96f09061.js" async=""></script><script src="/_next/static/chunks/turbopack-cd9704f0e36c424b.js" async=""></script><script src="/_next/static/chunks/39048bb5c98cef4d.js" async=""></script><script src="/_next/static/chunks/ff1a16fafef87110.js" async=""></script><script src="/_next/static/chunks/803a574de9eda7ae.js" async=""></script><script src="/_next/static/chunks/c1b976bb3ed182a2.js" async=""></script><script src="/_next/static/chunks/da2f50d302e01427.js" async=""></script><meta name="next-size-adjust" content=""/><title>Introducing Instinct: the world's best open Next Edit model, built by Continue | Continue - Blog</title><meta name="description" content="Meet Instinct, Continue's open Next Edit model that's built to predict your next edit and keep you in flow"/><link rel="canonical" href="https://blog.continue.dev/instinct/"/><meta property="og:title" content="Introducing Instinct: the world's best open Next Edit model, built by Continue"/><meta property="og:description" content="Meet Instinct, Continue's open Next Edit model that's built to predict your next edit and keep you in flow"/><meta property="og:image" content="https://blog.continue.dev/images/blog/instinct-hero.png"/><meta property="og:type" content="article"/><meta property="article:published_time" content="2025-09-04T10:17:55.000-07:00"/><meta property="article:modified_time" content="2025-09-04T11:32:20.000-07:00"/><meta property="article:author" content="Adarsh Iyer"/><meta property="article:author" content="Nate Sesti"/><meta name="twitter:card" content="summary_large_image"/><meta name="twitter:title" content="Introducing Instinct: the world's best open Next Edit model, built by Continue"/><meta name="twitter:description" content="Meet Instinct, Continue's open Next Edit model that's built to predict your next edit and keep you in flow"/><meta name="twitter:image" content="https://blog.continue.dev/images/blog/instinct-hero.png"/><link rel="icon" href="/favicon.png"/><script src="/_next/static/chunks/a6dad97d9634a72d.js" noModule=""></script></head><body><div hidden=""><!--$--><!--/$--></div><div class="fixed inset-0 overflow-y-auto bg-[hsl(0_0%_95.3%)] ibm_plex_sans_b42bcf6c-module__0s_T1q__variable ibm_plex_mono_ce326e86-module__Q4TVOq__variable" style="font-family:var(--font-homepage-sans), system-ui, sans-serif"><style>
.font-mono {
font-family: var(--font-homepage-mono), ui-monospace, monospace !important;
}
</style><div class="min-h-screen flex flex-col"><script type="application/ld+json">{"@context":"https://schema.org","@type":"BlogPosting","headline":"Introducing Instinct: the world's best open Next Edit model, built by Continue","description":"Meet Instinct, Continue's open Next Edit model that's built to predict your next edit and keep you in flow","datePublished":"2025-09-04T10:17:55.000-07:00","dateModified":"2025-09-04T11:32:20.000-07:00","image":"/images/blog/instinct-hero.png","author":[{"@type":"Person","name":"Adarsh Iyer"},{"@type":"Person","name":"Nate Sesti","url":"https://x.com/natesesti"}],"publisher":{"@type":"Organization","name":"Continue","url":"https://continue.dev"}}</script><main class="flex-1 w-full"><article class="max-w-3xl mx-auto px-6 sm:px-8 py-8"><a class="inline-flex items-center text-xs font-mono uppercase tracking-[0.15em] text-black/25 hover:text-black/50 transition-colors mb-8" href="/">← Back to blog</a><div class="mb-8 bg-black/[0.015] shadow-[0_2px_20px_rgb(0,0,0,0.04)] overflow-hidden"><img alt="Introducing Instinct: the world's best open Next Edit model, built by Continue" width="1200" height="675" decoding="async" data-nimg="1" class="w-full h-auto" style="color:transparent" src="/images/blog/instinct-hero.png"/></div><h1 class="text-[2.5rem] font-light tracking-tight text-black/90 mb-4">Introducing Instinct: the world's best open Next Edit model, built by Continue</h1><div class="flex flex-wrap items-center gap-x-4 gap-y-1 mb-8"><span class="text-[15px] text-black/50"><span>Adarsh Iyer</span><span>, <a href="https://x.com/natesesti" target="_blank" rel="noopener noreferrer" class="hover:text-black/70 transition-colors underline decoration-black/20 underline-offset-2 hover:decoration-black/50">Nate Sesti</a></span></span><span class="text-xs font-mono uppercase tracking-[0.15em] text-black/25">September 4, 2025<!-- --> · <!-- -->7<!-- --> min read</span></div><nav class="blog-toc"><h2 class="blog-toc-title">Table of Contents</h2><ul class="blog-toc-list"><li class=""><a href="#why-train-an-open-model">Why Train an Open Model?</a></li><li class=""><a href="#what-is-next-edit">What is Next Edit?</a></li><li class=""><a href="#training-the-model">Training the Model</a></li><li class="blog-toc-indent"><a href="#real-world-training-data">Real World Training Data</a></li><li class="blog-toc-indent"><a href="#maintaining-multilingual-support">Maintaining Multilingual Support</a></li><li class="blog-toc-indent"><a href="#robust-training-for-the-next-edit-task">Robust Training for the Next Edit Task</a></li><li class=""><a href="#performance-results-evals-for-quality-and-speed">Performance Results: Evals for Quality and Speed</a></li><li class=""><a href="#now-what">Now what?</a></li></ul></nav><div class="blog-content prose prose-lg max-w-none"><p>We're thrilled to share <a href="https://huggingface.co/continuedev/instinct?ref=blog.continue.dev">Instinct</a>, our open Next Edit model, which intelligently predicts your next move to keep you in flow.</p>
<p>When we launched <a href="https://blog.continue.dev/next-edit-powered-by-mercury-coder/">Next Edit</a>, we premiered it with <a href="https://hub.continue.dev/inceptionlabs/mercury-coder?ref=blog.continue.dev">Mercury Coder</a> from Inception. Today, we're expanding the possibilities with Instinct: an open, in-house–trained model that developers can run locally on their own GPUs.</p>
<p>As you code, Instinct sees your edit trajectory and carries out the next step automatically—an estimated 6.4x faster than manual editing.</p>
<blockquote>
<p>💡 <a href="https://docs.continue.dev/guides/instinct?ref=blog.continue.dev">Try it today</a> with Ollama and Continue in VS Code</p>
</blockquote>
<h2 id="why-train-an-open-model">Why Train an Open Model?</h2>
<p>While open models for agentic coding tasks have advanced rapidly in recent months, work on Next Edit remains nascent. Most progress had been made by Zed on their model Zeta and we are glad to be able to build upon what they learned. One of our goals is to highlight the open opportunity and lay the groundwork for future efforts that will benefit not only our team, but also our community and the broader developer ecosystem.</p>
<p>And while existing models like Mercury Coder have shown excellent performance, Instinct enables developers to run or customize a Next Edit model on their own GPUs, addressing privacy and customization needs.</p>
<h2 id="what-is-next-edit">What is Next Edit?</h2>
<table><thead><tr><th>Aspect</th><th>Traditional Autocomplete</th><th>Next Edit</th></tr></thead><tbody><tr><td>Scope of Change</td><td>Inserts text only at the cursor</td><td>Rewrites code windows (deletions, insertions, replacements)</td></tr><tr><td>Complex Changes</td><td>Requires multiple acceptances</td><td>Handles complex refactoring in a single operation</td></tr><tr><td>Code Restructuring</td><td>Cannot delete or restructure code</td><td>Understands editing trajectories and developer intent</td></tr><tr><td>Developer Flow</td><td>Frequent interruptions break flow</td><td>Maintains flow with fewer interruptions</td></tr></tbody></table>
<p>Traditional tab autocomplete can only <em>insert</em> code at the cursor as you type. This is helpful when ripping through boilerplate code, but most of the time developers are refactoring, maintaining, iterating on—<em>editing</em>—code.</p>
<p>As an example, refactoring a function might require: deleting old parameters (5 keystrokes), moving to the return statement (2 cursor jumps), changing the return type (8 keystrokes), and updating the function body (20+ keystrokes and 5+ cursor jumps). With Instinct, this entire sequence becomes a single tab-to-accept action, transforming what would be 40+ manual operations into one.</p>
<h2 id="training-the-model">Training the Model</h2>
<h3 id="real-world-training-data">Real World Training Data</h3>
<p>To create our Next Edit model, we needed high quality training data. Instead of synthetically generating examples, we automatically collected over 4,000 real-world edits from the Continue team as they worked on our open-source code. This is an order of magnitude more than the Zeta dataset released earlier this year, and represents real world development patterns better than the purely synthetic data that could be constructed from git commits.</p>
<p>Each data example includes:</p>
<ul>
<li>The five most recent edits the developer made</li>
<li>Relevant context from other files</li>
<li>The region of code to rewrite</li>
<li>The ground-truth change the developer made to that region</li>
</ul>
<p>A fundamental question we ran into was what constituted an "edit." Every keypress? Every time the user hit save? All keypresses made within some time window? After many iterations, we defined a set of line- and timing-based heuristics that "chunked" individual keypresses into well-formed and self-contained diffs.</p>
<p>When we examined sequences of such diffs, we noticed that developers sometimes jumped back and forth, iterating on the same one or two lines repetitively. We defined filters to throw out such examples since we wanted Instinct to make efficient, non-repetitive edits.</p>
<p>Continue's <a href="https://blog.continue.dev/root-path-context-the-secret-ingredient-in-continues-autocomplete-prompt/">autocomplete context pipelines</a> provided relevant information about the codebase, with content from the current file included in the prompt as well. We define the editable region (the span of code to be rewritten by the model) as starting one line above the cursor and ending five lines below. This decision was made based on the natural features of the diffs we saw in our dataset.</p>
<p>Taken together, the edit sequence, context, current file content, and editable region provide the necessary information to infer the developer's intentions and predict the next step in the edit sequence.</p>
<h3 id="maintaining-multilingual-support">Maintaining Multilingual Support</h3>
<p>One problem we ran into was that the Continue team works mostly with Typescript code. However, we wanted our model to preserve support for multiple languages. In addition to training the model in a robust way (more on this later), we used a self-hosted Qwen3-Coder-30B model to synthetically "translate" diffs, context, and file contents into Java, C, Python, and Rust—thus bootstrapping a multilingual dataset off of the Typescript data. A set of precise data correctors and filters ensured high quality and a similar distribution of edits among the 4,000+ synthetic examples.</p>
<h3 id="robust-training-for-the-next-edit-task">Robust Training for the Next Edit Task</h3>
<p>With a multilingual dataset ready to go, it was time to move on to supervised fine-tuning (SFT). SFT for specific tasks like Next Edit is typically done using Low-Rank Adaptation (LoRA), since although <a href="https://arxiv.org/abs/2405.09673?ref=blog.continue.dev">it learns less, it also forgets less</a> of the pre-trained model's general coding abilities.</p>
<p>The central problem with using LoRA, however, is that it fixes the parameters to be fine-tuned before training even begins. A preferable approach is to <em>discover</em> which parameters' weight updates are most important, and update only those. Rather than treating Next Edit as learning new <em>knowledge</em> at the expense of forgetting previous coding ability, the model can ideally adapt to the <em>task</em> of Next Editing while retaining its pre-trained coding knowledge.</p>
<p>We found exactly such a solution in the Selective Knowledge Transfer (SeleKT) algorithm used to train the <a href="https://www.microsoft.com/en-us/research/publication/nextcoder-robust-adaptation-of-code-lms-to-diverse-code-edits/?ref=blog.continue.dev">NextCoder</a> instructed code-editing models. SeleKT computes dense gradients (as in full fine-tuning), extracts the top-<em>k</em> gradients by magnitude, and then applies only those sparse weight updates. It discovers through practice what actually needs to change, rather than guessing beforehand, and as a result, only the most important weights for the Next Edit <em>task</em> are updated. Moreover, zero-ing out small weight updates helps prevent overfitting and erosion of previous coding knowledge, problems that full fine-tuning would suffer from.</p>
<p>We fine-tuned 5% of <a href="https://arxiv.org/abs/2409.12186?ref=blog.continue.dev">Qwen2.5-Coder-7B</a>'s parameters using SeleKT. After initial hyperparameter sweeps, the training process was standard, using a log warmup and cosine decay learning rate schedule for 5 epochs. We used the <a href="https://arxiv.org/abs/2009.10297?ref=blog.continue.dev">CodeBLEU</a> score as a quick proxy eval between predicted and ground truth rewrites during training. Ablating CodeBLEU scores across different languages in the dataset allowed us to adjust the data mixture, resulting in high validation performance across languages. Due to the robust training, only a small number of multilingual examples were needed.</p>
<h2 id="performance-results-evals-for-quality-and-speed">Performance Results: Evals for Quality and Speed</h2>
<table><thead><tr><th>3.877</th><th>6.4x faster*</th></tr></thead><tbody><tr><td>Average LLM Judge Score (new open-source state-of-the-art)</td><td>than manually typing out edits <em>* on our 8xH100 cluster</em></td></tr></tbody></table>
<p>It's nontrivial to evaluate the quality of a Next Edit suggestion formally, since there are often many ways to accomplish the same coding goal. Accordingly, we deployed Claude as an LLM judge with instructions to assess the quality of Next Edit suggestions on a five-point scale, where:</p>
<ul>
<li>A score of five corresponds to a functional match with the developer's ground-truth edit,</li>
<li>A score of four corresponds to a similar edit to the developer's ground truth edit, although not an exact functional match,</li>
<li>A score of three corresponds to an edit that does not match the ground truth but would reasonably be made by an expert developer in such a scenario,</li>
<li>A score of two corresponds to an edit that does not logically follow from the previous edits and context,</li>
<li>A score of one corresponds to an edit that is likely to hinder developer progress, such as large deletions or complete irrelevance, and</li>
<li>A score of zero corresponds to a malformed rewrite that does not line up with the editable region.</li>
</ul>
<p>This evaluation method is similar to the one Zed uses for Zeta. We noticed that their judge prompt caused the LLM to output scores of only zero or five, and that it relied on handcrafted assertions about the next edit. We created a different system prompt that helps represent the full spectrum of scores and compares the model's edit to the developer's ground truth change instead of relying on assertions. With just minor adjustments to account for our different Next Edit prompt structures, Instinct's average score of 3.877 outperforms Zeta's score of 3.735 on the held-out eval set. We're excited to see further work continuing to improve on this benchmark.</p>
<p>Not only does Instinct provide high-quality suggestions, our keystroke-distance eval, loosely based on that of <a href="https://arxiv.org/abs/2305.18584?ref=blog.continue.dev">Coeditor</a>, demonstrates that it also massively speeds up your workflow. By backtracking through the dynamic programming (DP) table of a Levenshtein distance calculator, it's possible to extract a character-level diff between the editable region and the suggested rewrite. That character-level diff can be "chunked" into edit operations, <em>e.g.</em>, adding three characters at one location, deleting five characters at another location, and such.</p>
<p>The minimum time to carry out the full edit is given by the optimal combination of keypresses and cursor jumps that accomplish all the edit operations. This can be posed as another DP problem. We assume the developer types at an average of 90 WPM, simulate the ability to highlight and delete as opposed to repeatedly hitting the backspace key, allow for small arrow movements instead of more time-intensive cursor jumps, and use the spatial distance between cursor locations (<em>i.e.</em> line and character) instead of just the index within the whole string. This lower bound on manual editing time is compared to average model inference time on our internal SGLang endpoint plus the time for one keypress (tab-to-accept).</p>
<p>The result is that even if you immediately knew exactly what edit to make, <em>and</em> took the DP-optimal sequence of actions to carry it out at 90 WPM, using the model would still provide the high-quality edit 6.4 times faster.</p>
<h2 id="now-what">Now what?</h2>
<p>First, we encourage you to try it out! Instinct is a 7B model so you should expect it to be slow on most laptops, but with sufficient hardware it is a great option for self-hosting. Read <a href="https://docs.continue.dev/guides/instinct?ref=blog.continue.dev">our guide</a> to learn more.</p>
<p>If you want to build upon our dataset, training pipelines, or open weights (for example to run <a href="https://arxiv.org/abs/2402.01306?ref=blog.continue.dev">KTO</a> on your own accept/reject data) we'd recommend exploring our <a href="https://github.com/continuedev/instinct?ref=blog.continue.dev">GitHub repository</a>, <a href="https://huggingface.co/continuedev/instinct?ref=blog.continue.dev">HuggingFace model card</a>, and dataset.</p>
<p>Most importantly, if you are interested in furthering the state of the art, either as part of the community or the Continue team, please <a href="mailto:nate@continue.dev">reach out</a>!</p></div></article></main></div><!--$--><!--/$--></div><script src="/_next/static/chunks/56c576438cbfc236.js" id="_R_" async=""></script><script>(self.__next_f=self.__next_f||[]).push([0])</script><script>self.__next_f.push([1,"1:\"$Sreact.fragment\"\n2:I[47344,[\"/_next/static/chunks/39048bb5c98cef4d.js\"],\"ForceTheme\"]\n3:I[39756,[\"/_next/static/chunks/ff1a16fafef87110.js\",\"/_next/static/chunks/803a574de9eda7ae.js\"],\"default\"]\n4:I[37457,[\"/_next/static/chunks/ff1a16fafef87110.js\",\"/_next/static/chunks/803a574de9eda7ae.js\"],\"default\"]\n5:I[22016,[\"/_next/static/chunks/39048bb5c98cef4d.js\",\"/_next/static/chunks/c1b976bb3ed182a2.js\",\"/_next/static/chunks/da2f50d302e01427.js\"],\"\"]\n7:I[97367,[\"/_next/static/chunks/ff1a16fafef87110.js\",\"/_next/static/chunks/803a574de9eda7ae.js\"],\"OutletBoundary\"]\n8:\"$Sreact.suspense\"\na:I[97367,[\"/_next/static/chunks/ff1a16fafef87110.js\",\"/_next/static/chunks/803a574de9eda7ae.js\"],\"ViewportBoundary\"]\nc:I[97367,[\"/_next/static/chunks/ff1a16fafef87110.js\",\"/_next/static/chunks/803a574de9eda7ae.js\"],\"MetadataBoundary\"]\ne:I[68027,[],\"default\"]\n:HL[\"/_next/static/chunks/4723c630d603d667.css\",\"style\"]\n:HL[\"/_next/static/media/03fc1b4a8d284b5e-s.p.af4fcd24.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/media/99e609270109b47d-s.p.64b9304e.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/chunks/d7fbbb6524d78f08.css\",\"style\"]\n"])</script><script>self.__next_f.push([1,"0:{\"P\":null,\"b\":\"UZagGqlyGLrSk-oeqt4G9\",\"c\":[\"\",\"instinct\"],\"q\":\"\",\"i\":false,\"f\":[[[\"\",{\"children\":[[\"slug\",\"instinct\",\"d\"],{\"children\":[\"__PAGE__\",{}]}]},\"$undefined\",\"$undefined\",true],[[\"$\",\"$1\",\"c\",{\"children\":[[[\"$\",\"link\",\"0\",{\"rel\":\"stylesheet\",\"href\":\"/_next/static/chunks/4723c630d603d667.css\",\"precedence\":\"next\",\"crossOrigin\":\"$undefined\",\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-0\",{\"src\":\"/_next/static/chunks/39048bb5c98cef4d.js\",\"async\":true,\"nonce\":\"$undefined\"}]],[\"$\",\"html\",null,{\"lang\":\"en\",\"children\":[\"$\",\"body\",null,{\"children\":[\"$\",\"$L2\",null,{\"theme\":\"light\",\"children\":[\"$\",\"div\",null,{\"className\":\"fixed inset-0 overflow-y-auto bg-[hsl(0_0%_95.3%)] ibm_plex_sans_b42bcf6c-module__0s_T1q__variable ibm_plex_mono_ce326e86-module__Q4TVOq__variable\",\"style\":{\"fontFamily\":\"var(--font-homepage-sans), system-ui, sans-serif\"},\"children\":[[\"$\",\"style\",null,{\"children\":\"\\n .font-mono {\\n font-family: var(--font-homepage-mono), ui-monospace, monospace !important;\\n }\\n \"}],[\"$\",\"$L3\",null,{\"parallelRouterKey\":\"children\",\"error\":\"$undefined\",\"errorStyles\":\"$undefined\",\"errorScripts\":\"$undefined\",\"template\":[\"$\",\"$L4\",null,{}],\"templateStyles\":\"$undefined\",\"templateScripts\":\"$undefined\",\"notFound\":[[\"$\",\"div\",null,{\"className\":\"min-h-screen flex flex-col\",\"children\":[\"$\",\"main\",null,{\"className\":\"flex-1 w-full px-6 sm:px-8 lg:px-16 py-8 lg:py-16\",\"children\":[\"$\",\"div\",null,{\"className\":\"w-full lg:grid lg:grid-cols-2 lg:gap-16 lg:items-center\",\"children\":[\"$\",\"div\",null,{\"children\":[[\"$\",\"p\",null,{\"className\":\"text-xs font-mono uppercase tracking-[0.15em] text-black/25 mb-4\",\"children\":\"404\"}],[\"$\",\"h1\",null,{\"className\":\"text-[2.5rem] sm:text-5xl font-light tracking-tight text-black/90 mb-8\",\"children\":\"Page not found\"}],[\"$\",\"div\",null,{\"className\":\"flex items-center gap-6\",\"children\":[[\"$\",\"$L5\",null,{\"href\":\"/\",\"className\":\"cr-nav-link text-[15px] text-black/40 hover:text-black/70 transition-colors tracking-wide uppercase font-mono\",\"children\":\"Back to Blog\"}],[\"$\",\"a\",null,{\"href\":\"https://continue.dev\",\"className\":\"cr-nav-link text-[15px] text-black/40 hover:text-black/70 transition-colors tracking-wide uppercase font-mono\",\"children\":\"Continue Home\"}]]}]]}]}]}]}],[]],\"forbidden\":\"$undefined\",\"unauthorized\":\"$undefined\"}]]}]}]}]}]]}],{\"children\":[[\"$\",\"$1\",\"c\",{\"children\":[null,[\"$\",\"$L3\",null,{\"parallelRouterKey\":\"children\",\"error\":\"$undefined\",\"errorStyles\":\"$undefined\",\"errorScripts\":\"$undefined\",\"template\":[\"$\",\"$L4\",null,{}],\"templateStyles\":\"$undefined\",\"templateScripts\":\"$undefined\",\"notFound\":\"$undefined\",\"forbidden\":\"$undefined\",\"unauthorized\":\"$undefined\"}]]}],{\"children\":[[\"$\",\"$1\",\"c\",{\"children\":[\"$L6\",[[\"$\",\"link\",\"0\",{\"rel\":\"stylesheet\",\"href\":\"/_next/static/chunks/d7fbbb6524d78f08.css\",\"precedence\":\"next\",\"crossOrigin\":\"$undefined\",\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-0\",{\"src\":\"/_next/static/chunks/c1b976bb3ed182a2.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-1\",{\"src\":\"/_next/static/chunks/da2f50d302e01427.js\",\"async\":true,\"nonce\":\"$undefined\"}]],[\"$\",\"$L7\",null,{\"children\":[\"$\",\"$8\",null,{\"name\":\"Next.MetadataOutlet\",\"children\":\"$@9\"}]}]]}],{},null,false,false]},null,false,false]},null,false,false],[\"$\",\"$1\",\"h\",{\"children\":[null,[\"$\",\"$La\",null,{\"children\":\"$Lb\"}],[\"$\",\"div\",null,{\"hidden\":true,\"children\":[\"$\",\"$Lc\",null,{\"children\":[\"$\",\"$8\",null,{\"name\":\"Next.Metadata\",\"children\":\"$Ld\"}]}]}],[\"$\",\"meta\",null,{\"name\":\"next-size-adjust\",\"content\":\"\"}]]}],false]],\"m\":\"$undefined\",\"G\":[\"$e\",[]],\"S\":true}\n"])</script><script>self.__next_f.push([1,"f:I[5500,[\"/_next/static/chunks/39048bb5c98cef4d.js\",\"/_next/static/chunks/c1b976bb3ed182a2.js\",\"/_next/static/chunks/da2f50d302e01427.js\"],\"Image\"]\n10:I[77844,[\"/_next/static/chunks/39048bb5c98cef4d.js\",\"/_next/static/chunks/c1b976bb3ed182a2.js\",\"/_next/static/chunks/da2f50d302e01427.js\"],\"default\"]\n11:T2e8f,"])</script><script>self.__next_f.push([1,"\nWe're thrilled to share [Instinct](https://huggingface.co/continuedev/instinct?ref=blog.continue.dev), our open Next Edit model, which intelligently predicts your next move to keep you in flow.\n\nWhen we launched [Next Edit](https://blog.continue.dev/next-edit-powered-by-mercury-coder/), we premiered it with [Mercury Coder](https://hub.continue.dev/inceptionlabs/mercury-coder?ref=blog.continue.dev) from Inception. Today, we're expanding the possibilities with Instinct: an open, in-house–trained model that developers can run locally on their own GPUs.\n\nAs you code, Instinct sees your edit trajectory and carries out the next step automatically—an estimated 6.4x faster than manual editing.\n\n\u003e 💡 [Try it today](https://docs.continue.dev/guides/instinct?ref=blog.continue.dev) with Ollama and Continue in VS Code\n\n## Why Train an Open Model?\n\nWhile open models for agentic coding tasks have advanced rapidly in recent months, work on Next Edit remains nascent. Most progress had been made by Zed on their model Zeta and we are glad to be able to build upon what they learned. One of our goals is to highlight the open opportunity and lay the groundwork for future efforts that will benefit not only our team, but also our community and the broader developer ecosystem.\n\nAnd while existing models like Mercury Coder have shown excellent performance, Instinct enables developers to run or customize a Next Edit model on their own GPUs, addressing privacy and customization needs.\n\n## What is Next Edit?\n\n| Aspect | Traditional Autocomplete | Next Edit |\n| --- | --- | --- |\n| Scope of Change | Inserts text only at the cursor | Rewrites code windows (deletions, insertions, replacements) |\n| Complex Changes | Requires multiple acceptances | Handles complex refactoring in a single operation |\n| Code Restructuring | Cannot delete or restructure code | Understands editing trajectories and developer intent |\n| Developer Flow | Frequent interruptions break flow | Maintains flow with fewer interruptions |\n\nTraditional tab autocomplete can only _insert_ code at the cursor as you type. This is helpful when ripping through boilerplate code, but most of the time developers are refactoring, maintaining, iterating on—_editing_—code.\n\nAs an example, refactoring a function might require: deleting old parameters (5 keystrokes), moving to the return statement (2 cursor jumps), changing the return type (8 keystrokes), and updating the function body (20+ keystrokes and 5+ cursor jumps). With Instinct, this entire sequence becomes a single tab-to-accept action, transforming what would be 40+ manual operations into one.\n\n## Training the Model\n\n### Real World Training Data\n\nTo create our Next Edit model, we needed high quality training data. Instead of synthetically generating examples, we automatically collected over 4,000 real-world edits from the Continue team as they worked on our open-source code. This is an order of magnitude more than the Zeta dataset released earlier this year, and represents real world development patterns better than the purely synthetic data that could be constructed from git commits.\n\nEach data example includes: \n\n- The five most recent edits the developer made\n- Relevant context from other files\n- The region of code to rewrite\n- The ground-truth change the developer made to that region\n\nA fundamental question we ran into was what constituted an \"edit.\" Every keypress? Every time the user hit save? All keypresses made within some time window? After many iterations, we defined a set of line- and timing-based heuristics that \"chunked\" individual keypresses into well-formed and self-contained diffs.\n\nWhen we examined sequences of such diffs, we noticed that developers sometimes jumped back and forth, iterating on the same one or two lines repetitively. We defined filters to throw out such examples since we wanted Instinct to make efficient, non-repetitive edits.\n\nContinue's [autocomplete context pipelines](https://blog.continue.dev/root-path-context-the-secret-ingredient-in-continues-autocomplete-prompt/) provided relevant information about the codebase, with content from the current file included in the prompt as well. We define the editable region (the span of code to be rewritten by the model) as starting one line above the cursor and ending five lines below. This decision was made based on the natural features of the diffs we saw in our dataset.\n\nTaken together, the edit sequence, context, current file content, and editable region provide the necessary information to infer the developer's intentions and predict the next step in the edit sequence.\n\n### Maintaining Multilingual Support\n\nOne problem we ran into was that the Continue team works mostly with Typescript code. However, we wanted our model to preserve support for multiple languages. In addition to training the model in a robust way (more on this later), we used a self-hosted Qwen3-Coder-30B model to synthetically \"translate\" diffs, context, and file contents into Java, C, Python, and Rust—thus bootstrapping a multilingual dataset off of the Typescript data. A set of precise data correctors and filters ensured high quality and a similar distribution of edits among the 4,000+ synthetic examples.\n\n### Robust Training for the Next Edit Task\n\nWith a multilingual dataset ready to go, it was time to move on to supervised fine-tuning (SFT). SFT for specific tasks like Next Edit is typically done using Low-Rank Adaptation (LoRA), since although [it learns less, it also forgets less](https://arxiv.org/abs/2405.09673?ref=blog.continue.dev) of the pre-trained model's general coding abilities.\n\nThe central problem with using LoRA, however, is that it fixes the parameters to be fine-tuned before training even begins. A preferable approach is to _discover_ which parameters' weight updates are most important, and update only those. Rather than treating Next Edit as learning new _knowledge_ at the expense of forgetting previous coding ability, the model can ideally adapt to the _task_ of Next Editing while retaining its pre-trained coding knowledge.\n\nWe found exactly such a solution in the Selective Knowledge Transfer (SeleKT) algorithm used to train the [NextCoder](https://www.microsoft.com/en-us/research/publication/nextcoder-robust-adaptation-of-code-lms-to-diverse-code-edits/?ref=blog.continue.dev) instructed code-editing models. SeleKT computes dense gradients (as in full fine-tuning), extracts the top-_k_ gradients by magnitude, and then applies only those sparse weight updates. It discovers through practice what actually needs to change, rather than guessing beforehand, and as a result, only the most important weights for the Next Edit _task_ are updated. Moreover, zero-ing out small weight updates helps prevent overfitting and erosion of previous coding knowledge, problems that full fine-tuning would suffer from.\n\nWe fine-tuned 5% of [Qwen2.5-Coder-7B](https://arxiv.org/abs/2409.12186?ref=blog.continue.dev)'s parameters using SeleKT. After initial hyperparameter sweeps, the training process was standard, using a log warmup and cosine decay learning rate schedule for 5 epochs. We used the [CodeBLEU](https://arxiv.org/abs/2009.10297?ref=blog.continue.dev) score as a quick proxy eval between predicted and ground truth rewrites during training. Ablating CodeBLEU scores across different languages in the dataset allowed us to adjust the data mixture, resulting in high validation performance across languages. Due to the robust training, only a small number of multilingual examples were needed.\n\n## Performance Results: Evals for Quality and Speed\n\n| 3.877 | 6.4x faster\\* |\n| --- | --- |\n| Average LLM Judge Score (new open-source state-of-the-art) | than manually typing out edits _\\* on our 8xH100 cluster_ |\n\nIt's nontrivial to evaluate the quality of a Next Edit suggestion formally, since there are often many ways to accomplish the same coding goal. Accordingly, we deployed Claude as an LLM judge with instructions to assess the quality of Next Edit suggestions on a five-point scale, where:\n\n- A score of five corresponds to a functional match with the developer's ground-truth edit,\n- A score of four corresponds to a similar edit to the developer's ground truth edit, although not an exact functional match,\n- A score of three corresponds to an edit that does not match the ground truth but would reasonably be made by an expert developer in such a scenario,\n- A score of two corresponds to an edit that does not logically follow from the previous edits and context,\n- A score of one corresponds to an edit that is likely to hinder developer progress, such as large deletions or complete irrelevance, and\n- A score of zero corresponds to a malformed rewrite that does not line up with the editable region.\n\nThis evaluation method is similar to the one Zed uses for Zeta. We noticed that their judge prompt caused the LLM to output scores of only zero or five, and that it relied on handcrafted assertions about the next edit. We created a different system prompt that helps represent the full spectrum of scores and compares the model's edit to the developer's ground truth change instead of relying on assertions. With just minor adjustments to account for our different Next Edit prompt structures, Instinct's average score of 3.877 outperforms Zeta's score of 3.735 on the held-out eval set. We're excited to see further work continuing to improve on this benchmark.\n\nNot only does Instinct provide high-quality suggestions, our keystroke-distance eval, loosely based on that of [Coeditor](https://arxiv.org/abs/2305.18584?ref=blog.continue.dev), demonstrates that it also massively speeds up your workflow. By backtracking through the dynamic programming (DP) table of a Levenshtein distance calculator, it's possible to extract a character-level diff between the editable region and the suggested rewrite. That character-level diff can be \"chunked\" into edit operations, _e.g._, adding three characters at one location, deleting five characters at another location, and such. \n\nThe minimum time to carry out the full edit is given by the optimal combination of keypresses and cursor jumps that accomplish all the edit operations. This can be posed as another DP problem. We assume the developer types at an average of 90 WPM, simulate the ability to highlight and delete as opposed to repeatedly hitting the backspace key, allow for small arrow movements instead of more time-intensive cursor jumps, and use the spatial distance between cursor locations (_i.e._ line and character) instead of just the index within the whole string. This lower bound on manual editing time is compared to average model inference time on our internal SGLang endpoint plus the time for one keypress (tab-to-accept).\n\nThe result is that even if you immediately knew exactly what edit to make, _and_ took the DP-optimal sequence of actions to carry it out at 90 WPM, using the model would still provide the high-quality edit 6.4 times faster. \n\n## Now what?\n\nFirst, we encourage you to try it out! Instinct is a 7B model so you should expect it to be slow on most laptops, but with sufficient hardware it is a great option for self-hosting. Read [our guide](https://docs.continue.dev/guides/instinct?ref=blog.continue.dev) to learn more.\n\nIf you want to build upon our dataset, training pipelines, or open weights (for example to run [KTO](https://arxiv.org/abs/2402.01306?ref=blog.continue.dev) on your own accept/reject data) we'd recommend exploring our [GitHub repository](https://github.com/continuedev/instinct?ref=blog.continue.dev), [HuggingFace model card](https://huggingface.co/continuedev/instinct?ref=blog.continue.dev), and dataset.\n\nMost importantly, if you are interested in furthering the state of the art, either as part of the community or the Continue team, please [reach out](mailto:nate@continue.dev)!\n"])</script><script>self.__next_f.push([1,"6:[\"$\",\"div\",null,{\"className\":\"min-h-screen flex flex-col\",\"children\":[[\"$\",\"script\",null,{\"type\":\"application/ld+json\",\"dangerouslySetInnerHTML\":{\"__html\":\"{\\\"@context\\\":\\\"https://schema.org\\\",\\\"@type\\\":\\\"BlogPosting\\\",\\\"headline\\\":\\\"Introducing Instinct: the world's best open Next Edit model, built by Continue\\\",\\\"description\\\":\\\"Meet Instinct, Continue's open Next Edit model that's built to predict your next edit and keep you in flow\\\",\\\"datePublished\\\":\\\"2025-09-04T10:17:55.000-07:00\\\",\\\"dateModified\\\":\\\"2025-09-04T11:32:20.000-07:00\\\",\\\"image\\\":\\\"/images/blog/instinct-hero.png\\\",\\\"author\\\":[{\\\"@type\\\":\\\"Person\\\",\\\"name\\\":\\\"Adarsh Iyer\\\"},{\\\"@type\\\":\\\"Person\\\",\\\"name\\\":\\\"Nate Sesti\\\",\\\"url\\\":\\\"https://x.com/natesesti\\\"}],\\\"publisher\\\":{\\\"@type\\\":\\\"Organization\\\",\\\"name\\\":\\\"Continue\\\",\\\"url\\\":\\\"https://continue.dev\\\"}}\"}}],[\"$\",\"main\",null,{\"className\":\"flex-1 w-full\",\"children\":[\"$\",\"article\",null,{\"className\":\"max-w-3xl mx-auto px-6 sm:px-8 py-8\",\"children\":[[\"$\",\"$L5\",null,{\"href\":\"/\",\"className\":\"inline-flex items-center text-xs font-mono uppercase tracking-[0.15em] text-black/25 hover:text-black/50 transition-colors mb-8\",\"children\":\"← Back to blog\"}],[\"$\",\"div\",null,{\"className\":\"mb-8 bg-black/[0.015] shadow-[0_2px_20px_rgb(0,0,0,0.04)] overflow-hidden\",\"children\":[\"$\",\"$Lf\",null,{\"src\":\"/images/blog/instinct-hero.png\",\"alt\":\"Introducing Instinct: the world's best open Next Edit model, built by Continue\",\"width\":1200,\"height\":675,\"sizes\":\"100vw\",\"priority\":true,\"className\":\"w-full h-auto\"}]}],[\"$\",\"h1\",null,{\"className\":\"text-[2.5rem] font-light tracking-tight text-black/90 mb-4\",\"children\":\"Introducing Instinct: the world's best open Next Edit model, built by Continue\"}],[\"$\",\"div\",null,{\"className\":\"flex flex-wrap items-center gap-x-4 gap-y-1 mb-8\",\"children\":[[\"$\",\"span\",null,{\"className\":\"text-[15px] text-black/50\",\"children\":[[\"$\",\"span\",\"Adarsh Iyer\",{\"children\":[false,\"Adarsh Iyer\"]}],[\"$\",\"span\",\"Nate Sesti\",{\"children\":[\", \",[\"$\",\"a\",null,{\"href\":\"https://x.com/natesesti\",\"target\":\"_blank\",\"rel\":\"noopener noreferrer\",\"className\":\"hover:text-black/70 transition-colors underline decoration-black/20 underline-offset-2 hover:decoration-black/50\",\"children\":\"Nate Sesti\"}]]}]]}],[\"$\",\"span\",null,{\"className\":\"text-xs font-mono uppercase tracking-[0.15em] text-black/25\",\"children\":[\"September 4, 2025\",[\" · \",7,\" min read\"]]}]]}],[\"$\",\"$L10\",null,{\"content\":\"$11\"}]]}]}]]}]\n"])</script><script>self.__next_f.push([1,"b:[[\"$\",\"meta\",\"0\",{\"charSet\":\"utf-8\"}],[\"$\",\"meta\",\"1\",{\"name\":\"viewport\",\"content\":\"width=device-width, initial-scale=1\"}]]\n"])</script><script>self.__next_f.push([1,"12:I[27201,[\"/_next/static/chunks/ff1a16fafef87110.js\",\"/_next/static/chunks/803a574de9eda7ae.js\"],\"IconMark\"]\n9:null\n"])</script><script>self.__next_f.push([1,"d:[[\"$\",\"title\",\"0\",{\"children\":\"Introducing Instinct: the world's best open Next Edit model, built by Continue | Continue - Blog\"}],[\"$\",\"meta\",\"1\",{\"name\":\"description\",\"content\":\"Meet Instinct, Continue's open Next Edit model that's built to predict your next edit and keep you in flow\"}],[\"$\",\"link\",\"2\",{\"rel\":\"canonical\",\"href\":\"https://blog.continue.dev/instinct/\"}],[\"$\",\"meta\",\"3\",{\"property\":\"og:title\",\"content\":\"Introducing Instinct: the world's best open Next Edit model, built by Continue\"}],[\"$\",\"meta\",\"4\",{\"property\":\"og:description\",\"content\":\"Meet Instinct, Continue's open Next Edit model that's built to predict your next edit and keep you in flow\"}],[\"$\",\"meta\",\"5\",{\"property\":\"og:image\",\"content\":\"https://blog.continue.dev/images/blog/instinct-hero.png\"}],[\"$\",\"meta\",\"6\",{\"property\":\"og:type\",\"content\":\"article\"}],[\"$\",\"meta\",\"7\",{\"property\":\"article:published_time\",\"content\":\"2025-09-04T10:17:55.000-07:00\"}],[\"$\",\"meta\",\"8\",{\"property\":\"article:modified_time\",\"content\":\"2025-09-04T11:32:20.000-07:00\"}],[\"$\",\"meta\",\"9\",{\"property\":\"article:author\",\"content\":\"Adarsh Iyer\"}],[\"$\",\"meta\",\"10\",{\"property\":\"article:author\",\"content\":\"Nate Sesti\"}],[\"$\",\"meta\",\"11\",{\"name\":\"twitter:card\",\"content\":\"summary_large_image\"}],[\"$\",\"meta\",\"12\",{\"name\":\"twitter:title\",\"content\":\"Introducing Instinct: the world's best open Next Edit model, built by Continue\"}],[\"$\",\"meta\",\"13\",{\"name\":\"twitter:description\",\"content\":\"Meet Instinct, Continue's open Next Edit model that's built to predict your next edit and keep you in flow\"}],[\"$\",\"meta\",\"14\",{\"name\":\"twitter:image\",\"content\":\"https://blog.continue.dev/images/blog/instinct-hero.png\"}],[\"$\",\"link\",\"15\",{\"rel\":\"icon\",\"href\":\"/favicon.png\"}],[\"$\",\"$L12\",\"16\",{}]]\n"])</script></body></html>