[{"data":1,"prerenderedAt":591},["ShallowReactive",2],{"navigation_docs":3,"-guides-how-detection-works":163,"-guides-how-detection-works-surround":586},[4,50,67,84,120,142,158],{"title":5,"icon":6,"path":7,"stem":8,"children":9,"page":6},"Getting Started",false,"\u002Fgetting-started","1.getting-started",[10,15,20,25,30,35,40,45],{"title":11,"path":12,"stem":13,"icon":14},"Introduction","\u002Fgetting-started\u002Fintroduction","1.getting-started\u002F2.introduction","i-lucide-house",{"title":16,"path":17,"stem":18,"icon":19},"Installation","\u002Fgetting-started\u002Finstallation","1.getting-started\u002F3.installation","i-lucide-download",{"title":21,"path":22,"stem":23,"icon":24},"Configuration","\u002Fgetting-started\u002Fconfiguration","1.getting-started\u002F4.configuration","i-lucide-settings",{"title":26,"path":27,"stem":28,"icon":29},"Supported Formats","\u002Fgetting-started\u002Fsupported-formats","1.getting-started\u002F5.supported-formats","i-lucide-file-code",{"title":31,"path":32,"stem":33,"icon":34},"Agent Skill","\u002Fgetting-started\u002Fagent-skill","1.getting-started\u002F6.agent-skill","i-lucide-cpu",{"title":36,"path":37,"stem":38,"icon":39},"Changelog","\u002Fgetting-started\u002Fchangelog","1.getting-started\u002F7.changelog","i-lucide-scroll-text",{"title":41,"path":42,"stem":43,"icon":44},"Migrating from v4 to v5","\u002Fgetting-started\u002Fmigration","1.getting-started\u002F8.migration","i-lucide-arrow-right-left",{"title":46,"path":47,"stem":48,"icon":49},"jscpd v4 (TypeScript)","\u002Fgetting-started\u002Fv4","1.getting-started\u002F9.v4","i-simple-icons-typescript",{"title":51,"path":52,"stem":53,"children":54,"icon":6},"Guides","\u002Fguides","2.guides\u002F1.index",[55,57,62],{"title":51,"path":52,"stem":53,"icon":56},"i-lucide-book-open",{"title":58,"path":59,"stem":60,"icon":61},"Types of Code Clones","\u002Fguides\u002Fclone-types","2.guides\u002F2.clone-types","i-lucide-copy",{"title":63,"path":64,"stem":65,"icon":66},"How Detection Works","\u002Fguides\u002Fhow-detection-works","2.guides\u002F3.how-detection-works","i-lucide-scan-search",{"title":68,"path":69,"stem":70,"children":71,"icon":6},"CI & Hooks","\u002Fci-and-hooks","3.ci-and-hooks\u002F1.index",[72,74,79],{"title":68,"path":69,"stem":70,"icon":73},"i-lucide-git-pull-request",{"title":75,"path":76,"stem":77,"icon":78},"Use jscpd in CI","\u002Fci-and-hooks\u002Fci","3.ci-and-hooks\u002F2.ci","i-lucide-github",{"title":80,"path":81,"stem":82,"icon":83},"Use jscpd in Pre-Commit Hook","\u002Fci-and-hooks\u002Fpre-commit","3.ci-and-hooks\u002F3.pre-commit","i-lucide-git-commit",{"title":85,"path":86,"stem":87,"children":88,"icon":90},"Reporters","\u002Freporters","4.reporters\u002F1.index",[89,91,95,100,105,110,115],{"title":85,"path":86,"stem":87,"icon":90},"i-lucide-file-chart-column",{"title":92,"path":93,"stem":94,"icon":29},"HTML Reporter","\u002Freporters\u002Fhtml","4.reporters\u002F2.html",{"title":96,"path":97,"stem":98,"icon":99},"JSON Reporter","\u002Freporters\u002Fjson","4.reporters\u002F3.json","i-lucide-braces",{"title":101,"path":102,"stem":103,"icon":104},"Badge Reporter","\u002Freporters\u002Fbadge","4.reporters\u002F4.badge","i-lucide-award",{"title":106,"path":107,"stem":108,"icon":109},"SARIF Reporter","\u002Freporters\u002Fsarif","4.reporters\u002F5.sarif","i-lucide-shield-check",{"title":111,"path":112,"stem":113,"icon":114},"CodeClimate \u002F GitLab Reporter","\u002Freporters\u002Fcodeclimate","4.reporters\u002F6.codeclimate","i-lucide-flag-triangle-right",{"title":116,"path":117,"stem":118,"icon":119},"OpenMetrics Reporter","\u002Freporters\u002Fopenmetrics","4.reporters\u002F7.openmetrics","i-lucide-activity",{"title":121,"path":122,"stem":123,"children":124,"icon":6},"Benchmarks","\u002Fbenchmarks","5.benchmarks\u002F1.index",[125,127,132,137],{"title":121,"path":122,"stem":123,"icon":126},"i-lucide-gauge",{"title":128,"path":129,"stem":130,"icon":131},"Detection Speed","\u002Fbenchmarks\u002Fdetection-speed","5.benchmarks\u002F2.detection-speed","i-lucide-zap",{"title":133,"path":134,"stem":135,"icon":136},"Multi-Format Detection","\u002Fbenchmarks\u002Fcross-format","5.benchmarks\u002F3.cross-format","i-lucide-layers",{"title":138,"path":139,"stem":140,"icon":141},"AI Token Efficiency","\u002Fbenchmarks\u002Fai-token-efficiency","5.benchmarks\u002F4.ai-token-efficiency","i-lucide-sparkles",{"title":143,"path":144,"stem":145,"children":146,"icon":148},"API","\u002Fapi","6.api\u002F1.index",[147,149,154],{"title":143,"path":144,"stem":145,"icon":148},"i-lucide-code",{"title":150,"path":151,"stem":152,"icon":153},"Rust Crates (v5)","\u002Fapi\u002Fcore","6.api\u002F2.core","i-lucide-box",{"title":155,"path":156,"stem":157,"icon":153},"MCP Server","\u002Fapi\u002Fmcp-server","6.api\u002F4.mcp-server",{"title":159,"path":160,"stem":161,"icon":162},"Trending","\u002Ftrending","7.trending","i-lucide-trending-up",{"id":164,"title":63,"body":165,"description":577,"extension":578,"links":579,"meta":580,"navigation":581,"path":64,"seo":582,"stem":65,"__hash__":585},"docs\u002F2.guides\u002F3.how-detection-works.md",{"type":166,"value":167,"toc":562},"minimark",[168,177,182,209,213,216,274,279,289,310,314,317,375,389,393,424,428,436,485,492,496,516,520,528,550,554],[169,170,171,172,176],"p",{},"jscpd is a token-based detector. The important word is ",[173,174,175],"em",{},"token",": a clone is a repeated sequence of language tokens, not a repeated run of characters. The tokens come from each language's own syntax, so what counts as a comment, a string, a keyword or an identifier is decided per language before any matching happens. This page walks through the pipeline.",[178,179,181],"h2",{"id":180},"_1-format-detection","1. Format detection",[169,183,184,185,189,190,194,195,198,199,204,205,208],{},"The file extension picks one of the ",[186,187,188],"a",{"href":27},"224 formats",". Extensionless files such as ",[191,192,193],"code",{},"Makefile"," or ",[191,196,197],{},"Dockerfile"," are mapped with ",[186,200,201],{"href":22},[191,202,203],{},"--formats-names",", and unknown extensions with ",[191,206,207],{},"--formats-exts",". The format selects the tokenizer and, for cross-format groups, the pool of files a clone may span.",[178,210,212],{"id":211},"_2-tokenization","2. Tokenization",[169,214,215],{},"Every format is read with its own lexical rules:",[217,218,219,262,268],"ul",{},[220,221,222,226,227,230,231,234,235,238,239,242,243,246,247,250,251,254,255,257,258,261],"li",{},[223,224,225],"strong",{},"Comments."," ",[191,228,229],{},"\u002F\u002F"," and ",[191,232,233],{},"\u002F* *\u002F"," for the C family, ",[191,236,237],{},"#"," for Python, Ruby, shells, YAML and friends, ",[191,240,241],{},"--"," for SQL, Haskell and Ada, ",[191,244,245],{},"--[[ ]]"," for Lua, ",[191,248,249],{},";"," for Lisp dialects and INI, ",[191,252,253],{},"'"," for Visual Basic, and none for Markdown. A ",[191,256,237],{}," inside a JavaScript string is not a comment, and a ",[191,259,260],{},"\u002F*"," inside Markdown prose does not open one that never closes.",[220,263,264,267],{},[223,265,266],{},"Literals."," String and numeric literals are recognized as single tokens, so a comment marker or a brace inside a string cannot break the stream.",[220,269,270,273],{},[223,271,272],{},"Classification."," Each token is an identifier, a literal, punctuation, a comment or whitespace. That classification is what the normalization options act on later.",[275,276,278],"h3",{"id":277},"javascript-and-typescript","JavaScript and TypeScript",[169,280,281,282,288],{},"JavaScript, TypeScript, JSX and TSX are not handled by the generic rules. They go through the ",[186,283,287],{"href":284,"rel":285},"https:\u002F\u002Foxc.rs",[286],"nofollow","oxc"," parser, the one behind oxlint. Template literals, regular expressions, JSX elements and decorators come out as the tokens the grammar defines. A recoverable parse error, a redeclaration for instance, leaves the token stream intact, so such files still match files that parse cleanly. Only a file the parser gives up on entirely falls back to a word-split tokenizer.",[169,290,291,292,297,298,301,302,305,306,309],{},"With ",[186,293,294],{"href":22},[191,295,296],{},"--cross-formats"," the parser also erases TypeScript-only syntax from the detection stream: type annotations, interfaces, type aliases, generics and access modifiers are removed via the syntax tree, so a ",[191,299,300],{},".ts"," file matches its plain-JavaScript equivalent. Constructs with runtime meaning stay, among them ",[191,303,304],{},"enum",", non-declare ",[191,307,308],{},"namespace"," and parameter properties, because removing them would change what the code does.",[275,311,313],{"id":312},"embedded-languages","Embedded languages",[169,315,316],{},"Some files are several languages at once, and jscpd splits them before tokenizing:",[318,319,320,333],"table",{},[321,322,323],"thead",{},[324,325,326,330],"tr",{},[327,328,329],"th",{},"File",[327,331,332],{},"What is extracted",[334,335,336,359,367],"tbody",{},[324,337,338,342],{},[339,340,341],"td",{},"Vue, Svelte, Astro",[339,343,344,347,348,230,351,354,355,358],{},[191,345,346],{},"\u003Ctemplate>",", ",[191,349,350],{},"\u003Cscript>",[191,352,353],{},"\u003Cstyle>"," blocks, each tokenized as the language its ",[191,356,357],{},"lang"," attribute names; Astro front matter as TypeScript",[324,360,361,364],{},[339,362,363],{},"Markdown",[339,365,366],{},"Fenced code blocks as the language of the fence, prose as Markdown",[324,368,369,372],{},[339,370,371],{},"Razor",[339,373,374],{},"C# code separated from the surrounding HTML",[169,376,377,378,381,382,384,385,388],{},"Clones inside a block are reported with the block's own line range in the original file, and a ",[191,379,380],{},"\u003Cscript lang=\"ts\">"," block matches ",[191,383,300],{}," files. See the ",[186,386,387],{"href":134},"cross-format benchmark"," for what this finds in practice.",[178,390,392],{"id":391},"_3-what-counts","3. What counts",[169,394,395,396,401,402,405,406,409,410,413,414,417,418,423],{},"Not every token takes part in matching. ",[186,397,398],{"href":22},[191,399,400],{},"--mode mild",", the default, drops whitespace tokens; ",[191,403,404],{},"--mode weak"," also drops comments, so duplicated comment blocks no longer count; ",[191,407,408],{},"--mode strict"," keeps every token. ",[191,411,412],{},"jscpd:ignore-start"," \u002F ",[191,415,416],{},"jscpd:ignore-end"," comments exclude a region, and ",[186,419,420],{"href":22},[191,421,422],{},"--ignore-pattern"," regular expressions exclude whatever they match, a license header for example. Skipped tokens leave the stream without shifting the positions reported for the remaining ones.",[178,425,427],{"id":426},"_4-normalization","4. Normalization",[169,429,430,431,435],{},"The classification from step 2 is what makes ",[186,432,434],{"href":433},"\u002Fguides\u002Fclone-types#type-2-renamed-clones","Type-2 clones"," detectable:",[217,437,438,459,465],{},[220,439,440,443,444,230,447,450,451,454,455,458],{},[191,441,442],{},"--ignore-identifiers"," hashes every identifier as the same placeholder, so ",[191,445,446],{},"total(items)",[191,448,449],{},"sum(rows)"," produce the same tokens. Keywords keep their value: the oxc token kinds tell them apart in JavaScript and TypeScript, and a shared keyword table does it for other languages, so ",[191,452,453],{},"return"," never matches ",[191,456,457],{},"retry",".",[220,460,461,464],{},[191,462,463],{},"--ignore-literals"," does the same for string and numeric literals.",[220,466,467,470,471,230,474,477,478,481,482,484],{},[191,468,469],{},"--ignore-annotations"," drops ",[191,472,473],{},"@Name",[191,475,476],{},"@Name(...)"," sequences, but only in languages where ",[191,479,480],{},"@"," introduces an annotation or decorator: Java, Kotlin, Scala, Groovy, Python, Dart, Swift, JavaScript and TypeScript. In Ruby, Perl, T-SQL, Razor and CSS, ",[191,483,480],{}," means a variable or a directive and is left alone.",[169,486,487,488,491],{},"Clones found this way are reported as ",[191,489,490],{},"renamed",", with the exact-match fingerprint kept separately so the default report is unchanged.",[178,493,495],{"id":494},"_5-matching","5. Matching",[169,497,498,499,504,505,508,509,512,513,515],{},"A rolling ",[186,500,503],{"href":501,"rel":502},"https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FRabin%E2%80%93Karp_algorithm",[286],"Rabin-Karp"," hash slides over the token stream and finds every repeated window of at least ",[191,506,507],{},"--min-tokens"," tokens spanning at least ",[191,510,511],{},"--min-lines"," lines. Files of one format share a pool, and ",[191,514,296],{}," groups merge pools, which is how a Vue script block can match a TypeScript file. Detection runs in parallel across pools.",[178,517,519],{"id":518},"_6-near-miss-passes","6. Near-miss passes",[169,521,522,523,527],{},"Two opt-in passes reach into ",[186,524,526],{"href":525},"\u002Fguides\u002Fclone-types#type-3-near-miss-clones","Type-3 territory",":",[217,529,530,544],{},[220,531,532,535,536,539,540,543],{},[191,533,534],{},"--max-gap-lines N"," merges clone pieces of the same file pair that are separated by at most ",[191,537,538],{},"N"," unmatched lines into one ",[191,541,542],{},"similar"," clone with a similarity score, which catches a copy where a few lines were inserted or changed.",[220,545,546,549],{},[191,547,548],{},"--similarity RATIO"," extracts every JavaScript and TypeScript function from the syntax tree and compares the sequence of node types inside it. Names, literal values and small edits do not change the sequence much, so two functions with the same shape match even when the token stream does not.",[178,551,553],{"id":552},"what-it-does-not-do","What it does not do",[169,555,556,557,561],{},"jscpd does not analyze what the code means. Two functions that compute the same result with different statements, ",[186,558,560],{"href":559},"\u002Fguides\u002Fclone-types#type-4-semantic-clones","Type-4 clones",", are out of scope, as they are for every token-based detector. What the language-aware tokenization buys is precision within that scope: fewer false positives from comments and strings, correct handling of embedded and typed code, and normalization that knows a keyword from a name.",{"title":563,"searchDepth":564,"depth":564,"links":565},"",2,[566,567,572,573,574,575,576],{"id":180,"depth":564,"text":181},{"id":211,"depth":564,"text":212,"children":568},[569,571],{"id":277,"depth":570,"text":278},3,{"id":312,"depth":570,"text":313},{"id":391,"depth":564,"text":392},{"id":426,"depth":564,"text":427},{"id":494,"depth":564,"text":495},{"id":518,"depth":564,"text":519},{"id":552,"depth":564,"text":553},"How jscpd turns source code into language tokens before it looks for duplicates, and what that means for the clones you get.","md",null,{},{"icon":66},{"title":583,"description":584},"How jscpd Detects Duplicate Code","jscpd reads each language by its own syntax. Per-format tokenization, the oxc parser for JavaScript and TypeScript, embedded-language extraction, normalization and Rabin-Karp matching, stage by stage.","xRkPsiVjybD8qZG0EUOSX2g2ATmHn2Z8Y95WNqqkA7g",[587,589],{"title":58,"path":59,"stem":60,"description":588,"icon":61,"children":-1},"Exact, renamed, near-miss and semantic clones, and which of them jscpd finds.",{"title":68,"path":69,"stem":70,"description":590,"icon":73,"children":-1},"Integrate jscpd into your CI pipeline and git hooks.",1789111943339]