From 6246b231d13079b2c62908267c62c424025b05bb Mon Sep 17 00:00:00 2001 From: raulkolaric <155586454+raulkolaric@users.noreply.github.com> Date: Thu, 13 Aug 2026 11:52:49 -0300 Subject: [PATCH] feat(content-linter): warn about consecutive duplicate words --- .../contributing/content-linter-rules.md | 1 + .../consecutive-duplicate-words.ts | 153 ++++++++++++++++++ src/content-linter/lib/linting-rules/index.ts | 2 + src/content-linter/style/github-docs.ts | 6 + .../tests/unit/consecutive-duplicate-words.ts | 135 ++++++++++++++++ 5 files changed, 297 insertions(+) create mode 100644 src/content-linter/lib/linting-rules/consecutive-duplicate-words.ts create mode 100644 src/content-linter/tests/unit/consecutive-duplicate-words.ts diff --git a/data/reusables/contributing/content-linter-rules.md b/data/reusables/contributing/content-linter-rules.md index e9a00f1331ad..74ef7b59d08c 100644 --- a/data/reusables/contributing/content-linter-rules.md +++ b/data/reusables/contributing/content-linter-rules.md @@ -69,6 +69,7 @@ | GHD065 | frontmatter-content-type | Content files in content-type directories must have a contentType frontmatter property that matches the parent directory. | error | frontmatter, content-type | | GHD066 | frontmatter-docs-team-metrics | Articles whose path contains a path-enforced docsTeamMetrics value must include that value in their docsTeamMetrics frontmatter property. | error | frontmatter, docs-team-metrics | | GHD067 | frontmatter-rest-api-category | Autogenerated REST API endpoint files must have a valid `category` frontmatter property | error | frontmatter, rest, category | +| GHD068 | consecutive-duplicate-words | Consecutive words must not be repeated | warning | format | | [search-replace](https://github.com/OnkarRuikar/markdownlint-rule-search-replace) | deprecated liquid syntax: octicon- | The octicon liquid syntax used is deprecated. Use this format instead `octicon "" aria-label=""` | error | | | [search-replace](https://github.com/OnkarRuikar/markdownlint-rule-search-replace) | deprecated liquid syntax: site.data | Catch occurrences of deprecated liquid data syntax. | error | | | [search-replace](https://github.com/OnkarRuikar/markdownlint-rule-search-replace) | developer-domain | Catch occurrences of developer.github.com domain. | error | | diff --git a/src/content-linter/lib/linting-rules/consecutive-duplicate-words.ts b/src/content-linter/lib/linting-rules/consecutive-duplicate-words.ts new file mode 100644 index 000000000000..fe130970a5dc --- /dev/null +++ b/src/content-linter/lib/linting-rules/consecutive-duplicate-words.ts @@ -0,0 +1,153 @@ +import { addError, ellipsify, filterTokens } from 'markdownlint-rule-helpers' + +import type { MarkdownToken, Rule, RuleErrorCallback, RuleParams } from '@/content-linter/types' + +interface MarkdownItToken extends MarkdownToken { + info?: string + map?: [number, number] +} + +interface DuplicateWordMatch { + firstWord: string + secondWord: string + start: number + text: string + whitespaceLength: number +} + +const DUPLICATE_WORD_PATTERN = + /(? 0 && letters === letters.toUpperCase() && letters !== letters.toLowerCase() + ) +} + +function isPlaceholderOrOperatorPair(firstWord: string, secondWord: string): boolean { + return firstWord !== secondWord && (isAllUppercase(firstWord) || isAllUppercase(secondWord)) +} + +function findDuplicateWords(text: string): DuplicateWordMatch[] { + const matches: DuplicateWordMatch[] = [] + + DUPLICATE_WORD_PATTERN.lastIndex = 0 + let match: RegExpExecArray | null + while ((match = DUPLICATE_WORD_PATTERN.exec(text)) !== null) { + const [, firstWord, whitespace, secondWord] = match + if (isPlaceholderOrOperatorPair(firstWord, secondWord)) continue + + matches.push({ + firstWord, + secondWord, + start: match.index, + text: match[0], + whitespaceLength: whitespace.length, + }) + } + + return matches +} + +function reportDuplicateWords( + text: string, + sourceLine: string, + lineNumber: number, + sourceOffset: number, + onError: RuleErrorCallback, + reportedLocations: Set, +): void { + for (const match of findDuplicateWords(text)) { + const expectedMatchOffset = sourceOffset + match.start + const sourceMatchOffset = + sourceLine.slice(expectedMatchOffset, expectedMatchOffset + match.text.length) === match.text + ? expectedMatchOffset + : sourceLine.indexOf(match.text, sourceOffset) + const resolvedMatchOffset = sourceMatchOffset >= 0 ? sourceMatchOffset : expectedMatchOffset + const duplicateOffset = match.start + match.firstWord.length + match.whitespaceLength + const duplicateColumn = resolvedMatchOffset + duplicateOffset - match.start + 1 + const location = `${lineNumber}:${duplicateColumn}` + if (reportedLocations.has(location)) continue + reportedLocations.add(location) + + addError( + onError, + lineNumber, + `Check whether the repeated word "${match.secondWord}" is intentional.`, + ellipsify(sourceLine), + [duplicateColumn, match.secondWord.length], + null, + ) + } +} + +function isSentenceLikeText(line: string): boolean { + return /[.!?]["')\]}]*$/.test(line.trim()) +} + +export const consecutiveDuplicateWords: Rule = { + names: ['GHD068', 'consecutive-duplicate-words'], + description: 'Consecutive words must not be repeated', + tags: ['format'], + parser: 'markdownit', + function: (params: RuleParams, onError: RuleErrorCallback) => { + const reportedLocations = new Set() + + filterTokens(params, 'inline', (token: MarkdownItToken) => { + let currentLineNumber = token.lineNumber || 1 + let sourceCursor = 0 + + for (const child of token.children || []) { + const childLineNumber = child.lineNumber || token.lineNumber || 1 + const sourceLine = child.line || token.line || params.lines[childLineNumber - 1] || '' + + if (childLineNumber !== currentLineNumber) { + currentLineNumber = childLineNumber + sourceCursor = 0 + } + + const childContent = child.content || '' + const childOffset = childContent ? sourceLine.indexOf(childContent, sourceCursor) : -1 + + if (child.type === 'text' && childContent) { + const sourceOffset = childOffset >= 0 ? childOffset : 0 + reportDuplicateWords( + childContent, + sourceLine, + childLineNumber, + sourceOffset, + onError, + reportedLocations, + ) + } + + if (childOffset >= 0) sourceCursor = childOffset + childContent.length + } + }) + + filterTokens(params, 'fence', (token: MarkdownItToken) => { + const language = token.info?.trim().split(/\s+/)[0]?.toLowerCase() + if (language !== 'text' || !token.map) return + + const contentLines = (token.content || '').split('\n') + const firstContentLineIndex = token.map[0] + 1 + + for (const [offset, contentLine] of contentLines.entries()) { + if (!contentLine || !isSentenceLikeText(contentLine)) continue + + const sourceLineIndex = firstContentLineIndex + offset + const sourceLine = params.lines[sourceLineIndex] || contentLine + const sourceOffset = Math.max(0, sourceLine.indexOf(contentLine)) + reportDuplicateWords( + contentLine, + sourceLine, + sourceLineIndex + 1, + sourceOffset, + onError, + reportedLocations, + ) + } + }) + }, +} diff --git a/src/content-linter/lib/linting-rules/index.ts b/src/content-linter/lib/linting-rules/index.ts index 18a630d3cdfb..d1a319faaddd 100644 --- a/src/content-linter/lib/linting-rules/index.ts +++ b/src/content-linter/lib/linting-rules/index.ts @@ -58,6 +58,7 @@ import { raiAppCardStructure } from '@/content-linter/lib/linting-rules/rai-app- import { frontmatterContentType } from '@/content-linter/lib/linting-rules/frontmatter-content-type' import { frontmatterDocsTeamMetrics } from '@/content-linter/lib/linting-rules/frontmatter-docs-team-metrics' import { frontmatterRestApiCategory } from '@/content-linter/lib/linting-rules/frontmatter-rest-api-category' +import { consecutiveDuplicateWords } from '@/content-linter/lib/linting-rules/consecutive-duplicate-words' const noDefaultAltText = markdownlintGitHub.find((elem: { names: string[] }) => elem.names.includes('no-default-alt-text'), @@ -124,6 +125,7 @@ export const gitHubDocsMarkdownlint = { frontmatterContentType, // GHD065 frontmatterDocsTeamMetrics, // GHD066 frontmatterRestApiCategory, // GHD067 + consecutiveDuplicateWords, // GHD068 // Search-replace rules searchReplace, // Open-source plugin diff --git a/src/content-linter/style/github-docs.ts b/src/content-linter/style/github-docs.ts index 2c6a8e89809b..4687a650bb95 100644 --- a/src/content-linter/style/github-docs.ts +++ b/src/content-linter/style/github-docs.ts @@ -192,6 +192,12 @@ const githubDocsConfig = { severity: 'error', 'partial-markdown-files': false, }, + 'consecutive-duplicate-words': { + // GHD068 + severity: 'warning', + 'partial-markdown-files': true, + 'yml-files': true, + }, } export const githubDocsFrontmatterConfig = { diff --git a/src/content-linter/tests/unit/consecutive-duplicate-words.ts b/src/content-linter/tests/unit/consecutive-duplicate-words.ts new file mode 100644 index 000000000000..cb57c86df90e --- /dev/null +++ b/src/content-linter/tests/unit/consecutive-duplicate-words.ts @@ -0,0 +1,135 @@ +import { describe, expect, test } from 'vitest' + +import { runRule } from '@/content-linter/lib/init-test' +import { consecutiveDuplicateWords } from '@/content-linter/lib/linting-rules/consecutive-duplicate-words' + +describe(consecutiveDuplicateWords.names.join(' - '), () => { + test('reports lowercase and case-insensitive duplicate words', async () => { + const markdown = [ + 'You will use use this process.', + 'For more more information, see the guide.', + 'This version is newer than the the published version.', + 'Open the Terminal Chat chat window.', + 'Create an Azure Blob Storage storage account.', + 'The café café is nearby.', + ].join('\n') + + const result = await runRule(consecutiveDuplicateWords, { strings: { markdown } }) + const errors = result.markdown + + expect(errors).toHaveLength(6) + expect(errors.map((error) => error.lineNumber)).toEqual([1, 2, 3, 4, 5, 6]) + expect(errors[0].errorDetail).toBe('Check whether the repeated word "use" is intentional.') + expect(errors[0].errorRange).toEqual([14, 3]) + expect(errors[3].errorRange).toEqual([24, 4]) + expect(errors[4].errorRange).toEqual([30, 7]) + }) + + test('reports duplicates in common Markdown prose constructs', async () => { + const markdown = [ + '# A repeated repeated heading', + '', + '* A duplicate duplicate list item.', + '', + '> A repeated repeated blockquote.', + '', + '**Terminal Chat chat** window.', + '', + '[More more information](https://example.com).', + '', + '| Value | Description |', + '| --- | --- |', + '| Test | A repeated repeated value. |', + ].join('\n') + + const result = await runRule(consecutiveDuplicateWords, { strings: { markdown } }) + const errors = result.markdown + + expect(errors).toHaveLength(6) + expect(errors.map((error) => error.lineNumber)).toEqual([1, 3, 5, 7, 9, 13]) + expect(errors[3].errorRange).toEqual([17, 4]) + expect(errors[4].errorRange).toEqual([7, 4]) + }) + + test('reports every duplicate pair on the same line', async () => { + const markdown = 'This is is wrong, and that that is also wrong.' + + const result = await runRule(consecutiveDuplicateWords, { strings: { markdown } }) + const errors = result.markdown + + expect(errors).toHaveLength(2) + expect(errors.map((error) => error.errorRange)).toEqual([ + [9, 2], + [28, 4], + ]) + }) + + test('reports accurate ranges after encoded text and tabs', async () => { + const markdown = ['An & repeated repeated phrase.', 'This is\tis wrong.'].join('\n') + + const result = await runRule(consecutiveDuplicateWords, { strings: { markdown } }) + const errors = result.markdown + + expect(errors).toHaveLength(2) + expect(errors.map((error) => error.errorRange)).toEqual([ + [19, 8], + [9, 2], + ]) + }) + + test('reports sentence-like prose in text fences', async () => { + const markdown = [ + '```text', + 'Currently, the option is used to to pass a token.', + '', + 'view View sub-issues', + '```', + ].join('\n') + + const result = await runRule(consecutiveDuplicateWords, { strings: { markdown } }) + const errors = result.markdown + + expect(errors).toHaveLength(1) + expect(errors[0].lineNumber).toBe(2) + expect(errors[0].errorRange).toEqual([34, 2]) + }) + + test('ignores code, placeholders, operators, punctuation, and hyphenated words', async () => { + const markdown = [ + 'This is very, very important.', + 'Follow the how-to to complete the setup.', + 'Use logical OR or logical AND.', + 'Replace hostname HOSTNAME in the command.', + 'Click **Delete Tag TAG NAME**.', + 'The API API pair is intentionally tested separately.', + '`use use` is an inline code example.', + '', + '```shell', + 'python -m venv venv', + '```', + '', + 'Button', + ].join('\n') + + const result = await runRule(consecutiveDuplicateWords, { strings: { markdown } }) + const errors = result.markdown + + expect(errors).toHaveLength(1) + expect(errors[0].errorDetail).toBe('Check whether the repeated word "API" is intentional.') + expect(errors[0].lineNumber).toBe(6) + }) + + test('respects Markdownlint suppression comments', async () => { + const markdown = [ + '', + 'The words had had a deliberate meaning.', + 'This is is still an error.', + ].join('\n') + + const result = await runRule(consecutiveDuplicateWords, { strings: { markdown } }) + const errors = result.markdown + + expect(errors).toHaveLength(1) + expect(errors[0].lineNumber).toBe(3) + }) +})