mirror of
https://github.com/stablyai/orca.git
synced 2026-10-01 00:02:10 +00:00
fix: accept single-quoted attributes in README local link check
The parser only matched double-quoted src/srcset/href, so <img src='missing.gif'> was skipped and the guard passed a README that GitHub renders with a broken image. Regression test fails without the parser change.
This commit is contained in:
@@ -10,7 +10,7 @@ import { pathToFileURL } from 'node:url'
|
||||
// git index rather than the working tree.
|
||||
const TRANSLATED_README_DIR = path.join('docs', 'readme')
|
||||
const EXTERNAL_TARGET = /^(?:[a-z][a-z0-9+.-]*:|#|\/\/)/i
|
||||
const HTML_ATTRIBUTE = /\b(?:src|srcset|href)\s*=\s*"([^"]*)"/g
|
||||
const HTML_ATTRIBUTE = /\b(?:src|srcset|href)\s*=\s*(?:"([^"]*)"|'([^']*)')/g
|
||||
const MARKDOWN_LINK = /!?\[[^\]]*\]\(([^)\s]+)(?:\s+"[^"]*")?\)/g
|
||||
|
||||
function readmeFiles(root) {
|
||||
@@ -38,7 +38,7 @@ function trackedFiles(root, candidates) {
|
||||
function* localTargets(markdown) {
|
||||
for (const match of markdown.matchAll(HTML_ATTRIBUTE)) {
|
||||
// Why: srcset is a candidate list ("a.gif 1x, b.gif 2x"); each entry starts with a URL.
|
||||
for (const candidate of match[1].split(',')) {
|
||||
for (const candidate of (match[1] ?? match[2]).split(',')) {
|
||||
const target = candidate.trim().split(/\s+/)[0]
|
||||
if (target) {
|
||||
yield target
|
||||
|
||||
@@ -39,6 +39,7 @@ const validReadmes = {
|
||||
'<img src="resources/build/icon.png" />',
|
||||
'<picture><source srcset="docs/site/public/docs/tab-split.gif" type="image/gif"><img src="resources/onboarding/feature-wall/tile-01.poster.jpg" /></picture>',
|
||||
'<a href="docs/readme/README.ja.md">日本語</a>',
|
||||
"<img src='resources/build/icon.png' />",
|
||||
'<img src="https://img.shields.io/badge/x-y-z" />',
|
||||
'[Contributing](.github/CONTRIBUTING.md) [Docs](https://example.com/docs) [Top](#top)',
|
||||
''
|
||||
@@ -119,6 +120,23 @@ describe('README local link check', () => {
|
||||
])
|
||||
})
|
||||
|
||||
// Why: a single-quoted attribute is valid HTML and GitHub renders it, so a parser
|
||||
// that only reads double quotes would pass a README with a broken image.
|
||||
it('reports a missing target in a single-quoted attribute', () => {
|
||||
const files = {
|
||||
...validReadmes,
|
||||
'README.md': `${validReadmes['README.md']}\n<img src='docs/assets/missing.gif' />`
|
||||
}
|
||||
|
||||
expect(findBrokenReadmeLinks(makeFixture(files))).toEqual([
|
||||
{
|
||||
readme: 'README.md',
|
||||
target: 'docs/assets/missing.gif',
|
||||
resolved: 'docs/assets/missing.gif'
|
||||
}
|
||||
])
|
||||
})
|
||||
|
||||
// Why the ungated job: static_analysis is skipped for docs-only diffs, which is
|
||||
// exactly the kind of PR that deletes a docs-site GIF the README embeds.
|
||||
it('runs on every PR through the ungated guard job and in the lint script', () => {
|
||||
|
||||
Reference in New Issue
Block a user