Strip Al Jazeera's Recommended Stories section from readable articles
Al Jazeera's article pages embed a <section class="more-on"> block (a 'Recommended Stories' heading plus a list of unrelated article teasers) that was getting pulled into the parsed content by Readability. Remove it from the fetched DOM before Readability runs, same as the existing social-bar/abo/teaser stripping in getReadable().
This commit is contained in:
@@ -934,4 +934,55 @@ describe('useFeeds', () => {
|
|||||||
expect(feeds.value[0].content).not.toContain('%7Bsize%7D')
|
expect(feeds.value[0].content).not.toContain('%7Bsize%7D')
|
||||||
expect(feeds.value[0].content).not.toContain('<img')
|
expect(feeds.value[0].content).not.toContain('<img')
|
||||||
})
|
})
|
||||||
|
|
||||||
|
it('strips Al-Jazeera-style "Recommended Stories" sections', async () => {
|
||||||
|
feeds.value = [{
|
||||||
|
id: 1,
|
||||||
|
title: 'Article one',
|
||||||
|
url: 'https://www.aljazeera.com/news/2026/9/14/article-one',
|
||||||
|
content: '',
|
||||||
|
}]
|
||||||
|
axios.post.mockResolvedValueOnce({
|
||||||
|
data: {
|
||||||
|
content: `<html><body><article>
|
||||||
|
<p>some article text long enough for readability to keep the paragraph as the main content body, padded with extra words to pass the content-length heuristics used by Mozilla Readability when scoring candidate nodes.</p>
|
||||||
|
<section class="more-on"><h2 class="more-on__heading">Recommended Stories<!-- --> </h2><span>list of 3 items</span><ul>
|
||||||
|
<li><span>list 1 of 3</span><a href="https://www.aljazeera.com/a">Unrelated story one</a></li>
|
||||||
|
<li><span>list 2 of 3</span><a href="https://www.aljazeera.com/b">Unrelated story two</a></li>
|
||||||
|
<li><span>list 3 of 3</span><a href="https://www.aljazeera.com/c">Unrelated story three</a></li>
|
||||||
|
</ul><span>end of list</span></section>
|
||||||
|
<p>more article text long enough to survive readability's content-length heuristics as well, since it needs to look like part of the main body too.</p>
|
||||||
|
</article></body></html>`,
|
||||||
|
},
|
||||||
|
})
|
||||||
|
|
||||||
|
await getReadable(feeds.value[0], 0)
|
||||||
|
|
||||||
|
expect(feeds.value[0].readable).toBe(true)
|
||||||
|
expect(feeds.value[0].content).toContain('main content body')
|
||||||
|
expect(feeds.value[0].content).not.toContain('Recommended Stories')
|
||||||
|
expect(feeds.value[0].content).not.toContain('Unrelated story')
|
||||||
|
})
|
||||||
|
|
||||||
|
it('keeps other sections that are not the "Recommended Stories" block', async () => {
|
||||||
|
feeds.value = [{
|
||||||
|
id: 1,
|
||||||
|
title: 'Article one',
|
||||||
|
url: 'https://www.aljazeera.com/news/2026/9/14/article-one',
|
||||||
|
content: '',
|
||||||
|
}]
|
||||||
|
axios.post.mockResolvedValueOnce({
|
||||||
|
data: {
|
||||||
|
content: `<html><body><article>
|
||||||
|
<section><p>some article text long enough for readability to keep the paragraph as the main content body, padded with extra words to pass the content-length heuristics used by Mozilla Readability when scoring candidate nodes.</p></section>
|
||||||
|
<p>more article text long enough to survive readability's content-length heuristics as well, since it needs to look like part of the main body too.</p>
|
||||||
|
</article></body></html>`,
|
||||||
|
},
|
||||||
|
})
|
||||||
|
|
||||||
|
await getReadable(feeds.value[0], 0)
|
||||||
|
|
||||||
|
expect(feeds.value[0].readable).toBe(true)
|
||||||
|
expect(feeds.value[0].content).toContain('main content body')
|
||||||
|
})
|
||||||
})
|
})
|
||||||
|
|||||||
@@ -271,6 +271,11 @@ async function getReadable(feed, index) {
|
|||||||
const container = el.closest('section') ?? el.closest('article')
|
const container = el.closest('section') ?? el.closest('article')
|
||||||
if (container) container.remove()
|
if (container) container.remove()
|
||||||
})
|
})
|
||||||
|
// Al Jazeera embeds a "Recommended Stories" section (a heading followed by
|
||||||
|
// a list of unrelated article teasers), marked up as <section class="more-on">.
|
||||||
|
// It's not part of the article, so strip it before Readability pulls it
|
||||||
|
// into the parsed content.
|
||||||
|
doc.querySelectorAll('section.more-on').forEach(el => el.remove())
|
||||||
const article = new Readability(doc).parse();
|
const article = new Readability(doc).parse();
|
||||||
if (!article) {
|
if (!article) {
|
||||||
showMessageForXSeconds('Could not extract readable content.', 5)
|
showMessageForXSeconds('Could not extract readable content.', 5)
|
||||||
|
|||||||
Reference in New Issue
Block a user