Strip Al Jazeera's Recommended Stories section from readable articles

Al Jazeera's article pages embed a <section class="more-on"> block (a
'Recommended Stories' heading plus a list of unrelated article teasers)
that was getting pulled into the parsed content by Readability. Remove
it from the fetched DOM before Readability runs, same as the existing
social-bar/abo/teaser stripping in getReadable().
This commit is contained in:
2026-09-14 18:40:34 +02:00
parent c584ac4f60
commit a7cfd267c6
2 changed files with 56 additions and 0 deletions
@@ -934,4 +934,55 @@ describe('useFeeds', () => {
expect(feeds.value[0].content).not.toContain('%7Bsize%7D')
expect(feeds.value[0].content).not.toContain('<img')
})
it('strips Al-Jazeera-style "Recommended Stories" sections', async () => {
feeds.value = [{
id: 1,
title: 'Article one',
url: 'https://www.aljazeera.com/news/2026/9/14/article-one',
content: '',
}]
axios.post.mockResolvedValueOnce({
data: {
content: `<html><body><article>
<p>some article text long enough for readability to keep the paragraph as the main content body, padded with extra words to pass the content-length heuristics used by Mozilla Readability when scoring candidate nodes.</p>
<section class="more-on"><h2 class="more-on__heading">Recommended Stories<!-- --> </h2><span>list of 3 items</span><ul>
<li><span>list 1 of 3</span><a href="https://www.aljazeera.com/a">Unrelated story one</a></li>
<li><span>list 2 of 3</span><a href="https://www.aljazeera.com/b">Unrelated story two</a></li>
<li><span>list 3 of 3</span><a href="https://www.aljazeera.com/c">Unrelated story three</a></li>
</ul><span>end of list</span></section>
<p>more article text long enough to survive readability's content-length heuristics as well, since it needs to look like part of the main body too.</p>
</article></body></html>`,
},
})
await getReadable(feeds.value[0], 0)
expect(feeds.value[0].readable).toBe(true)
expect(feeds.value[0].content).toContain('main content body')
expect(feeds.value[0].content).not.toContain('Recommended Stories')
expect(feeds.value[0].content).not.toContain('Unrelated story')
})
it('keeps other sections that are not the "Recommended Stories" block', async () => {
feeds.value = [{
id: 1,
title: 'Article one',
url: 'https://www.aljazeera.com/news/2026/9/14/article-one',
content: '',
}]
axios.post.mockResolvedValueOnce({
data: {
content: `<html><body><article>
<section><p>some article text long enough for readability to keep the paragraph as the main content body, padded with extra words to pass the content-length heuristics used by Mozilla Readability when scoring candidate nodes.</p></section>
<p>more article text long enough to survive readability's content-length heuristics as well, since it needs to look like part of the main body too.</p>
</article></body></html>`,
},
})
await getReadable(feeds.value[0], 0)
expect(feeds.value[0].readable).toBe(true)
expect(feeds.value[0].content).toContain('main content body')
})
})
+5
View File
@@ -271,6 +271,11 @@ async function getReadable(feed, index) {
const container = el.closest('section') ?? el.closest('article')
if (container) container.remove()
})
// Al Jazeera embeds a "Recommended Stories" section (a heading followed by
// a list of unrelated article teasers), marked up as <section class="more-on">.
// It's not part of the article, so strip it before Readability pulls it
// into the parsed content.
doc.querySelectorAll('section.more-on').forEach(el => el.remove())
const article = new Readability(doc).parse();
if (!article) {
showMessageForXSeconds('Could not extract readable content.', 5)