Strip Al Jazeera's Recommended Stories section from readable articles
Al Jazeera's article pages embed a <section class="more-on"> block (a 'Recommended Stories' heading plus a list of unrelated article teasers) that was getting pulled into the parsed content by Readability. Remove it from the fetched DOM before Readability runs, same as the existing social-bar/abo/teaser stripping in getReadable().
This commit is contained in:
@@ -934,4 +934,55 @@ describe('useFeeds', () => {
|
||||
expect(feeds.value[0].content).not.toContain('%7Bsize%7D')
|
||||
expect(feeds.value[0].content).not.toContain('<img')
|
||||
})
|
||||
|
||||
it('strips Al-Jazeera-style "Recommended Stories" sections', async () => {
|
||||
feeds.value = [{
|
||||
id: 1,
|
||||
title: 'Article one',
|
||||
url: 'https://www.aljazeera.com/news/2026/9/14/article-one',
|
||||
content: '',
|
||||
}]
|
||||
axios.post.mockResolvedValueOnce({
|
||||
data: {
|
||||
content: `<html><body><article>
|
||||
<p>some article text long enough for readability to keep the paragraph as the main content body, padded with extra words to pass the content-length heuristics used by Mozilla Readability when scoring candidate nodes.</p>
|
||||
<section class="more-on"><h2 class="more-on__heading">Recommended Stories<!-- --> </h2><span>list of 3 items</span><ul>
|
||||
<li><span>list 1 of 3</span><a href="https://www.aljazeera.com/a">Unrelated story one</a></li>
|
||||
<li><span>list 2 of 3</span><a href="https://www.aljazeera.com/b">Unrelated story two</a></li>
|
||||
<li><span>list 3 of 3</span><a href="https://www.aljazeera.com/c">Unrelated story three</a></li>
|
||||
</ul><span>end of list</span></section>
|
||||
<p>more article text long enough to survive readability's content-length heuristics as well, since it needs to look like part of the main body too.</p>
|
||||
</article></body></html>`,
|
||||
},
|
||||
})
|
||||
|
||||
await getReadable(feeds.value[0], 0)
|
||||
|
||||
expect(feeds.value[0].readable).toBe(true)
|
||||
expect(feeds.value[0].content).toContain('main content body')
|
||||
expect(feeds.value[0].content).not.toContain('Recommended Stories')
|
||||
expect(feeds.value[0].content).not.toContain('Unrelated story')
|
||||
})
|
||||
|
||||
it('keeps other sections that are not the "Recommended Stories" block', async () => {
|
||||
feeds.value = [{
|
||||
id: 1,
|
||||
title: 'Article one',
|
||||
url: 'https://www.aljazeera.com/news/2026/9/14/article-one',
|
||||
content: '',
|
||||
}]
|
||||
axios.post.mockResolvedValueOnce({
|
||||
data: {
|
||||
content: `<html><body><article>
|
||||
<section><p>some article text long enough for readability to keep the paragraph as the main content body, padded with extra words to pass the content-length heuristics used by Mozilla Readability when scoring candidate nodes.</p></section>
|
||||
<p>more article text long enough to survive readability's content-length heuristics as well, since it needs to look like part of the main body too.</p>
|
||||
</article></body></html>`,
|
||||
},
|
||||
})
|
||||
|
||||
await getReadable(feeds.value[0], 0)
|
||||
|
||||
expect(feeds.value[0].readable).toBe(true)
|
||||
expect(feeds.value[0].content).toContain('main content body')
|
||||
})
|
||||
})
|
||||
|
||||
@@ -271,6 +271,11 @@ async function getReadable(feed, index) {
|
||||
const container = el.closest('section') ?? el.closest('article')
|
||||
if (container) container.remove()
|
||||
})
|
||||
// Al Jazeera embeds a "Recommended Stories" section (a heading followed by
|
||||
// a list of unrelated article teasers), marked up as <section class="more-on">.
|
||||
// It's not part of the article, so strip it before Readability pulls it
|
||||
// into the parsed content.
|
||||
doc.querySelectorAll('section.more-on').forEach(el => el.remove())
|
||||
const article = new Readability(doc).parse();
|
||||
if (!article) {
|
||||
showMessageForXSeconds('Could not extract readable content.', 5)
|
||||
|
||||
Reference in New Issue
Block a user