Files
synapse/tests/media/test_html_preview.py
T
4f5524043c Use beautifulsoup4 instead of lxml for URL previews (#19301)
Use `beautifulsoup4` instead of `lxml` for URL previews. This offers
some nicer APIs when parsing HTML and avoids using `libxml`, [which is
unmaintained](https://gitlab.gnome.org/GNOME/libxml2/-/commit/9c80a89af2fdf4f853892f84e46580f4902658ba).

I haven’t done a full regression against commonly previewed sites, but I
expect this will give similar (or better) results.

beautiulsoup also handles decoding the charset for us, which is less
custom code.

---------

Co-authored-by: Andrew Morgan <andrew@amorgan.xyz>
Co-authored-by: Andrew Morgan <1342360+anoadragon453@users.noreply.github.com>
2026-10-02 15:46:34 +01:00

589 lines
22 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#
# This file is licensed under the Affero General Public License (AGPL) version 3.
#
# Copyright 2014-2016 OpenMarket Ltd
# Copyright (C) 2023 New Vector, Ltd
#
# This program is free software: you can redistribute it and/or modify
# it under the terms of the GNU Affero General Public License as
# published by the Free Software Foundation, either version 3 of the
# License, or (at your option) any later version.
#
# See the GNU Affero General Public License for more details:
# <https://www.gnu.org/licenses/agpl-3.0.html>.
#
# Originally licensed under the Apache License, Version 2.0:
# <http://www.apache.org/licenses/LICENSE-2.0>.
#
# [This file includes modifications made by New Vector Limited]
#
#
from unittest.mock import patch
from synapse.media.preview_html import (
decode_body,
parse_html_to_open_graph,
summarize_paragraphs,
)
from tests import unittest
try:
import bs4
except ImportError:
bs4 = None # type: ignore[assignment]
class SummarizeTestCase(unittest.TestCase):
if not bs4:
skip = "url preview feature requires beautifulsoup4"
def test_long_summarize(self) -> None:
example_paras = [
"""Tromsø (Norwegian pronunciation: [ˈtrʊmsœ] ( listen); Northern Sami:
Romsa; Finnish: Tromssa[2] Kven: Tromssa) is a city and municipality in
Troms county, Norway. The administrative centre of the municipality is
the city of Tromsø. Outside of Norway, Tromso and Tromsö are
alternative spellings of the city.Tromsø is considered the northernmost
city in the world with a population above 50,000. The most populous town
north of it is Alta, Norway, with a population of 14,272 (2013).""",
"""Tromsø lies in Northern Norway. The municipality has a population of
(2015) 72,066, but with an annual influx of students it has over 75,000
most of the year. It is the largest urban area in Northern Norway and the
third largest north of the Arctic Circle (following Murmansk and Norilsk).
Most of Tromsø, including the city centre, is located on the island of
Tromsøya, 350 kilometres (217 mi) north of the Arctic Circle. In 2012,
Tromsøya had a population of 36,088. Substantial parts of the urban area
are also situated on the mainland to the east, and on parts of Kvaløya—a
large island to the west. Tromsøya is connected to the mainland by the Tromsø
Bridge and the Tromsøysund Tunnel, and to the island of Kvaløya by the
Sandnessund Bridge. Tromsø Airport connects the city to many destinations
in Europe. The city is warmer than most other places located on the same
latitude, due to the warming effect of the Gulf Stream.""",
"""The city centre of Tromsø contains the highest number of old wooden
houses in Northern Norway, the oldest house dating from 1789. The Arctic
Cathedral, a modern church from 1965, is probably the most famous landmark
in Tromsø. The city is a cultural centre for its region, with several
festivals taking place in the summer. Some of Norway's best-known
musicians, Torbjørn Brundtland and Svein Berge of the electronica duo
Röyksopp and Lene Marlin grew up and started their careers in Tromsø.
Noted electronic musician Geir Jenssen also hails from Tromsø.""",
]
desc = summarize_paragraphs(example_paras, min_size=200, max_size=500)
self.assertEqual(
desc,
"Tromsø (Norwegian pronunciation: [ˈtrʊmsœ] ( listen); Northern Sami:"
" Romsa; Finnish: Tromssa[2] Kven: Tromssa) is a city and municipality in"
" Troms county, Norway. The administrative centre of the municipality is"
" the city of Tromsø. Outside of Norway, Tromso and Tromsö are"
" alternative spellings of the city.Tromsø is considered the northernmost"
" city in the world with a population above 50,000. The most populous town"
" north of it is Alta, Norway, with a population of 14,272 (2013).",
)
desc = summarize_paragraphs(example_paras[1:], min_size=200, max_size=500)
self.assertEqual(
desc,
"Tromsø lies in Northern Norway. The municipality has a population of"
" (2015) 72,066, but with an annual influx of students it has over 75,000"
" most of the year. It is the largest urban area in Northern Norway and the"
" third largest north of the Arctic Circle (following Murmansk and Norilsk)."
" Most of Tromsø, including the city centre, is located on the island of"
" Tromsøya, 350 kilometres (217 mi) north of the Arctic Circle. In 2012,"
" Tromsøya had a population of 36,088. Substantial parts of the urban…",
)
def test_short_summarize(self) -> None:
example_paras = [
"Tromsø (Norwegian pronunciation: [ˈtrʊmsœ] ( listen); Northern Sami:"
" Romsa; Finnish: Tromssa[2] Kven: Tromssa) is a city and municipality in"
" Troms county, Norway.",
"Tromsø lies in Northern Norway. The municipality has a population of"
" (2015) 72,066, but with an annual influx of students it has over 75,000"
" most of the year.",
"The city centre of Tromsø contains the highest number of old wooden"
" houses in Northern Norway, the oldest house dating from 1789. The Arctic"
" Cathedral, a modern church from 1965, is probably the most famous landmark"
" in Tromsø.",
]
desc = summarize_paragraphs(example_paras, min_size=200, max_size=500)
self.assertEqual(
desc,
"Tromsø (Norwegian pronunciation: [ˈtrʊmsœ] ( listen); Northern Sami:"
" Romsa; Finnish: Tromssa[2] Kven: Tromssa) is a city and municipality in"
" Troms county, Norway.\n"
"\n"
"Tromsø lies in Northern Norway. The municipality has a population of"
" (2015) 72,066, but with an annual influx of students it has over 75,000"
" most of the year.",
)
def test_small_then_large_summarize(self) -> None:
example_paras = [
"Tromsø (Norwegian pronunciation: [ˈtrʊmsœ] ( listen); Northern Sami:"
" Romsa; Finnish: Tromssa[2] Kven: Tromssa) is a city and municipality in"
" Troms county, Norway.",
"Tromsø lies in Northern Norway. The municipality has a population of"
" (2015) 72,066, but with an annual influx of students it has over 75,000"
" most of the year."
" The city centre of Tromsø contains the highest number of old wooden"
" houses in Northern Norway, the oldest house dating from 1789. The Arctic"
" Cathedral, a modern church from 1965, is probably the most famous landmark"
" in Tromsø.",
]
desc = summarize_paragraphs(example_paras, min_size=200, max_size=500)
self.assertEqual(
desc,
"Tromsø (Norwegian pronunciation: [ˈtrʊmsœ] ( listen); Northern Sami:"
" Romsa; Finnish: Tromssa[2] Kven: Tromssa) is a city and municipality in"
" Troms county, Norway.\n"
"\n"
"Tromsø lies in Northern Norway. The municipality has a population of"
" (2015) 72,066, but with an annual influx of students it has over 75,000"
" most of the year. The city centre of Tromsø contains the highest number"
" of old wooden houses in Northern Norway, the oldest house dating from"
" 1789. The Arctic Cathedral, a modern church from…",
)
class OpenGraphFromHtmlTestCase(unittest.TestCase):
if not bs4:
skip = "url preview feature requires beautifulsoup4"
def test_simple(self) -> None:
html = b"""
<html>
<head><title>Foo</title></head>
<body>
Some text.
</body>
</html>
"""
soup = decode_body(html, "http://example.com/test.html")
assert soup is not None
og = parse_html_to_open_graph(soup)
self.assertEqual(og, {"og:title": "Foo", "og:description": "Some text."})
def test_comment(self) -> None:
html = b"""
<html>
<head><title>Foo</title></head>
<body>
<!-- HTML comment -->
Some text.
</body>
</html>
"""
soup = decode_body(html, "http://example.com/test.html")
assert soup is not None
og = parse_html_to_open_graph(soup)
self.assertEqual(og, {"og:title": "Foo", "og:description": "Some text."})
def test_comment2(self) -> None:
html = b"""
<html>
<head><title>Foo</title></head>
<body>
Some text.
<!-- HTML comment -->
Some more text.
<p>Text</p>
More text
</body>
</html>
"""
soup = decode_body(html, "http://example.com/test.html")
assert soup is not None
og = parse_html_to_open_graph(soup)
self.assertEqual(
og,
{
"og:title": "Foo",
"og:description": "Some text.\n\nSome more text.\n\nText\n\nMore text",
},
)
def test_script(self) -> None:
html = b"""
<html>
<head><title>Foo</title></head>
<body>
<script> (function() {})() </script>
Some text.
</body>
</html>
"""
soup = decode_body(html, "http://example.com/test.html")
assert soup is not None
og = parse_html_to_open_graph(soup)
self.assertEqual(og, {"og:title": "Foo", "og:description": "Some text."})
def test_missing_title(self) -> None:
html = b"""
<html>
<body>
Some text.
</body>
</html>
"""
soup = decode_body(html, "http://example.com/test.html")
assert soup is not None
og = parse_html_to_open_graph(soup)
self.assertEqual(og, {"og:title": None, "og:description": "Some text."})
# Another variant is a title with no content.
html = b"""
<html>
<head><title></title></head>
<body>
<h1>Title</h1>
</body>
</html>
"""
tree = decode_body(html, "http://example.com/test.html")
assert tree is not None
og = parse_html_to_open_graph(tree)
self.assertEqual(og, {"og:title": "Title", "og:description": "Title"})
def test_h1_as_title(self) -> None:
html = b"""
<html>
<meta property="og:description" content="Some text."/>
<body>
<h1>Title</h1>
</body>
</html>
"""
soup = decode_body(html, "http://example.com/test.html")
assert soup is not None
og = parse_html_to_open_graph(soup)
self.assertEqual(og, {"og:title": "Title", "og:description": "Some text."})
def test_empty_description(self) -> None:
"""Description tags with empty content should be ignored."""
html = b"""
<html>
<meta property="og:description" content=""/>
<meta property="og:description"/>
<meta name="description" content=""/>
<meta name="description"/>
<meta name="description" content="Finally!"/>
<body>
<h1>Title</h1>
</body>
</html>
"""
tree = decode_body(html, "http://example.com/test.html")
assert tree is not None
og = parse_html_to_open_graph(tree)
self.assertEqual(og, {"og:title": "Title", "og:description": "Finally!"})
def test_missing_title_and_broken_h1(self) -> None:
html = b"""
<html>
<body>
<h1><a href="foo"/></h1>
Some text.
</body>
</html>
"""
soup = decode_body(html, "http://example.com/test.html")
assert soup is not None
og = parse_html_to_open_graph(soup)
self.assertEqual(og, {"og:title": None, "og:description": "Some text."})
def test_empty(self) -> None:
"""Test a body with no data in it."""
html = b""
soup = decode_body(html, "http://example.com/test.html")
self.assertIsNone(soup)
def test_no_soup(self) -> None:
"""A valid body with no tree in it."""
with patch(
"bs4.BeautifulSoup",
side_effect=bs4.ParserRejectedMarkup("Invalid markup"),
):
self.assertIsNone(decode_body(b"<html></html>", "https://example.com/"))
def test_xml(self) -> None:
"""Test decoding XML and ensure it works properly."""
# Note that the strip() call is important to ensure the xml tag starts
# at the initial byte.
html = b"""
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN" "http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head><title>Foo</title></head><body>Some text.</body></html>
""".strip()
soup = decode_body(html, "http://example.com/test.html")
assert soup is not None
og = parse_html_to_open_graph(soup)
self.assertEqual(og, {"og:title": "Foo", "og:description": "Some text."})
def test_invalid_encoding2(self) -> None:
"""A body which doesn't match the sent character encoding."""
# Note that this contains an invalid UTF-8 sequence in the title.
html = b"""
<html>
<head><title>\xff\xff Foo</title></head>
<body>
Some text.
</body>
</html>
"""
soup = decode_body(html, "http://example.com/test.html")
assert soup is not None
og = parse_html_to_open_graph(soup)
self.assertEqual(og, {"og:title": "˙˙ Foo", "og:description": "Some text."})
def test_windows_1252(self) -> None:
"""A body which uses cp1252, but doesn't declare that."""
html = b"""
<html>
<head><title>\xf3</title></head>
<body>
Some text.
</body>
</html>
"""
soup = decode_body(html, "http://example.com/test.html")
assert soup is not None
og = parse_html_to_open_graph(soup)
self.assertIn("og:title", og)
og.pop("og:title")
self.assertEqual(og, {"og:description": "Some text."})
def test_image(self) -> None:
"""Test that ensures an image can be pulled from the HTML."""
# Tags is a list of two-element tuples: the HTML tag and the expected image which
# is chosen.
#
# They're in a particular order such that we can prove "higher" priority tags
# are parsed out first, e.g. OpenGraph, then meta tags, then images of a certain
# height/width, favicons, etc.
tags = [
(
b"""<meta property="og:image" content="https://example.com/meta-prop.png">""",
"meta-prop",
),
(
b"""<meta itemprop="IMAGE" content="https://example.com/meta-IMAGE.png">""",
"meta-IMAGE",
),
(
b"""<meta itemprop="image" content="https://example.com/meta-image.png">""",
"meta-image",
),
(b"""<img src="https://example.com/img-no-width-no-height.png">""", "img"),
(
b"""<img src="https://example.com/img-no-height.png" width="100">""",
"img",
),
(
b"""<img src="https://example.com/img-no-width.png" height="100">""",
"img",
),
(
b"""<img src="https://example.com/img-small.png" width="100" height="100">""",
"img",
),
(
b"""<img src="https://example.com/img.png" width="200" height="100">""",
"img",
),
# Put this image again since if it is the *only* image it will be used.
(
b"""<img src="https://example.com/img-no-width-no-height.png">""",
"img-no-width-no-height",
),
(
b"""<link rel="icon" href="https://example.com/favicon.png">""",
"favicon",
),
]
while tags:
html = b"<html>" + b"".join(t[0] for t in tags) + b"</html>"
tree = decode_body(html, "http://example.com/test.html")
assert tree is not None
og = parse_html_to_open_graph(tree)
self.assertEqual(
og,
{
"og:title": None,
"og:description": None,
"og:image": f"https://example.com/{tags[0][1]}.png",
},
)
# Remove the highest remaining priority item.
tags.pop(0)
def test_image_bad_height_width(self) -> None:
"""A bad height/width should be ignored."""
html = b"""<html>
<img src="http://example.com/no-height-width.png">
<img src="http://example.com/bad-height-width.png" height="a" width="a">
</html>"""
tree = decode_body(html, "http://example.com/test.html")
assert tree is not None
og = parse_html_to_open_graph(tree)
self.assertEqual(
og,
{
"og:title": None,
"og:description": None,
"og:image": "http://example.com/no-height-width.png",
},
)
def test_twitter_tag(self) -> None:
"""Twitter card tags should be used if nothing else is available."""
html = b"""
<html>
<meta name="twitter:card" content="summary">
<meta name="twitter:description" content="Description">
<meta name="twitter:site" content="@matrixdotorg">
<meta name="twitter:image" content="https://example.com/test.png">
</html>
"""
tree = decode_body(html, "http://example.com/test.html")
assert tree is not None
og = parse_html_to_open_graph(tree)
self.assertEqual(
og,
{
"og:title": None,
"og:description": "Description",
"og:site_name": "@matrixdotorg",
"og:image": "https://example.com/test.png",
},
)
# But they shouldn't override Open Graph values.
html = b"""
<html>
<meta name="twitter:card" content="summary">
<meta name="twitter:description" content="Description">
<meta property="og:description" content="Real Description">
<meta name="twitter:site" content="@matrixdotorg">
<meta property="og:site_name" content="matrix.org">
<meta name="twitter:image" content="https://example.com/bad.png">
<meta property="og:image" content="https://example.com/good.png">
</html>
"""
tree = decode_body(html, "http://example.com/test.html")
assert tree is not None
og = parse_html_to_open_graph(tree)
self.assertEqual(
og,
{
"og:title": None,
"og:description": "Real Description",
"og:site_name": "matrix.org",
"og:image": "https://example.com/good.png",
},
)
def test_extended_opengraph(self) -> None:
"""Ensure we pull in profile and article data from opengraph."""
html = b"""
<html>
<meta property="og:description" content="My description" />
<meta property="profile:username" content="myname">
<meta property="article:published_time" content="2026-04-07T10:07:37Z">
</html>
"""
tree = decode_body(html, "http://example.com/test.html")
assert tree is not None
og = parse_html_to_open_graph(tree)
self.assertEqual(
og,
{
"og:title": None,
"og:description": "My description",
"profile:username": "myname",
"article:published_time": "2026-04-07T10:07:37Z",
},
)
def test_nested_nodes(self) -> None:
"""A body with some nested nodes. Tests that we iterate over children
in the right order (and don't reverse the order of the text)."""
html = b"""
<a href="somewhere">Welcome <b>the bold <u>and underlined text <svg>
with a cheeky SVG</svg></u> and <strong>some</strong> tail text</b></a>
"""
tree = decode_body(html, "http://example.com/test.html")
assert tree is not None
og = parse_html_to_open_graph(tree)
self.assertEqual(
og,
{
"og:title": None,
"og:description": "Welcome\n\nthe bold\n\nand underlined text\n\nand\n\nsome\n\ntail text",
},
)
def test_ignored_tags(self) -> None:
"""Test ignored elements."""
html = b"""
<header>Foo bar</header>
<div>A real description</div>
"""
tree = decode_body(html, "http://example.com/test.html")
assert tree is not None
og = parse_html_to_open_graph(tree)
self.assertEqual(
og,
{
"og:title": None,
"og:description": "A real description",
},
)
def test_aria_tags(self) -> None:
"""Test ignored elements."""
html = b"""
<div role="menu">Foo bar</div>
<div>A real description</div>
"""
tree = decode_body(html, "http://example.com/test.html")
assert tree is not None
og = parse_html_to_open_graph(tree)
self.assertEqual(
og,
{
"og:title": None,
"og:description": "A real description",
},
)