From 84c4a530a8007698b677771cbfaa4ae25d7cdb95 Mon Sep 17 00:00:00 2001
From: Patrick Cloke <patrickc@matrix.org>
Date: Wed, 8 Sep 2021 14:45:43 -0400
Subject: [PATCH 01/19] Factor out calculating a description from an HTML tree.

---
 synapse/rest/media/v1/preview_url_resource.py | 99 ++++++++++++-------
 1 file changed, 66 insertions(+), 33 deletions(-)
diff --git a/synapse/rest/media/v1/preview_url_resource.py b/synapse/rest/media/v1/preview_url_resource.py
index f108da05db55..a7b58b1556c8 100644
--- a/synapse/rest/media/v1/preview_url_resource.py
+++ b/synapse/rest/media/v1/preview_url_resource.py
@@ -668,7 +668,18 @@ def _attempt_calc_og(body_attempt: Union[bytes, str]) -> Dict[str, Optional[str]
 
 
 def _calc_og(tree: "etree.Element", media_uri: str) -> Dict[str, Optional[str]]:
-    # suck our tree into lxml and define our OG response.
+    """
+    Calculate metadata for an HTML document.
+
+    This uses lxml to search the HTML document for Open Graph data.
+
+    Args:
+        tree: The parsed HTML document.
+        media_url: The URI used to download the body.
+
+    Returns:
+        The Open Graph response as a dictionary.
+    """
 
     # if we see any image URLs in the OG response, then spider them
     # (although the client could choose to do this by asking for previews of those
@@ -742,35 +753,7 @@ def _calc_og(tree: "etree.Element", media_uri: str) -> Dict[str, Optional[str]]:
         if meta_description:
             og["og:description"] = meta_description[0]
         else:
-            # grab any text nodes which are inside the <body/> tag...
-            # unless they are within an HTML5 semantic markup tag...
-            # <header/>, <nav/>, <aside/>, <footer/>
-            # ...or if they are within a <script/> or <style/> tag.
-            # This is a very very very coarse approximation to a plain text
-            # render of the page.
-
-            # We don't just use XPATH here as that is slow on some machines.
-
-            from lxml import etree
-
-            TAGS_TO_REMOVE = (
-                "header",
-                "nav",
-                "aside",
-                "footer",
-                "script",
-                "noscript",
-                "style",
-                etree.Comment,
-            )
-
-            # Split all the text nodes into paragraphs (by splitting on new
-            # lines)
-            text_nodes = (
-                re.sub(r"\s+", "\n", el).strip()
-                for el in _iterate_over_text(tree.find("body"), *TAGS_TO_REMOVE)
-            )
-            og["og:description"] = summarize_paragraphs(text_nodes)
+            og["og:description"] = _calc_description(tree)
     elif og["og:description"]:
         # This must be a non-empty string at this point.
         assert isinstance(og["og:description"], str)
@@ -781,6 +764,46 @@ def _calc_og(tree: "etree.Element", media_uri: str) -> Dict[str, Optional[str]]:
     return og
 
 
+def _calc_description(tree: "etree.Element") -> Optional[str]:
+    """
+    Calculate a text description based on an HTML document.
+
+    Grabs any text nodes which are inside the <body/> tag, unless they are within
+    an HTML5 semantic markup tag (<header/>, <nav/>, <aside/>, <footer/>), or
+    if they are within a <script/> or <style/> tag.
+
+    This is a very very very coarse approximation to a plain text render of the page.
+
+    Args:
+        tree: The parsed HTML document.
+
+    Returns:
+        The plain text description, or None if one cannot be generated.
+    """
+    # We don't just use XPATH here as that is slow on some machines.
+
+    from lxml import etree
+
+    TAGS_TO_REMOVE = (
+        "header",
+        "nav",
+        "aside",
+        "footer",
+        "script",
+        "noscript",
+        "style",
+        etree.Comment,
+    )
+
+    # Split all the text nodes into paragraphs (by splitting on new
+    # lines)
+    text_nodes = (
+        re.sub(r"\s+", "\n", el).strip()
+        for el in _iterate_over_text(tree.find("body"), *TAGS_TO_REMOVE)
+    )
+    return summarize_paragraphs(text_nodes)
+
+
 def _iterate_over_text(
     tree, *tags_to_ignore: Iterable[Union[str, "etree.Comment"]]
 ) -> Generator[str, None, None]:
@@ -843,8 +866,18 @@ def _is_html(content_type: str) -> bool:
 def summarize_paragraphs(
     text_nodes: Iterable[str], min_size: int = 200, max_size: int = 500
 ) -> Optional[str]:
-    # Try to get a summary of between 200 and 500 words, respecting
-    # first paragraph and then word boundaries.
+    """
+    Try to get a summary respecting first paragraph and then word boundaries.
+
+    Args:
+        text_nodes: The paragraphs to summarize.
+        min_size: The minimum number of words to include.
+        max_size: The maximum number of words to include.
+
+    Returns:
+        A summary of the text nodes, or None if that was not possible.
+    """
+
     # TODO: Respect sentences?
 
     description = ""
@@ -867,7 +900,7 @@ def summarize_paragraphs(
         new_desc = ""
 
         # This splits the paragraph into words, but keeping the
-        # (preceeding) whitespace intact so we can easily concat
+        # (preceding) whitespace intact so we can easily concat
         # words back together.
         for match in re.finditer(r"\s*\S+", description):
             word = match.group()

From 160b2ca2273a635c8c140ac080dfdeb7995d8a58 Mon Sep 17 00:00:00 2001
From: Patrick Cloke <patrickc@matrix.org>
Date: Wed, 8 Sep 2021 14:46:34 -0400
Subject: [PATCH 02/19] Factor out calculating the expiration timestamp.

---
 synapse/rest/media/v1/preview_url_resource.py | 5 ++++-
 1 file changed, 4 insertions(+), 1 deletion(-)

diff --git a/synapse/rest/media/v1/preview_url_resource.py b/synapse/rest/media/v1/preview_url_resource.py
index a7b58b1556c8..b9dee17d44f7 100644
--- a/synapse/rest/media/v1/preview_url_resource.py
+++ b/synapse/rest/media/v1/preview_url_resource.py
@@ -258,6 +258,9 @@ async def _do_preview(self, url: str, user: str, ts: int) -> bytes:
 
         logger.debug("got media_info of '%s'", media_info)
 
+        # The timestamp of when this media expires.
+        expiration_ts_ms = media_info.expires + media_info.created_ts_ms
+
         if _is_media(media_info.media_type):
             file_id = media_info.filesystem_id
             dims = await self.media_repo._generate_thumbnails(
@@ -340,7 +343,7 @@ async def _do_preview(self, url: str, user: str, ts: int) -> bytes:
             url,
             media_info.response_code,
             media_info.etag,
-            media_info.expires + media_info.created_ts_ms,
+            expiration_ts_ms,
             jsonog,
             media_info.filesystem_id,
             media_info.created_ts_ms,

From 611849dbb6f62fdc312cd7381038a9e075d9ba6a Mon Sep 17 00:00:00 2001
From: Patrick Cloke <patrickc@matrix.org>
Date: Wed, 8 Sep 2021 14:51:40 -0400
Subject: [PATCH 03/19] Factor out pre-caching a related image.

---
 synapse/rest/media/v1/preview_url_resource.py | 66 +++++++++++--------
 1 file changed, 39 insertions(+), 27 deletions(-)

diff --git a/synapse/rest/media/v1/preview_url_resource.py b/synapse/rest/media/v1/preview_url_resource.py
index b9dee17d44f7..9f109070d832 100644
--- a/synapse/rest/media/v1/preview_url_resource.py
+++ b/synapse/rest/media/v1/preview_url_resource.py
@@ -290,34 +290,8 @@ async def _do_preview(self, url: str, user: str, ts: int) -> bytes:
             encoding = get_html_media_encoding(body, media_info.media_type)
             og = decode_and_calc_og(body, media_info.uri, encoding)
 
-            # pre-cache the image for posterity
-            # FIXME: it might be cleaner to use the same flow as the main /preview_url
-            # request itself and benefit from the same caching etc.  But for now we
-            # just rely on the caching on the master request to speed things up.
-            if "og:image" in og and og["og:image"]:
-                image_info = await self._download_url(
-                    _rebase_url(og["og:image"], media_info.uri), user
-                )
+            await self._precache_image_url(user, media_info, og)
 
-                if _is_media(image_info.media_type):
-                    # TODO: make sure we don't choke on white-on-transparent images
-                    file_id = image_info.filesystem_id
-                    dims = await self.media_repo._generate_thumbnails(
-                        None, file_id, file_id, image_info.media_type, url_cache=True
-                    )
-                    if dims:
-                        og["og:image:width"] = dims["width"]
-                        og["og:image:height"] = dims["height"]
-                    else:
-                        logger.warning("Couldn't get dims for %s", og["og:image"])
-
-                    og[
-                        "og:image"
-                    ] = f"mxc://{self.server_name}/{image_info.filesystem_id}"
-                    og["og:image:type"] = image_info.media_type
-                    og["matrix:image:size"] = image_info.media_length
-                else:
-                    del og["og:image"]
         else:
             logger.warning("Failed to find any OG data in %s", url)
             og = {}
@@ -476,6 +450,44 @@ async def _download_url(self, url: str, user: str) -> MediaInfo:
             etag=etag,
         )
 
+    async def _precache_image_url(
+        self, user: str, media_info: MediaInfo, og: JsonDict
+    ) -> None:
+        """
+        Pre-cache the image (if one exists) for posterity
+
+        Args:
+            user: The user requesting the preview.
+            media_info: The media being previewed.
+            og: The Open Graph dictionary. This is modified with image information.
+        """
+        #
+        # FIXME: it might be cleaner to use the same flow as the main /preview_url
+        # request itself and benefit from the same caching etc.  But for now we
+        # just rely on the caching on the master request to speed things up.
+        if "og:image" in og and og["og:image"]:
+            image_info = await self._download_url(
+                _rebase_url(og["og:image"], media_info.uri), user
+            )
+
+            if _is_media(image_info.media_type):
+                # TODO: make sure we don't choke on white-on-transparent images
+                file_id = image_info.filesystem_id
+                dims = await self.media_repo._generate_thumbnails(
+                    None, file_id, file_id, image_info.media_type, url_cache=True
+                )
+                if dims:
+                    og["og:image:width"] = dims["width"]
+                    og["og:image:height"] = dims["height"]
+                else:
+                    logger.warning("Couldn't get dims for %s", og["og:image"])
+
+                og["og:image"] = f"mxc://{self.server_name}/{image_info.filesystem_id}"
+                og["og:image:type"] = image_info.media_type
+                og["matrix:image:size"] = image_info.media_length
+            else:
+                del og["og:image"]
+
     def _start_expire_url_cache_data(self):
         return run_as_background_process(
             "expire_url_cache_data", self._expire_url_cache_data

From 05076b7050e8ffb93ada3c2794a2ac8a9d1eea55 Mon Sep 17 00:00:00 2001
From: Patrick Cloke <patrickc@matrix.org>
Date: Tue, 14 Sep 2021 11:31:32 -0400
Subject: [PATCH 04/19] Update test to actually return an image.

---
 tests/rest/media/v1/test_url_preview.py | 23 ++++++++++++++---------
 1 file changed, 14 insertions(+), 9 deletions(-)

diff --git a/tests/rest/media/v1/test_url_preview.py b/tests/rest/media/v1/test_url_preview.py
index 9f6fbfe6def8..1c3242de3a42 100644
--- a/tests/rest/media/v1/test_url_preview.py
+++ b/tests/rest/media/v1/test_url_preview.py
@@ -14,6 +14,7 @@
 import json
 import os
 import re
+from binascii import unhexlify
 
 from twisted.internet._resolver import HostResolution
 from twisted.internet.address import IPv4Address, IPv6Address
@@ -576,11 +577,10 @@ def test_oembed_photo(self):
         }
         oembed_content = json.dumps(result).encode("utf-8")
 
-        end_content = (
-            b"<html><head>"
-            b"<title>Some Title</title>"
-            b'<meta property="og:description" content="hi" />'
-            b"</head></html>"
+        end_content = unhexlify(
+            b"89504e470d0a1a0a0000000d4948445200000001000000010806"
+            b"0000001f15c4890000000a49444154789c63000100000500010d"
+            b"0a2db40000000049454e44ae426082"
         )
 
         channel = self.make_request(
@@ -606,6 +606,7 @@ def test_oembed_photo(self):
 
         self.pump()
 
+        # Ensure a second request is made to the photo URL.
         client = self.reactor.tcpClients[1][2].buildProtocol(None)
         server = AccumulatingProtocol()
         server.makeConnection(FakeTransport(client, self.reactor))
@@ -613,7 +614,7 @@ def test_oembed_photo(self):
         client.dataReceived(
             (
                 b"HTTP/1.0 200 OK\r\nContent-Length: %d\r\n"
-                b'Content-Type: text/html; charset="utf8"\r\n\r\n'
+                b'Content-Type: image/png; charset="utf8"\r\n\r\n'
             )
             % (len(end_content),)
             + end_content
@@ -621,10 +622,14 @@ def test_oembed_photo(self):
 
         self.pump()
 
+        # Ensure the URL is what was requested.
+        self.assertIn(b"/matrixdotorg", server.data)
+
         self.assertEqual(channel.code, 200)
-        self.assertEqual(
-            channel.json_body, {"og:title": "Some Title", "og:description": "hi"}
-        )
+        self.assertTrue(channel.json_body["og:image"].startswith("mxc://"))
+        self.assertEqual(channel.json_body["og:image:height"], 1)
+        self.assertEqual(channel.json_body["og:image:width"], 1)
+        self.assertTrue(channel.json_body["og:image:type"].startswith("image/png"))
 
     def test_oembed_rich(self):
         """Test an oEmbed endpoint which returns HTML content via the 'rich' type."""

From 0501d08c86a4be0f7dd899731d31e64a81018b59 Mon Sep 17 00:00:00 2001
From: Patrick Cloke <patrickc@matrix.org>
Date: Wed, 8 Sep 2021 15:11:39 -0400
Subject: [PATCH 05/19] Process oEmbed by downloading the result to a file for
 later processing.

This makes oEmbed previews more like image / HTML previews and
takes beter advantage of the caching layer.
---
 docs/development/url_previews.md              |  16 +-
 synapse/rest/media/v1/oembed.py               | 130 +++++++++------
 synapse/rest/media/v1/preview_url_resource.py | 150 ++++++++----------
 tests/rest/media/v1/test_url_preview.py       |   1 +
 4 files changed, 160 insertions(+), 137 deletions(-)

diff --git a/docs/development/url_previews.md b/docs/development/url_previews.md
index bbe05e281ca7..a65fb0b8f62c 100644
--- a/docs/development/url_previews.md
+++ b/docs/development/url_previews.md
@@ -25,16 +25,15 @@ When Synapse is asked to preview a URL it does the following:
 3. Kicks off a background process to generate a preview:
    1. Checks the database cache by URL and timestamp and returns the result if it
       has not expired and was successful (a 2xx return code).
-   2. Checks if the URL matches an oEmbed pattern. If it does, fetch the oEmbed
-      response. If this is an image, replace the URL to fetch and continue. If
-      if it is HTML content, use the HTML as the document and continue.
+   2. Checks if the URL matches an oEmbed pattern. If it does, replace the URL
+      to fetch.
    3. If it doesn't match an oEmbed pattern, downloads the URL and stores it
       into a file via the media storage provider and saves the local media
       metadata.
-   5. If the media is an image:
+   4. If the media is an image:
       1. Generates thumbnails.
       2. Generates an Open Graph response based on image properties.
-   6. If the media is HTML:
+   5. If the media is HTML:
       1. Decodes the HTML via the stored file.
       2. Generates an Open Graph response from the HTML.
       3. If an image exists in the Open Graph response:
@@ -42,6 +41,13 @@ When Synapse is asked to preview a URL it does the following:
             provider and saves the local media metadata.
          2. Generates thumbnails.
          3. Updates the Open Graph response based on image properties.
+   6. If the media is JSON and oEmbed was used:
+      1. Convert the oEmbed response to an Open Graph response.
+      2. If a thumbnail or image is in the oEmbed response:
+         1. Downloads the URL and stores it into a file via the media storage
+            provider and saves the local media metadata.
+         2. Generates thumbnails.
+         3. Updates the Open Graph response based on image properties.
    7. Stores the result in the database cache.
 4. Returns the result.
 
diff --git a/synapse/rest/media/v1/oembed.py b/synapse/rest/media/v1/oembed.py
index 2e6706dbfa7f..96fee6322151 100644
--- a/synapse/rest/media/v1/oembed.py
+++ b/synapse/rest/media/v1/oembed.py
@@ -12,11 +12,14 @@
 #  See the License for the specific language governing permissions and
 #  limitations under the License.
 import logging
+import urllib.parse
 from typing import TYPE_CHECKING, Optional
 
 import attr
 
 from synapse.http.client import SimpleHttpClient
+from synapse.types import JsonDict
+from synapse.util import json_decoder
 
 if TYPE_CHECKING:
     from synapse.server import HomeServer
@@ -24,18 +27,11 @@
 logger = logging.getLogger(__name__)
 
 
-@attr.s(slots=True, auto_attribs=True)
+@attr.s(slots=True, frozen=True, auto_attribs=True)
 class OEmbedResult:
-    # Either HTML content or URL must be provided.
-    html: Optional[str]
-    url: Optional[str]
-    title: Optional[str]
+    og: JsonDict
     # Number of seconds to cache the content.
-    cache_age: int
-
-
-class OEmbedError(Exception):
-    """An error occurred processing the oEmbed object."""
+    cache_age: Optional[int]
 
 
 class OEmbedProvider:
@@ -81,41 +77,38 @@ def get_oembed_url(self, url: str) -> Optional[str]:
         """
         for url_pattern, endpoint in self._oembed_patterns.items():
             if url_pattern.fullmatch(url):
-                return endpoint
+                # TODO Specify max height / width.
+
+                # Note that only the JSON format is supported, some endpoints want
+                # this in the URL, others want it as an argument.
+                endpoint = endpoint.replace("{format}", "json")
+
+                args = {"url": url, "format": "json"}
+                query_str = urllib.parse.urlencode(args, True)
+                return f"{endpoint}?{query_str}"
 
         # No match.
         return None
 
-    async def get_oembed_content(self, endpoint: str, url: str) -> OEmbedResult:
+    def parse_oembed_response(self, url: str, body: str) -> OEmbedResult:
         """
-        Request content from an oEmbed endpoint.
+        Parse the oEmbed response into an Open Graph response.
 
         Args:
-            endpoint: The oEmbed API endpoint.
-            url: The URL to pass to the API.
+            url: The URL which is being previewed (not the one which was
+                requested).
+            body: The oEmbed response as JSON.
 
         Returns:
-            An object representing the metadata returned.
-
-        Raises:
-            OEmbedError if fetching or parsing of the oEmbed information fails.
+            json-encoded Open Graph data
         """
-        try:
-            logger.debug("Trying to get oEmbed content for url '%s'", url)
 
-            # Note that only the JSON format is supported, some endpoints want
-            # this in the URL, others want it as an argument.
-            endpoint = endpoint.replace("{format}", "json")
-
-            result = await self._client.get_json(
-                endpoint,
-                # TODO Specify max height / width.
-                args={"url": url, "format": "json"},
-            )
+        try:
+            result = json_decoder.decode(body)
 
             # Ensure there's a version of 1.0.
             if result.get("version") != "1.0":
-                raise OEmbedError("Invalid version: %s" % (result.get("version"),))
+                raise RuntimeError("Invalid version: %s" % (result.get("version"),))
 
             oembed_type = result.get("type")
 
@@ -124,32 +117,65 @@ async def get_oembed_content(self, endpoint: str, url: str) -> OEmbedResult:
             if cache_age:
                 cache_age = int(cache_age)
 
-            oembed_result = OEmbedResult(None, None, result.get("title"), cache_age)
+            # The results.
+            og = {"og:title": result.get("title")}
 
-            # HTML content.
+            # If a thumbnail exists, use it. Note that dimensions will be calculated later.
+            if "thumbnail_url" in result:
+                og["og:image"] = result["thumbnail_url"]
+
+            # Process each type separately.
             if oembed_type == "rich":
-                oembed_result.html = result.get("html")
-                return oembed_result
+                calc_description_and_urls(og, result.get("html"))
 
-            if oembed_type == "photo":
-                oembed_result.url = result.get("url")
-                return oembed_result
+            elif oembed_type == "photo":
+                # If this is a photo, use the full image, not the thumbnail.
+                og["og:image"] = result.get("url")
 
-            # TODO Handle link and video types.
+            else:
+                raise RuntimeError(f"Unknown oEmbed type: {oembed_type}")
 
-            if "thumbnail_url" in result:
-                oembed_result.url = result.get("thumbnail_url")
-                return oembed_result
+        except Exception as e:
+            # Trap any exception and let the code follow as usual.
+            logger.warning(f"Error parsing oEmbed metadata from {url}: {e:r}")
+            og = {}
+            cache_age = None
 
-            raise OEmbedError("Incompatible oEmbed information.")
+        return OEmbedResult(og, cache_age)
 
-        except OEmbedError as e:
-            # Trap OEmbedErrors first so we can directly re-raise them.
-            logger.warning("Error parsing oEmbed metadata from %s: %r", url, e)
-            raise
 
-        except Exception as e:
-            # Trap any exception and let the code follow as usual.
-            # FIXME: pass through 404s and other error messages nicely
-            logger.warning("Error downloading oEmbed metadata from %s: %r", url, e)
-            raise OEmbedError() from e
+def calc_description_and_urls(og: JsonDict, body: str) -> None:
+    """
+    Calculate description for an HTML document.
+
+    This uses lxml to convert the HTML document into plaintext. If errors
+    occur during processing of the document, an empty response is returned.
+
+    Args:
+        og: The current Open Graph summary. This is updated with additional fields.
+        body: The HTML document, as bytes.
+
+    Returns:
+        The summary
+    """
+    # If there's no body, nothing useful is going to be found.
+    if not body:
+        return
+
+    from lxml import etree
+
+    # Create an HTML parser. If this fails, log and return no metadata.
+    parser = etree.HTMLParser(recover=True, encoding="utf-8")
+
+    # Attempt to parse the body. If this fails, log and return no metadata.
+    tree = etree.fromstring(body, parser)
+
+    # The data was successfully parsed, but no tree was found.
+    if tree is None:
+        return
+
+    from synapse.rest.media.v1.preview_url_resource import _calc_description
+
+    description = _calc_description(tree)
+    if description:
+        og["og:description"] = description
diff --git a/synapse/rest/media/v1/preview_url_resource.py b/synapse/rest/media/v1/preview_url_resource.py
index 9f109070d832..679018864245 100644
--- a/synapse/rest/media/v1/preview_url_resource.py
+++ b/synapse/rest/media/v1/preview_url_resource.py
@@ -43,7 +43,7 @@
 from synapse.metrics.background_process_metrics import run_as_background_process
 from synapse.rest.media.v1._base import get_filename_from_headers
 from synapse.rest.media.v1.media_storage import MediaStorage
-from synapse.rest.media.v1.oembed import OEmbedError, OEmbedProvider
+from synapse.rest.media.v1.oembed import OEmbedProvider
 from synapse.types import JsonDict
 from synapse.util import json_encoder
 from synapse.util.async_helpers import ObservableDeferred
@@ -254,7 +254,13 @@ async def _do_preview(self, url: str, user: str, ts: int) -> bytes:
                 og = og.encode("utf8")
             return og
 
-        media_info = await self._download_url(url, user)
+        # If this URL can be accessed via oEmbed, use that instead.
+        url_to_download = url
+        oembed_url = self._oembed.get_oembed_url(url)
+        if oembed_url:
+            url_to_download = oembed_url
+
+        media_info = await self._download_url(url_to_download, user)
 
         logger.debug("got media_info of '%s'", media_info)
 
@@ -292,6 +298,22 @@ async def _do_preview(self, url: str, user: str, ts: int) -> bytes:
 
             await self._precache_image_url(user, media_info, og)
 
+        elif oembed_url and _is_json(media_info.media_type):
+            # Handle an oEmbed response.
+            with open(media_info.filename, "r") as utf8_file:
+                oembed_body = utf8_file.read()
+
+            oembed_response = self._oembed.parse_oembed_response(
+                media_info.uri, oembed_body
+            )
+            og = oembed_response.og
+
+            # Use the cache age from the oEmbed result, instead of the HTTP response.
+            if oembed_response.cache_age is not None:
+                expiration_ts_ms = oembed_response.cache_age + media_info.created_ts_ms
+
+            await self._precache_image_url(user, media_info, og)
+
         else:
             logger.warning("Failed to find any OG data in %s", url)
             og = {}
@@ -334,88 +356,52 @@ async def _download_url(self, url: str, user: str) -> MediaInfo:
 
         file_info = FileInfo(server_name=None, file_id=file_id, url_cache=True)
 
-        # If this URL can be accessed via oEmbed, use that instead.
-        url_to_download: Optional[str] = url
-        oembed_url = self._oembed.get_oembed_url(url)
-        if oembed_url:
-            # The result might be a new URL to download, or it might be HTML content.
+        with self.media_storage.store_into_file(file_info) as (f, fname, finish):
             try:
-                oembed_result = await self._oembed.get_oembed_content(oembed_url, url)
-                if oembed_result.url:
-                    url_to_download = oembed_result.url
-                elif oembed_result.html:
-                    url_to_download = None
-            except OEmbedError:
-                # If an error occurs, try doing a normal preview.
-                pass
+                logger.debug("Trying to get preview for url '%s'", url)
+                length, headers, uri, code = await self.client.get_file(
+                    url,
+                    output_stream=f,
+                    max_size=self.max_spider_size,
+                    headers={"Accept-Language": self.url_preview_accept_language},
+                )
+            except SynapseError:
+                # Pass SynapseErrors through directly, so that the servlet
+                # handler will return a SynapseError to the client instead of
+                # blank data or a 500.
+                raise
+            except DNSLookupError:
+                # DNS lookup returned no results
+                # Note: This will also be the case if one of the resolved IP
+                # addresses is blacklisted
+                raise SynapseError(
+                    502,
+                    "DNS resolution failure during URL preview generation",
+                    Codes.UNKNOWN,
+                )
+            except Exception as e:
+                # FIXME: pass through 404s and other error messages nicely
+                logger.warning("Error downloading %s: %r", url, e)
 
-        if url_to_download:
-            with self.media_storage.store_into_file(file_info) as (f, fname, finish):
-                try:
-                    logger.debug("Trying to get preview for url '%s'", url_to_download)
-                    length, headers, uri, code = await self.client.get_file(
-                        url_to_download,
-                        output_stream=f,
-                        max_size=self.max_spider_size,
-                        headers={"Accept-Language": self.url_preview_accept_language},
-                    )
-                except SynapseError:
-                    # Pass SynapseErrors through directly, so that the servlet
-                    # handler will return a SynapseError to the client instead of
-                    # blank data or a 500.
-                    raise
-                except DNSLookupError:
-                    # DNS lookup returned no results
-                    # Note: This will also be the case if one of the resolved IP
-                    # addresses is blacklisted
-                    raise SynapseError(
-                        502,
-                        "DNS resolution failure during URL preview generation",
-                        Codes.UNKNOWN,
-                    )
-                except Exception as e:
-                    # FIXME: pass through 404s and other error messages nicely
-                    logger.warning("Error downloading %s: %r", url_to_download, e)
-
-                    raise SynapseError(
-                        500,
-                        "Failed to download content: %s"
-                        % (traceback.format_exception_only(sys.exc_info()[0], e),),
-                        Codes.UNKNOWN,
-                    )
-                await finish()
-
-                if b"Content-Type" in headers:
-                    media_type = headers[b"Content-Type"][0].decode("ascii")
-                else:
-                    media_type = "application/octet-stream"
+                raise SynapseError(
+                    500,
+                    "Failed to download content: %s"
+                    % (traceback.format_exception_only(sys.exc_info()[0], e),),
+                    Codes.UNKNOWN,
+                )
+            await finish()
 
-                download_name = get_filename_from_headers(headers)
+            if b"Content-Type" in headers:
+                media_type = headers[b"Content-Type"][0].decode("ascii")
+            else:
+                media_type = "application/octet-stream"
 
-                # FIXME: we should calculate a proper expiration based on the
-                # Cache-Control and Expire headers.  But for now, assume 1 hour.
-                expires = ONE_HOUR
-                etag = (
-                    headers[b"ETag"][0].decode("ascii") if b"ETag" in headers else None
-                )
-        else:
-            # we can only get here if we did an oembed request and have an oembed_result.html
-            assert oembed_result.html is not None
-            assert oembed_url is not None
-
-            html_bytes = oembed_result.html.encode("utf-8")
-            with self.media_storage.store_into_file(file_info) as (f, fname, finish):
-                f.write(html_bytes)
-                await finish()
-
-            media_type = "text/html"
-            download_name = oembed_result.title
-            length = len(html_bytes)
-            # If a specific cache age was not given, assume 1 hour.
-            expires = oembed_result.cache_age or ONE_HOUR
-            uri = oembed_url
-            code = 200
-            etag = None
+            download_name = get_filename_from_headers(headers)
+
+            # FIXME: we should calculate a proper expiration based on the
+            # Cache-Control and Expire headers.  But for now, assume 1 hour.
+            expires = ONE_HOUR
+            etag = headers[b"ETag"][0].decode("ascii") if b"ETag" in headers else None
 
         try:
             time_now_ms = self.clock.time_msec()
@@ -878,6 +864,10 @@ def _is_html(content_type: str) -> bool:
     )
 
 
+def _is_json(content_type: str) -> bool:
+    return content_type.lower().startswith("application/json")
+
+
 def summarize_paragraphs(
     text_nodes: Iterable[str], min_size: int = 200, max_size: int = 500
 ) -> Optional[str]:
diff --git a/tests/rest/media/v1/test_url_preview.py b/tests/rest/media/v1/test_url_preview.py
index 1c3242de3a42..2a8e7b83b294 100644
--- a/tests/rest/media/v1/test_url_preview.py
+++ b/tests/rest/media/v1/test_url_preview.py
@@ -626,6 +626,7 @@ def test_oembed_photo(self):
         self.assertIn(b"/matrixdotorg", server.data)
 
         self.assertEqual(channel.code, 200)
+        self.assertIsNone(channel.json_body["og:title"])
         self.assertTrue(channel.json_body["og:image"].startswith("mxc://"))
         self.assertEqual(channel.json_body["og:image:height"], 1)
         self.assertEqual(channel.json_body["og:image:width"], 1)

From 0c763bfea2f7c884fcda249c21f2f61cccce0fd7 Mon Sep 17 00:00:00 2001
From: Patrick Cloke <patrickc@matrix.org>
Date: Wed, 8 Sep 2021 15:22:28 -0400
Subject: [PATCH 06/19] Assume required properties exist.

---
 synapse/rest/media/v1/oembed.py | 12 ++++++------
 1 file changed, 6 insertions(+), 6 deletions(-)

diff --git a/synapse/rest/media/v1/oembed.py b/synapse/rest/media/v1/oembed.py
index 96fee6322151..6d436c4d0816 100644
--- a/synapse/rest/media/v1/oembed.py
+++ b/synapse/rest/media/v1/oembed.py
@@ -107,10 +107,9 @@ def parse_oembed_response(self, url: str, body: str) -> OEmbedResult:
             result = json_decoder.decode(body)
 
             # Ensure there's a version of 1.0.
-            if result.get("version") != "1.0":
-                raise RuntimeError("Invalid version: %s" % (result.get("version"),))
-
-            oembed_type = result.get("type")
+            oembed_version = result["version"]
+            if oembed_version != "1.0":
+                raise RuntimeError(f"Invalid version: {oembed_version}")
 
             # Ensure the cache age is None or an int.
             cache_age = result.get("cache_age")
@@ -125,12 +124,13 @@ def parse_oembed_response(self, url: str, body: str) -> OEmbedResult:
                 og["og:image"] = result["thumbnail_url"]
 
             # Process each type separately.
+            oembed_type = result["type"]
             if oembed_type == "rich":
-                calc_description_and_urls(og, result.get("html"))
+                calc_description_and_urls(og, result["html"])
 
             elif oembed_type == "photo":
                 # If this is a photo, use the full image, not the thumbnail.
-                og["og:image"] = result.get("url")
+                og["og:image"] = result["url"]
 
             else:
                 raise RuntimeError(f"Unknown oEmbed type: {oembed_type}")

From 3ffaee602067b245bca656351c72ebf592078c83 Mon Sep 17 00:00:00 2001
From: Patrick Cloke <patrickc@matrix.org>
Date: Mon, 13 Sep 2021 14:56:06 -0400
Subject: [PATCH 07/19] Newsfragment

---
 changelog.d/10814.feature | 1 +
 1 file changed, 1 insertion(+)
 create mode 100644 changelog.d/10814.feature

diff --git a/changelog.d/10814.feature b/changelog.d/10814.feature
new file mode 100644
index 000000000000..4fa95a6cc968
--- /dev/null
+++ b/changelog.d/10814.feature
@@ -0,0 +1 @@
+Improve oEmbed previews by processing the author name, photo, and video information.

From c98626c42c63cb6e4c4a408c3e0cbdcd924762af Mon Sep 17 00:00:00 2001
From: Patrick Cloke <patrickc@matrix.org>
Date: Thu, 16 Sep 2021 10:55:07 -0400
Subject: [PATCH 08/19] Rename some variables.

---
 synapse/rest/media/v1/oembed.py | 38 ++++++++++++++++-----------------
 1 file changed, 19 insertions(+), 19 deletions(-)

diff --git a/synapse/rest/media/v1/oembed.py b/synapse/rest/media/v1/oembed.py
index 6d436c4d0816..68ce2b760227 100644
--- a/synapse/rest/media/v1/oembed.py
+++ b/synapse/rest/media/v1/oembed.py
@@ -90,47 +90,47 @@ def get_oembed_url(self, url: str) -> Optional[str]:
         # No match.
         return None
 
-    def parse_oembed_response(self, url: str, body: str) -> OEmbedResult:
+    def parse_oembed_response(self, url: str, raw_body: str) -> OEmbedResult:
         """
         Parse the oEmbed response into an Open Graph response.
 
         Args:
             url: The URL which is being previewed (not the one which was
                 requested).
-            body: The oEmbed response as JSON.
+            raw_body: The oEmbed response as JSON.
 
         Returns:
             json-encoded Open Graph data
         """
 
         try:
-            result = json_decoder.decode(body)
+            oembed = json_decoder.decode(raw_body)
 
             # Ensure there's a version of 1.0.
-            oembed_version = result["version"]
+            oembed_version = oembed["version"]
             if oembed_version != "1.0":
                 raise RuntimeError(f"Invalid version: {oembed_version}")
 
             # Ensure the cache age is None or an int.
-            cache_age = result.get("cache_age")
+            cache_age = oembed.get("cache_age")
             if cache_age:
                 cache_age = int(cache_age)
 
             # The results.
-            og = {"og:title": result.get("title")}
+            ppen_graph_response = {"og:title": oembed.get("title")}
 
             # If a thumbnail exists, use it. Note that dimensions will be calculated later.
-            if "thumbnail_url" in result:
-                og["og:image"] = result["thumbnail_url"]
+            if "thumbnail_url" in oembed:
+                ppen_graph_response["og:image"] = oembed["thumbnail_url"]
 
             # Process each type separately.
-            oembed_type = result["type"]
+            oembed_type = oembed["type"]
             if oembed_type == "rich":
-                calc_description_and_urls(og, result["html"])
+                calc_description_and_urls(ppen_graph_response, oembed["html"])
 
             elif oembed_type == "photo":
                 # If this is a photo, use the full image, not the thumbnail.
-                og["og:image"] = result["url"]
+                ppen_graph_response["og:image"] = oembed["url"]
 
             else:
                 raise RuntimeError(f"Unknown oEmbed type: {oembed_type}")
@@ -138,13 +138,13 @@ def parse_oembed_response(self, url: str, body: str) -> OEmbedResult:
         except Exception as e:
             # Trap any exception and let the code follow as usual.
             logger.warning(f"Error parsing oEmbed metadata from {url}: {e:r}")
-            og = {}
+            ppen_graph_response = {}
             cache_age = None
 
-        return OEmbedResult(og, cache_age)
+        return OEmbedResult(ppen_graph_response, cache_age)
 
 
-def calc_description_and_urls(og: JsonDict, body: str) -> None:
+def calc_description_and_urls(open_graph_response: JsonDict, html_body: str) -> None:
     """
     Calculate description for an HTML document.
 
@@ -152,14 +152,14 @@ def calc_description_and_urls(og: JsonDict, body: str) -> None:
     occur during processing of the document, an empty response is returned.
 
     Args:
-        og: The current Open Graph summary. This is updated with additional fields.
-        body: The HTML document, as bytes.
+        open_graph_response: The current Open Graph summary. This is updated with additional fields.
+        html_body: The HTML document, as bytes.
 
     Returns:
         The summary
     """
     # If there's no body, nothing useful is going to be found.
-    if not body:
+    if not html_body:
         return
 
     from lxml import etree
@@ -168,7 +168,7 @@ def calc_description_and_urls(og: JsonDict, body: str) -> None:
     parser = etree.HTMLParser(recover=True, encoding="utf-8")
 
     # Attempt to parse the body. If this fails, log and return no metadata.
-    tree = etree.fromstring(body, parser)
+    tree = etree.fromstring(html_body, parser)
 
     # The data was successfully parsed, but no tree was found.
     if tree is None:
@@ -178,4 +178,4 @@ def calc_description_and_urls(og: JsonDict, body: str) -> None:
 
     description = _calc_description(tree)
     if description:
-        og["og:description"] = description
+        open_graph_response["og:description"] = description

From c29713f5c64c15d426a7e3bb8128975cd3c58c83 Mon Sep 17 00:00:00 2001
From: Patrick Cloke <patrickc@matrix.org>
Date: Thu, 16 Sep 2021 10:57:45 -0400
Subject: [PATCH 09/19] Clarify docs more.

---
 docs/development/url_previews.md | 11 +++++------
 1 file changed, 5 insertions(+), 6 deletions(-)

diff --git a/docs/development/url_previews.md b/docs/development/url_previews.md
index a65fb0b8f62c..aff38136091d 100644
--- a/docs/development/url_previews.md
+++ b/docs/development/url_previews.md
@@ -25,11 +25,10 @@ When Synapse is asked to preview a URL it does the following:
 3. Kicks off a background process to generate a preview:
    1. Checks the database cache by URL and timestamp and returns the result if it
       has not expired and was successful (a 2xx return code).
-   2. Checks if the URL matches an oEmbed pattern. If it does, replace the URL
-      to fetch.
-   3. If it doesn't match an oEmbed pattern, downloads the URL and stores it
-      into a file via the media storage provider and saves the local media
-      metadata.
+   2. Checks if the URL matches an [oEmbed](https://oembed.com/) pattern. If it
+      does, update the URL to download.
+   3. Downloads the URL and stores it into a file via the media storage provider
+      and saves the local media metadata.
    4. If the media is an image:
       1. Generates thumbnails.
       2. Generates an Open Graph response based on image properties.
@@ -41,7 +40,7 @@ When Synapse is asked to preview a URL it does the following:
             provider and saves the local media metadata.
          2. Generates thumbnails.
          3. Updates the Open Graph response based on image properties.
-   6. If the media is JSON and oEmbed was used:
+   6. If the media is JSON and an oEmbed URL was found:
       1. Convert the oEmbed response to an Open Graph response.
       2. If a thumbnail or image is in the oEmbed response:
          1. Downloads the URL and stores it into a file via the media storage

From e9cf11ffd71f40ea705fcd54f19ca08712c606ac Mon Sep 17 00:00:00 2001
From: Patrick Cloke <patrickc@matrix.org>
Date: Thu, 16 Sep 2021 12:05:19 -0400
Subject: [PATCH 10/19] Use png constant.

---
 tests/rest/media/v1/test_url_preview.py | 12 +++---------
 1 file changed, 3 insertions(+), 9 deletions(-)

diff --git a/tests/rest/media/v1/test_url_preview.py b/tests/rest/media/v1/test_url_preview.py
index 2a8e7b83b294..ceefa58d28bd 100644
--- a/tests/rest/media/v1/test_url_preview.py
+++ b/tests/rest/media/v1/test_url_preview.py
@@ -14,7 +14,6 @@
 import json
 import os
 import re
-from binascii import unhexlify
 
 from twisted.internet._resolver import HostResolution
 from twisted.internet.address import IPv4Address, IPv6Address
@@ -25,6 +24,7 @@
 
 from tests import unittest
 from tests.server import FakeTransport
+from tests.test_utils import SMALL_PNG
 
 try:
     import lxml
@@ -577,12 +577,6 @@ def test_oembed_photo(self):
         }
         oembed_content = json.dumps(result).encode("utf-8")
 
-        end_content = unhexlify(
-            b"89504e470d0a1a0a0000000d4948445200000001000000010806"
-            b"0000001f15c4890000000a49444154789c63000100000500010d"
-            b"0a2db40000000049454e44ae426082"
-        )
-
         channel = self.make_request(
             "GET",
             "preview_url?url=http://twitter.com/matrixdotorg/status/12345",
@@ -616,8 +610,8 @@ def test_oembed_photo(self):
                 b"HTTP/1.0 200 OK\r\nContent-Length: %d\r\n"
                 b'Content-Type: image/png; charset="utf8"\r\n\r\n'
             )
-            % (len(end_content),)
-            + end_content
+            % (len(SMALL_PNG),)
+            + SMALL_PNG
         )
 
         self.pump()

From 53b733ba78c0c2e68bfe54a27b9053cdb9d64024 Mon Sep 17 00:00:00 2001
From: Patrick Cloke <patrickc@matrix.org>
Date: Thu, 16 Sep 2021 12:07:55 -0400
Subject: [PATCH 11/19] Clearer property names.

---
 synapse/rest/media/v1/oembed.py               | 3 ++-
 synapse/rest/media/v1/preview_url_resource.py | 2 +-
 2 files changed, 3 insertions(+), 2 deletions(-)

diff --git a/synapse/rest/media/v1/oembed.py b/synapse/rest/media/v1/oembed.py
index 68ce2b760227..b309c81be0c6 100644
--- a/synapse/rest/media/v1/oembed.py
+++ b/synapse/rest/media/v1/oembed.py
@@ -29,7 +29,8 @@
 
 @attr.s(slots=True, frozen=True, auto_attribs=True)
 class OEmbedResult:
-    og: JsonDict
+    # The Open Graph result (converted from the oEmbed result).
+    open_graph_result: JsonDict
     # Number of seconds to cache the content.
     cache_age: Optional[int]
 
diff --git a/synapse/rest/media/v1/preview_url_resource.py b/synapse/rest/media/v1/preview_url_resource.py
index 014f35cea81c..2d0523b5cf0a 100644
--- a/synapse/rest/media/v1/preview_url_resource.py
+++ b/synapse/rest/media/v1/preview_url_resource.py
@@ -307,7 +307,7 @@ async def _do_preview(self, url: str, user: str, ts: int) -> bytes:
             oembed_response = self._oembed.parse_oembed_response(
                 media_info.uri, oembed_body
             )
-            og = oembed_response.og
+            og = oembed_response.open_graph_result
 
             # Use the cache age from the oEmbed result, instead of the HTTP response.
             if oembed_response.cache_age is not None:

From 86b75fb4fd60ae9be4063fe8d9fcbf2f28dfc7c9 Mon Sep 17 00:00:00 2001
From: Patrick Cloke <patrickc@matrix.org>
Date: Tue, 21 Sep 2021 09:28:30 -0400
Subject: [PATCH 12/19] Return early in _precache_image_url.

---
 synapse/rest/media/v1/preview_url_resource.py | 44 ++++++++++---------
 1 file changed, 23 insertions(+), 21 deletions(-)

diff --git a/synapse/rest/media/v1/preview_url_resource.py b/synapse/rest/media/v1/preview_url_resource.py
index 2d0523b5cf0a..992abdb51edf 100644
--- a/synapse/rest/media/v1/preview_url_resource.py
+++ b/synapse/rest/media/v1/preview_url_resource.py
@@ -448,32 +448,34 @@ async def _precache_image_url(
             media_info: The media being previewed.
             og: The Open Graph dictionary. This is modified with image information.
         """
-        #
+        # If there's no image or it is blank, there's nothing to do.
+        if "og:image" not in og or not og["og:image"]:
+            return
+
         # FIXME: it might be cleaner to use the same flow as the main /preview_url
         # request itself and benefit from the same caching etc.  But for now we
         # just rely on the caching on the master request to speed things up.
-        if "og:image" in og and og["og:image"]:
-            image_info = await self._download_url(
-                _rebase_url(og["og:image"], media_info.uri), user
-            )
-
-            if _is_media(image_info.media_type):
-                # TODO: make sure we don't choke on white-on-transparent images
-                file_id = image_info.filesystem_id
-                dims = await self.media_repo._generate_thumbnails(
-                    None, file_id, file_id, image_info.media_type, url_cache=True
-                )
-                if dims:
-                    og["og:image:width"] = dims["width"]
-                    og["og:image:height"] = dims["height"]
-                else:
-                    logger.warning("Couldn't get dims for %s", og["og:image"])
+        image_info = await self._download_url(
+            _rebase_url(og["og:image"], media_info.uri), user
+        )
 
-                og["og:image"] = f"mxc://{self.server_name}/{image_info.filesystem_id}"
-                og["og:image:type"] = image_info.media_type
-                og["matrix:image:size"] = image_info.media_length
+        if _is_media(image_info.media_type):
+            # TODO: make sure we don't choke on white-on-transparent images
+            file_id = image_info.filesystem_id
+            dims = await self.media_repo._generate_thumbnails(
+                None, file_id, file_id, image_info.media_type, url_cache=True
+            )
+            if dims:
+                og["og:image:width"] = dims["width"]
+                og["og:image:height"] = dims["height"]
             else:
-                del og["og:image"]
+                logger.warning("Couldn't get dims for %s", og["og:image"])
+
+            og["og:image"] = f"mxc://{self.server_name}/{image_info.filesystem_id}"
+            og["og:image:type"] = image_info.media_type
+            og["matrix:image:size"] = image_info.media_length
+        else:
+            del og["og:image"]
 
     def _start_expire_url_cache_data(self) -> Deferred:
         return run_as_background_process(

From 93c0c250ec6cfbf962cfbbe46306f7f119d525c5 Mon Sep 17 00:00:00 2001
From: Patrick Cloke <patrickc@matrix.org>
Date: Tue, 21 Sep 2021 09:39:40 -0400
Subject: [PATCH 13/19] Fix charset in tests.

---
 tests/rest/media/v1/test_url_preview.py | 4 ++--
 1 file changed, 2 insertions(+), 2 deletions(-)

diff --git a/tests/rest/media/v1/test_url_preview.py b/tests/rest/media/v1/test_url_preview.py
index ceefa58d28bd..9d1389958406 100644
--- a/tests/rest/media/v1/test_url_preview.py
+++ b/tests/rest/media/v1/test_url_preview.py
@@ -608,7 +608,7 @@ def test_oembed_photo(self):
         client.dataReceived(
             (
                 b"HTTP/1.0 200 OK\r\nContent-Length: %d\r\n"
-                b'Content-Type: image/png; charset="utf8"\r\n\r\n'
+                b"Content-Type: image/png\r\n\r\n"
             )
             % (len(SMALL_PNG),)
             + SMALL_PNG
@@ -624,7 +624,7 @@ def test_oembed_photo(self):
         self.assertTrue(channel.json_body["og:image"].startswith("mxc://"))
         self.assertEqual(channel.json_body["og:image:height"], 1)
         self.assertEqual(channel.json_body["og:image:width"], 1)
-        self.assertTrue(channel.json_body["og:image:type"].startswith("image/png"))
+        self.assertEqual(channel.json_body["og:image:type"], "image/png")
 
     def test_oembed_rich(self):
         """Test an oEmbed endpoint which returns HTML content via the 'rich' type."""

From 53270531d736021db39d564329bacf68d2ed643a Mon Sep 17 00:00:00 2001
From: Patrick Cloke <patrickc@matrix.org>
Date: Tue, 21 Sep 2021 09:41:26 -0400
Subject: [PATCH 14/19] Clarify when cache_age is None.

---
 synapse/rest/media/v1/oembed.py | 3 +++
 1 file changed, 3 insertions(+)

diff --git a/synapse/rest/media/v1/oembed.py b/synapse/rest/media/v1/oembed.py
index b309c81be0c6..7c003837f7c4 100644
--- a/synapse/rest/media/v1/oembed.py
+++ b/synapse/rest/media/v1/oembed.py
@@ -32,6 +32,9 @@ class OEmbedResult:
     # The Open Graph result (converted from the oEmbed result).
     open_graph_result: JsonDict
     # Number of seconds to cache the content.
+    #
+    # This will be None if no cache-age is provided in the oEmbed response (or
+    # if the oEmbed response cannot be turned into an Open Graph response).
     cache_age: Optional[int]
 
 

From a2a40d7020aa109f034b585891b621e3dd5a5ccc Mon Sep 17 00:00:00 2001
From: Patrick Cloke <patrickc@matrix.org>
Date: Tue, 21 Sep 2021 09:45:50 -0400
Subject: [PATCH 15/19] Cap the amount of time which URL previews will be kept
 to 24 hours.

---
 synapse/rest/media/v1/preview_url_resource.py | 12 ++++++++----
 1 file changed, 8 insertions(+), 4 deletions(-)

diff --git a/synapse/rest/media/v1/preview_url_resource.py b/synapse/rest/media/v1/preview_url_resource.py
index 992abdb51edf..b60842d97264 100644
--- a/synapse/rest/media/v1/preview_url_resource.py
+++ b/synapse/rest/media/v1/preview_url_resource.py
@@ -73,6 +73,7 @@
 OG_TAG_VALUE_MAXLEN = 1000
 
 ONE_HOUR = 60 * 60 * 1000
+ONE_DAY = 24 * ONE_HOUR
 
 
 @attr.s(slots=True, frozen=True, auto_attribs=True)
@@ -265,8 +266,8 @@ async def _do_preview(self, url: str, user: str, ts: int) -> bytes:
 
         logger.debug("got media_info of '%s'", media_info)
 
-        # The timestamp of when this media expires.
-        expiration_ts_ms = media_info.expires + media_info.created_ts_ms
+        # The number of milliseconds that the response should be considered valid.
+        expiration_ms = media_info.expires
 
         if _is_media(media_info.media_type):
             file_id = media_info.filesystem_id
@@ -311,7 +312,7 @@ async def _do_preview(self, url: str, user: str, ts: int) -> bytes:
 
             # Use the cache age from the oEmbed result, instead of the HTTP response.
             if oembed_response.cache_age is not None:
-                expiration_ts_ms = oembed_response.cache_age + media_info.created_ts_ms
+                expiration_ms = oembed_response.cache_age
 
             await self._precache_image_url(user, media_info, og)
 
@@ -335,12 +336,15 @@ async def _do_preview(self, url: str, user: str, ts: int) -> bytes:
 
         jsonog = json_encoder.encode(og)
 
+        # Cap the amount of time to consider a response valid.
+        expiration_ms = min(expiration_ms, ONE_DAY)
+
         # store OG in history-aware DB cache
         await self.store.store_url_cache(
             url,
             media_info.response_code,
             media_info.etag,
-            expiration_ts_ms,
+            media_info.created_ts_ms + expiration_ms,
             jsonog,
             media_info.filesystem_id,
             media_info.created_ts_ms,

From 6643b3f81606b90798dcb6e80c5cf7fb6f605f5a Mon Sep 17 00:00:00 2001
From: Patrick Cloke <patrickc@matrix.org>
Date: Tue, 21 Sep 2021 09:46:12 -0400
Subject: [PATCH 16/19] Fix typo.

---
 synapse/rest/media/v1/oembed.py | 12 ++++++------
 1 file changed, 6 insertions(+), 6 deletions(-)

diff --git a/synapse/rest/media/v1/oembed.py b/synapse/rest/media/v1/oembed.py
index 7c003837f7c4..f3fe3c8eab2a 100644
--- a/synapse/rest/media/v1/oembed.py
+++ b/synapse/rest/media/v1/oembed.py
@@ -121,20 +121,20 @@ def parse_oembed_response(self, url: str, raw_body: str) -> OEmbedResult:
                 cache_age = int(cache_age)
 
             # The results.
-            ppen_graph_response = {"og:title": oembed.get("title")}
+            open_graph_response = {"og:title": oembed.get("title")}
 
             # If a thumbnail exists, use it. Note that dimensions will be calculated later.
             if "thumbnail_url" in oembed:
-                ppen_graph_response["og:image"] = oembed["thumbnail_url"]
+                open_graph_response["og:image"] = oembed["thumbnail_url"]
 
             # Process each type separately.
             oembed_type = oembed["type"]
             if oembed_type == "rich":
-                calc_description_and_urls(ppen_graph_response, oembed["html"])
+                calc_description_and_urls(open_graph_response, oembed["html"])
 
             elif oembed_type == "photo":
                 # If this is a photo, use the full image, not the thumbnail.
-                ppen_graph_response["og:image"] = oembed["url"]
+                open_graph_response["og:image"] = oembed["url"]
 
             else:
                 raise RuntimeError(f"Unknown oEmbed type: {oembed_type}")
@@ -142,10 +142,10 @@ def parse_oembed_response(self, url: str, raw_body: str) -> OEmbedResult:
         except Exception as e:
             # Trap any exception and let the code follow as usual.
             logger.warning(f"Error parsing oEmbed metadata from {url}: {e:r}")
-            ppen_graph_response = {}
+            open_graph_response = {}
             cache_age = None
 
-        return OEmbedResult(ppen_graph_response, cache_age)
+        return OEmbedResult(open_graph_response, cache_age)
 
 
 def calc_description_and_urls(open_graph_response: JsonDict, html_body: str) -> None:

From 1f975d66ace4382eaaa481e2b29eceb340196270 Mon Sep 17 00:00:00 2001
From: Patrick Cloke <patrickc@matrix.org>
Date: Tue, 21 Sep 2021 11:43:09 -0400
Subject: [PATCH 17/19] Pass raw-bytes to the oEmbed code.

---
 synapse/rest/media/v1/oembed.py               | 7 ++++---
 synapse/rest/media/v1/preview_url_resource.py | 8 +++-----
 2 files changed, 7 insertions(+), 8 deletions(-)

diff --git a/synapse/rest/media/v1/oembed.py b/synapse/rest/media/v1/oembed.py
index f3fe3c8eab2a..1693af673a06 100644
--- a/synapse/rest/media/v1/oembed.py
+++ b/synapse/rest/media/v1/oembed.py
@@ -94,21 +94,22 @@ def get_oembed_url(self, url: str) -> Optional[str]:
         # No match.
         return None
 
-    def parse_oembed_response(self, url: str, raw_body: str) -> OEmbedResult:
+    def parse_oembed_response(self, url: str, raw_body: bytes) -> OEmbedResult:
         """
         Parse the oEmbed response into an Open Graph response.
 
         Args:
             url: The URL which is being previewed (not the one which was
                 requested).
-            raw_body: The oEmbed response as JSON.
+            raw_body: The oEmbed response as JSON encoded as bytes.
 
         Returns:
             json-encoded Open Graph data
         """
 
         try:
-            oembed = json_decoder.decode(raw_body)
+            # oEmbed responses *must* be UTF-8 according to the spec.
+            oembed = json_decoder.decode(raw_body.decode("utf-8"))
 
             # Ensure there's a version of 1.0.
             oembed_version = oembed["version"]
diff --git a/synapse/rest/media/v1/preview_url_resource.py b/synapse/rest/media/v1/preview_url_resource.py
index b60842d97264..438dca364130 100644
--- a/synapse/rest/media/v1/preview_url_resource.py
+++ b/synapse/rest/media/v1/preview_url_resource.py
@@ -302,12 +302,10 @@ async def _do_preview(self, url: str, user: str, ts: int) -> bytes:
 
         elif oembed_url and _is_json(media_info.media_type):
             # Handle an oEmbed response.
-            with open(media_info.filename, "r") as utf8_file:
-                oembed_body = utf8_file.read()
+            with open(media_info.filename, "rb") as file:
+                body = file.read()
 
-            oembed_response = self._oembed.parse_oembed_response(
-                media_info.uri, oembed_body
-            )
+            oembed_response = self._oembed.parse_oembed_response(media_info.uri, body)
             og = oembed_response.open_graph_result
 
             # Use the cache age from the oEmbed result, instead of the HTTP response.

From c31f23b1ce83faf519f2bacb355b9889b3212c29 Mon Sep 17 00:00:00 2001
From: Patrick Cloke <clokep@users.noreply.github.com>
Date: Tue, 21 Sep 2021 11:43:40 -0400
Subject: [PATCH 18/19] Clarify comment.

Co-authored-by: Richard van der Hoff <1389908+richvdh@users.noreply.github.com>
---
 synapse/rest/media/v1/oembed.py | 2 +-
 1 file changed, 1 insertion(+), 1 deletion(-)

diff --git a/synapse/rest/media/v1/oembed.py b/synapse/rest/media/v1/oembed.py
index 1693af673a06..8b74e72655bc 100644
--- a/synapse/rest/media/v1/oembed.py
+++ b/synapse/rest/media/v1/oembed.py
@@ -31,7 +31,7 @@
 class OEmbedResult:
     # The Open Graph result (converted from the oEmbed result).
     open_graph_result: JsonDict
-    # Number of seconds to cache the content.
+    # Number of seconds to cache the content, according to the oEmbed response.
     #
     # This will be None if no cache-age is provided in the oEmbed response (or
     # if the oEmbed response cannot be turned into an Open Graph response).

From 62b36c0e108f7f1ed0dc59304c6f9bd82135e250 Mon Sep 17 00:00:00 2001
From: Patrick Cloke <patrickc@matrix.org>
Date: Tue, 21 Sep 2021 11:45:57 -0400
Subject: [PATCH 19/19] Re-use a constant.

---
 synapse/rest/media/v1/preview_url_resource.py | 2 +-
 1 file changed, 1 insertion(+), 1 deletion(-)

diff --git a/synapse/rest/media/v1/preview_url_resource.py b/synapse/rest/media/v1/preview_url_resource.py
index 438dca364130..0a0b476d2b26 100644
--- a/synapse/rest/media/v1/preview_url_resource.py
+++ b/synapse/rest/media/v1/preview_url_resource.py
@@ -532,7 +532,7 @@ async def _expire_url_cache_data(self) -> None:
         # These may be cached for a bit on the client (i.e., they
         # may have a room open with a preview url thing open).
         # So we wait a couple of days before deleting, just in case.
-        expire_before = now - 2 * 24 * ONE_HOUR
+        expire_before = now - 2 * ONE_DAY
         media_ids = await self.store.get_url_cache_media_before(expire_before)
 
         removed_media = []