diff options
| author | Jan Tuomi <jans.tuomi@gmail.com> | 2023-09-11 10:39:36 +0300 |
|---|---|---|
| committer | Jan Tuomi <jans.tuomi@gmail.com> | 2023-09-11 10:39:36 +0300 |
| commit | 8c54989f182e2485564a071cbae43afb64cb8905 (patch) | |
| tree | 461c0925857039f8f9f3cd56cd9b027ef5e99402 /plugins/FacebookSourcePlugin.py | |
| parent | a8aed014dbbfed67ef56fdda6219e7f17922f383 (diff) | |
Fix html parsing
Diffstat (limited to 'plugins/FacebookSourcePlugin.py')
| -rw-r--r-- | plugins/FacebookSourcePlugin.py | 12 |
1 files changed, 3 insertions, 9 deletions
diff --git a/plugins/FacebookSourcePlugin.py b/plugins/FacebookSourcePlugin.py index a43aee7..55abbc1 100644 --- a/plugins/FacebookSourcePlugin.py +++ b/plugins/FacebookSourcePlugin.py @@ -83,9 +83,7 @@ def fetch_page_posts(email: str, password: str, page_id: str, limit: int) -> lis if cookie_page_resp.status_code >= 400: raise Exception(cookie_page_resp.text) - cookie_page = BeautifulSoup( - cookie_page_resp.text, features=["xml", "lxml", "lxml-xml"] - ) + cookie_page = BeautifulSoup(cookie_page_resp.text, "html.parser") lsd: str = cookie_page.find("input", {"name": "lsd"})["value"] # type: ignore jazoest: str = cookie_page.find("input", {"name": "jazoest"})["value"] # type: ignore @@ -108,9 +106,7 @@ def fetch_page_posts(email: str, password: str, page_id: str, limit: int) -> lis if login_page_resp.status_code >= 400: raise Exception(login_page_resp.text) - login_page = BeautifulSoup( - login_page_resp.text, features=["xml", "lxml", "lxml-xml"] - ) + login_page = BeautifulSoup(login_page_resp.text, "html.parser") lsd: str = login_page.find("input", {"name": "lsd"})["value"] # type: ignore jazoest: str = login_page.find("input", {"name": "jazoest"})["value"] # type: ignore @@ -162,9 +158,7 @@ def fetch_page_posts(email: str, password: str, page_id: str, limit: int) -> lis if timeline_resp.status_code >= 400: raise Exception(timeline_resp.text) - timeline = BeautifulSoup( - timeline_resp.text, features=["xml", "lxml", "lxml-xml"] - ) + timeline = BeautifulSoup(timeline_resp.text, "html.parser") posts = timeline.select("section > article") for post in posts: |
