mirror of
https://github.com/yt-dlp/yt-dlp.git
synced 2026-08-10 14:18:37 +03:00
Compare commits
159
Commits
2021.12.25
...
2022.01.21
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b54e1255ce | ||
|
|
f20d607b0e | ||
|
|
ed40877833 | ||
|
|
935f5a4209 | ||
|
|
6970b6005e | ||
|
|
fc5fa964c7 | ||
|
|
e0ddbd02bd | ||
|
|
0bfc53d05c | ||
|
|
78ab4f447c | ||
|
|
85fee22152 | ||
|
|
ad9158d5f4 | ||
|
|
f81c62a6a4 | ||
|
|
6c73052c0a | ||
|
|
593e43c030 | ||
|
|
8fe514d382 | ||
|
|
b1156c1e59 | ||
|
|
311b6615d8 | ||
|
|
396a76f7bf | ||
|
|
301d07fc4b | ||
|
|
d14cbdd92d | ||
|
|
19b4c74d40 | ||
|
|
135dfa2c7e | ||
|
|
e0585e6562 | ||
|
|
426764371f | ||
|
|
64f36541c9 | ||
|
|
0ff1e0fba3 | ||
|
|
1a20d29552 | ||
|
|
f7085283e1 | ||
|
|
e25ca9b017 | ||
|
|
4259402c56 | ||
|
|
dfb7f2a25d | ||
|
|
42c5458a02 | ||
|
|
ba1c671d2e | ||
|
|
b143e83ec9 | ||
|
|
4a77fb1d6b | ||
|
|
66f7c6a3e0 | ||
|
|
baf599effa | ||
|
|
8bd1c00bf3 | ||
|
|
596379e260 | ||
|
|
b6ce9bb038 | ||
|
|
eea1b0358e | ||
|
|
32b95bb643 | ||
|
|
fdf80059d9 | ||
|
|
aa062713c1 | ||
|
|
71738b1451 | ||
|
|
0bb5ac1ac4 | ||
|
|
77b28f000a | ||
|
|
d57576b9d9 | ||
|
|
11c861702d | ||
|
|
a4a426023d | ||
|
|
3b603dbdf1 | ||
|
|
5df1ac92bd | ||
|
|
b2db8102dc | ||
|
|
e9a6a65a55 | ||
|
|
ed8d87f911 | ||
|
|
397235c52b | ||
|
|
4636548463 | ||
|
|
cb3c5682ae | ||
|
|
7d449fff53 | ||
|
|
80fa6e5327 | ||
|
|
fabb27fcea | ||
|
|
e04938ab88 | ||
|
|
8bcd404818 | ||
|
|
0df11dafdd | ||
|
|
dc5f409cdc | ||
|
|
99d6f9461d | ||
|
|
8130779db6 | ||
|
|
ed5835b451 | ||
|
|
e88e1febd8 | ||
|
|
faca674510 | ||
|
|
0931ba94ab | ||
|
|
b31874334d | ||
|
|
f1150b9e1e | ||
|
|
d6579d532b | ||
|
|
2be56f2242 | ||
|
|
f95a7b93e6 | ||
|
|
62c955efc9 | ||
|
|
0254f16274 | ||
|
|
a70b71e85a | ||
|
|
4c968755fc | ||
|
|
be1f331f21 | ||
|
|
3cf5429a21 | ||
|
|
bfa0e270cf | ||
|
|
f76ca2dd56 | ||
|
|
5f969a78b0 | ||
|
|
443f8de820 | ||
|
|
768145d48a | ||
|
|
976ae3eabb | ||
|
|
f0d785d3ed | ||
|
|
97a6b117d9 | ||
|
|
6f32a0b5b7 | ||
|
|
e8736539f3 | ||
|
|
9c634ef857 | ||
|
|
9f517bb1f3 | ||
|
|
b8eeced286 | ||
|
|
db47787024 | ||
|
|
fdeab99eab | ||
|
|
9e907ebddf | ||
|
|
21df2117e4 | ||
|
|
06e57990f7 | ||
|
|
b62fa6d75f | ||
|
|
be72c62480 | ||
|
|
61e9d9268c | ||
|
|
a13e684813 | ||
|
|
f46e2f9d92 | ||
|
|
9c906919ae | ||
|
|
6020e05d23 | ||
|
|
ebed8b3732 | ||
|
|
1e43a6f733 | ||
|
|
ca30f449a1 | ||
|
|
af3cbd8782 | ||
|
|
7141ced57d | ||
|
|
18c7683d27 | ||
|
|
f5c2c2c9b0 | ||
|
|
8896899216 | ||
|
|
1797b073ed | ||
|
|
4c922dd3fc | ||
|
|
b8e976a445 | ||
|
|
a9f5f5d6eb | ||
|
|
f522573787 | ||
|
|
7592749cbe | ||
|
|
767f999b53 | ||
|
|
8efffafa53 | ||
|
|
26f2aa3db9 | ||
|
|
3464a2727b | ||
|
|
497d77e1aa | ||
|
|
9040e2d6e3 | ||
|
|
6134fbeb65 | ||
|
|
cfcf60ea99 | ||
|
|
4afa3ec4b6 | ||
|
|
11aa91a12f | ||
|
|
abbeeebc4c | ||
|
|
2c539d493a | ||
|
|
042931a507 | ||
|
|
96f13f01a6 | ||
|
|
4b9353239e | ||
|
|
dd5e60b15d | ||
|
|
e540c56f39 | ||
|
|
45d86abeb4 | ||
|
|
f02d24d8d2 | ||
|
|
ceb98323f2 | ||
|
|
7537e35b64 | ||
|
|
1e5c83b26b | ||
|
|
6223f67a8c | ||
|
|
6a34813a0d | ||
|
|
f59f5ef8b6 | ||
|
|
f44afb54ef | ||
|
|
77cee0f188 | ||
|
|
6a17677577 | ||
|
|
ee7b9bdf5d | ||
|
|
185bf31070 | ||
|
|
0b77924a38 | ||
|
|
8126298c1b | ||
|
|
6da22e7d4f | ||
|
|
c62ecf0d90 | ||
|
|
3774f4f427 | ||
|
|
9980d3d213 | ||
|
|
8eb4b1bb8e | ||
|
|
332da56f52 |
@@ -11,7 +11,7 @@ body:
|
|||||||
options:
|
options:
|
||||||
- label: I'm reporting a broken site
|
- label: I'm reporting a broken site
|
||||||
required: true
|
required: true
|
||||||
- label: I've verified that I'm running yt-dlp version **2021.12.25**. ([update instructions](https://github.com/yt-dlp/yt-dlp#update))
|
- label: I've verified that I'm running yt-dlp version **2022.01.21**. ([update instructions](https://github.com/yt-dlp/yt-dlp#update))
|
||||||
required: true
|
required: true
|
||||||
- label: I've checked that all provided URLs are alive and playable in a browser
|
- label: I've checked that all provided URLs are alive and playable in a browser
|
||||||
required: true
|
required: true
|
||||||
@@ -44,19 +44,19 @@ body:
|
|||||||
label: Verbose log
|
label: Verbose log
|
||||||
description: |
|
description: |
|
||||||
Provide the complete verbose output of yt-dlp **that clearly demonstrates the problem**.
|
Provide the complete verbose output of yt-dlp **that clearly demonstrates the problem**.
|
||||||
Add the `-Uv` flag to your command line you run yt-dlp with (`yt-dlp -Uv <your command line>`), copy the WHOLE output and insert it below.
|
Add the `-vU` flag to your command line you run yt-dlp with (`yt-dlp -vU <your command line>`), copy the WHOLE output and insert it below.
|
||||||
It should look similar to this:
|
It should look similar to this:
|
||||||
placeholder: |
|
placeholder: |
|
||||||
[debug] Command-line config: ['-Uv', 'http://www.youtube.com/watch?v=BaW_jenozKc']
|
[debug] Command-line config: ['-vU', 'http://www.youtube.com/watch?v=BaW_jenozKc']
|
||||||
[debug] Portable config file: yt-dlp.conf
|
[debug] Portable config file: yt-dlp.conf
|
||||||
[debug] Portable config: ['-i']
|
[debug] Portable config: ['-i']
|
||||||
[debug] Encodings: locale cp1252, fs utf-8, stdout utf-8, stderr utf-8, pref cp1252
|
[debug] Encodings: locale cp1252, fs utf-8, stdout utf-8, stderr utf-8, pref cp1252
|
||||||
[debug] yt-dlp version 2021.12.25 (exe)
|
[debug] yt-dlp version 2022.01.21 (exe)
|
||||||
[debug] Python version 3.8.8 (CPython 64bit) - Windows-10-10.0.19041-SP0
|
[debug] Python version 3.8.8 (CPython 64bit) - Windows-10-10.0.19041-SP0
|
||||||
[debug] exe versions: ffmpeg 3.0.1, ffprobe 3.0.1
|
[debug] exe versions: ffmpeg 3.0.1, ffprobe 3.0.1
|
||||||
[debug] Optional libraries: Cryptodome, keyring, mutagen, sqlite, websockets
|
[debug] Optional libraries: Cryptodome, keyring, mutagen, sqlite, websockets
|
||||||
[debug] Proxy map: {}
|
[debug] Proxy map: {}
|
||||||
yt-dlp is up to date (2021.12.25)
|
yt-dlp is up to date (2022.01.21)
|
||||||
<more lines>
|
<more lines>
|
||||||
render: shell
|
render: shell
|
||||||
validations:
|
validations:
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ body:
|
|||||||
options:
|
options:
|
||||||
- label: I'm reporting a new site support request
|
- label: I'm reporting a new site support request
|
||||||
required: true
|
required: true
|
||||||
- label: I've verified that I'm running yt-dlp version **2021.12.25**. ([update instructions](https://github.com/yt-dlp/yt-dlp#update))
|
- label: I've verified that I'm running yt-dlp version **2022.01.21**. ([update instructions](https://github.com/yt-dlp/yt-dlp#update))
|
||||||
required: true
|
required: true
|
||||||
- label: I've checked that all provided URLs are alive and playable in a browser
|
- label: I've checked that all provided URLs are alive and playable in a browser
|
||||||
required: true
|
required: true
|
||||||
@@ -55,19 +55,19 @@ body:
|
|||||||
label: Verbose log
|
label: Verbose log
|
||||||
description: |
|
description: |
|
||||||
Provide the complete verbose output **using one of the example URLs provided above**.
|
Provide the complete verbose output **using one of the example URLs provided above**.
|
||||||
Add the `-Uv` flag to your command line you run yt-dlp with (`yt-dlp -Uv <your command line>`), copy the WHOLE output and insert it below.
|
Add the `-vU` flag to your command line you run yt-dlp with (`yt-dlp -vU <your command line>`), copy the WHOLE output and insert it below.
|
||||||
It should look similar to this:
|
It should look similar to this:
|
||||||
placeholder: |
|
placeholder: |
|
||||||
[debug] Command-line config: ['-Uv', 'http://www.youtube.com/watch?v=BaW_jenozKc']
|
[debug] Command-line config: ['-vU', 'http://www.youtube.com/watch?v=BaW_jenozKc']
|
||||||
[debug] Portable config file: yt-dlp.conf
|
[debug] Portable config file: yt-dlp.conf
|
||||||
[debug] Portable config: ['-i']
|
[debug] Portable config: ['-i']
|
||||||
[debug] Encodings: locale cp1252, fs utf-8, stdout utf-8, stderr utf-8, pref cp1252
|
[debug] Encodings: locale cp1252, fs utf-8, stdout utf-8, stderr utf-8, pref cp1252
|
||||||
[debug] yt-dlp version 2021.12.25 (exe)
|
[debug] yt-dlp version 2022.01.21 (exe)
|
||||||
[debug] Python version 3.8.8 (CPython 64bit) - Windows-10-10.0.19041-SP0
|
[debug] Python version 3.8.8 (CPython 64bit) - Windows-10-10.0.19041-SP0
|
||||||
[debug] exe versions: ffmpeg 3.0.1, ffprobe 3.0.1
|
[debug] exe versions: ffmpeg 3.0.1, ffprobe 3.0.1
|
||||||
[debug] Optional libraries: Cryptodome, keyring, mutagen, sqlite, websockets
|
[debug] Optional libraries: Cryptodome, keyring, mutagen, sqlite, websockets
|
||||||
[debug] Proxy map: {}
|
[debug] Proxy map: {}
|
||||||
yt-dlp is up to date (2021.12.25)
|
yt-dlp is up to date (2022.01.21)
|
||||||
<more lines>
|
<more lines>
|
||||||
render: shell
|
render: shell
|
||||||
validations:
|
validations:
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ body:
|
|||||||
options:
|
options:
|
||||||
- label: I'm reporting a site feature request
|
- label: I'm reporting a site feature request
|
||||||
required: true
|
required: true
|
||||||
- label: I've verified that I'm running yt-dlp version **2021.12.25**. ([update instructions](https://github.com/yt-dlp/yt-dlp#update))
|
- label: I've verified that I'm running yt-dlp version **2022.01.21**. ([update instructions](https://github.com/yt-dlp/yt-dlp#update))
|
||||||
required: true
|
required: true
|
||||||
- label: I've checked that all provided URLs are alive and playable in a browser
|
- label: I've checked that all provided URLs are alive and playable in a browser
|
||||||
required: true
|
required: true
|
||||||
@@ -32,7 +32,7 @@ body:
|
|||||||
label: Example URLs
|
label: Example URLs
|
||||||
description: |
|
description: |
|
||||||
Example URLs that can be used to demonstrate the requested feature
|
Example URLs that can be used to demonstrate the requested feature
|
||||||
value: |
|
placeholder: |
|
||||||
https://www.youtube.com/watch?v=BaW_jenozKc
|
https://www.youtube.com/watch?v=BaW_jenozKc
|
||||||
validations:
|
validations:
|
||||||
required: true
|
required: true
|
||||||
@@ -53,19 +53,19 @@ body:
|
|||||||
label: Verbose log
|
label: Verbose log
|
||||||
description: |
|
description: |
|
||||||
Provide the complete verbose output of yt-dlp that demonstrates the need for the enhancement.
|
Provide the complete verbose output of yt-dlp that demonstrates the need for the enhancement.
|
||||||
Add the `-Uv` flag to your command line you run yt-dlp with (`yt-dlp -Uv <your command line>`), copy the WHOLE output and insert it below.
|
Add the `-vU` flag to your command line you run yt-dlp with (`yt-dlp -vU <your command line>`), copy the WHOLE output and insert it below.
|
||||||
It should look similar to this:
|
It should look similar to this:
|
||||||
placeholder: |
|
placeholder: |
|
||||||
[debug] Command-line config: ['-Uv', 'http://www.youtube.com/watch?v=BaW_jenozKc']
|
[debug] Command-line config: ['-vU', 'http://www.youtube.com/watch?v=BaW_jenozKc']
|
||||||
[debug] Portable config file: yt-dlp.conf
|
[debug] Portable config file: yt-dlp.conf
|
||||||
[debug] Portable config: ['-i']
|
[debug] Portable config: ['-i']
|
||||||
[debug] Encodings: locale cp1252, fs utf-8, stdout utf-8, stderr utf-8, pref cp1252
|
[debug] Encodings: locale cp1252, fs utf-8, stdout utf-8, stderr utf-8, pref cp1252
|
||||||
[debug] yt-dlp version 2021.12.25 (exe)
|
[debug] yt-dlp version 2022.01.21 (exe)
|
||||||
[debug] Python version 3.8.8 (CPython 64bit) - Windows-10-10.0.19041-SP0
|
[debug] Python version 3.8.8 (CPython 64bit) - Windows-10-10.0.19041-SP0
|
||||||
[debug] exe versions: ffmpeg 3.0.1, ffprobe 3.0.1
|
[debug] exe versions: ffmpeg 3.0.1, ffprobe 3.0.1
|
||||||
[debug] Optional libraries: Cryptodome, keyring, mutagen, sqlite, websockets
|
[debug] Optional libraries: Cryptodome, keyring, mutagen, sqlite, websockets
|
||||||
[debug] Proxy map: {}
|
[debug] Proxy map: {}
|
||||||
yt-dlp is up to date (2021.12.25)
|
yt-dlp is up to date (2022.01.21)
|
||||||
<more lines>
|
<more lines>
|
||||||
render: shell
|
render: shell
|
||||||
validations:
|
validations:
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ body:
|
|||||||
options:
|
options:
|
||||||
- label: I'm reporting a bug unrelated to a specific site
|
- label: I'm reporting a bug unrelated to a specific site
|
||||||
required: true
|
required: true
|
||||||
- label: I've verified that I'm running yt-dlp version **2021.12.25**. ([update instructions](https://github.com/yt-dlp/yt-dlp#update))
|
- label: I've verified that I'm running yt-dlp version **2022.01.21**. ([update instructions](https://github.com/yt-dlp/yt-dlp#update))
|
||||||
required: true
|
required: true
|
||||||
- label: I've checked that all provided URLs are alive and playable in a browser
|
- label: I've checked that all provided URLs are alive and playable in a browser
|
||||||
required: true
|
required: true
|
||||||
@@ -38,19 +38,19 @@ body:
|
|||||||
label: Verbose log
|
label: Verbose log
|
||||||
description: |
|
description: |
|
||||||
Provide the complete verbose output of yt-dlp **that clearly demonstrates the problem**.
|
Provide the complete verbose output of yt-dlp **that clearly demonstrates the problem**.
|
||||||
Add the `-Uv` flag to **your** command line you run yt-dlp with (`yt-dlp -Uv <your command line>`), copy the WHOLE output and insert it below.
|
Add the `-vU` flag to **your** command line you run yt-dlp with (`yt-dlp -vU <your command line>`), copy the WHOLE output and insert it below.
|
||||||
It should look similar to this:
|
It should look similar to this:
|
||||||
placeholder: |
|
placeholder: |
|
||||||
[debug] Command-line config: ['-Uv', 'http://www.youtube.com/watch?v=BaW_jenozKc']
|
[debug] Command-line config: ['-vU', 'http://www.youtube.com/watch?v=BaW_jenozKc']
|
||||||
[debug] Portable config file: yt-dlp.conf
|
[debug] Portable config file: yt-dlp.conf
|
||||||
[debug] Portable config: ['-i']
|
[debug] Portable config: ['-i']
|
||||||
[debug] Encodings: locale cp1252, fs utf-8, stdout utf-8, stderr utf-8, pref cp1252
|
[debug] Encodings: locale cp1252, fs utf-8, stdout utf-8, stderr utf-8, pref cp1252
|
||||||
[debug] yt-dlp version 2021.12.25 (exe)
|
[debug] yt-dlp version 2022.01.21 (exe)
|
||||||
[debug] Python version 3.8.8 (CPython 64bit) - Windows-10-10.0.19041-SP0
|
[debug] Python version 3.8.8 (CPython 64bit) - Windows-10-10.0.19041-SP0
|
||||||
[debug] exe versions: ffmpeg 3.0.1, ffprobe 3.0.1
|
[debug] exe versions: ffmpeg 3.0.1, ffprobe 3.0.1
|
||||||
[debug] Optional libraries: Cryptodome, keyring, mutagen, sqlite, websockets
|
[debug] Optional libraries: Cryptodome, keyring, mutagen, sqlite, websockets
|
||||||
[debug] Proxy map: {}
|
[debug] Proxy map: {}
|
||||||
yt-dlp is up to date (2021.12.25)
|
yt-dlp is up to date (2022.01.21)
|
||||||
<more lines>
|
<more lines>
|
||||||
render: shell
|
render: shell
|
||||||
validations:
|
validations:
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ body:
|
|||||||
options:
|
options:
|
||||||
- label: I'm reporting a feature request
|
- label: I'm reporting a feature request
|
||||||
required: true
|
required: true
|
||||||
- label: I've verified that I'm running yt-dlp version **2021.12.25**. ([update instructions](https://github.com/yt-dlp/yt-dlp#update))
|
- label: I've verified that I'm running yt-dlp version **2022.01.21**. ([update instructions](https://github.com/yt-dlp/yt-dlp#update))
|
||||||
required: true
|
required: true
|
||||||
- label: I've searched the [bugtracker](https://github.com/yt-dlp/yt-dlp/issues?q=) for similar issues including closed ones. DO NOT post duplicates
|
- label: I've searched the [bugtracker](https://github.com/yt-dlp/yt-dlp/issues?q=) for similar issues including closed ones. DO NOT post duplicates
|
||||||
required: true
|
required: true
|
||||||
|
|||||||
@@ -35,10 +35,10 @@ body:
|
|||||||
label: Verbose log
|
label: Verbose log
|
||||||
description: |
|
description: |
|
||||||
If your question involes a yt-dlp command, provide the complete verbose output of that command.
|
If your question involes a yt-dlp command, provide the complete verbose output of that command.
|
||||||
Add the `-Uv` flag to **your** command line you run yt-dlp with (`yt-dlp -Uv <your command line>`), copy the WHOLE output and insert it below.
|
Add the `-vU` flag to **your** command line you run yt-dlp with (`yt-dlp -vU <your command line>`), copy the WHOLE output and insert it below.
|
||||||
It should look similar to this:
|
It should look similar to this:
|
||||||
placeholder: |
|
placeholder: |
|
||||||
[debug] Command-line config: ['-Uv', 'http://www.youtube.com/watch?v=BaW_jenozKc']
|
[debug] Command-line config: ['-vU', 'http://www.youtube.com/watch?v=BaW_jenozKc']
|
||||||
[debug] Portable config file: yt-dlp.conf
|
[debug] Portable config file: yt-dlp.conf
|
||||||
[debug] Portable config: ['-i']
|
[debug] Portable config: ['-i']
|
||||||
[debug] Encodings: locale cp1252, fs utf-8, stdout utf-8, stderr utf-8, pref cp1252
|
[debug] Encodings: locale cp1252, fs utf-8, stdout utf-8, stderr utf-8, pref cp1252
|
||||||
|
|||||||
@@ -44,10 +44,10 @@ body:
|
|||||||
label: Verbose log
|
label: Verbose log
|
||||||
description: |
|
description: |
|
||||||
Provide the complete verbose output of yt-dlp **that clearly demonstrates the problem**.
|
Provide the complete verbose output of yt-dlp **that clearly demonstrates the problem**.
|
||||||
Add the `-Uv` flag to your command line you run yt-dlp with (`yt-dlp -Uv <your command line>`), copy the WHOLE output and insert it below.
|
Add the `-vU` flag to your command line you run yt-dlp with (`yt-dlp -vU <your command line>`), copy the WHOLE output and insert it below.
|
||||||
It should look similar to this:
|
It should look similar to this:
|
||||||
placeholder: |
|
placeholder: |
|
||||||
[debug] Command-line config: ['-Uv', 'http://www.youtube.com/watch?v=BaW_jenozKc']
|
[debug] Command-line config: ['-vU', 'http://www.youtube.com/watch?v=BaW_jenozKc']
|
||||||
[debug] Portable config file: yt-dlp.conf
|
[debug] Portable config file: yt-dlp.conf
|
||||||
[debug] Portable config: ['-i']
|
[debug] Portable config: ['-i']
|
||||||
[debug] Encodings: locale cp1252, fs utf-8, stdout utf-8, stderr utf-8, pref cp1252
|
[debug] Encodings: locale cp1252, fs utf-8, stdout utf-8, stderr utf-8, pref cp1252
|
||||||
|
|||||||
@@ -55,10 +55,10 @@ body:
|
|||||||
label: Verbose log
|
label: Verbose log
|
||||||
description: |
|
description: |
|
||||||
Provide the complete verbose output **using one of the example URLs provided above**.
|
Provide the complete verbose output **using one of the example URLs provided above**.
|
||||||
Add the `-Uv` flag to your command line you run yt-dlp with (`yt-dlp -Uv <your command line>`), copy the WHOLE output and insert it below.
|
Add the `-vU` flag to your command line you run yt-dlp with (`yt-dlp -vU <your command line>`), copy the WHOLE output and insert it below.
|
||||||
It should look similar to this:
|
It should look similar to this:
|
||||||
placeholder: |
|
placeholder: |
|
||||||
[debug] Command-line config: ['-Uv', 'http://www.youtube.com/watch?v=BaW_jenozKc']
|
[debug] Command-line config: ['-vU', 'http://www.youtube.com/watch?v=BaW_jenozKc']
|
||||||
[debug] Portable config file: yt-dlp.conf
|
[debug] Portable config file: yt-dlp.conf
|
||||||
[debug] Portable config: ['-i']
|
[debug] Portable config: ['-i']
|
||||||
[debug] Encodings: locale cp1252, fs utf-8, stdout utf-8, stderr utf-8, pref cp1252
|
[debug] Encodings: locale cp1252, fs utf-8, stdout utf-8, stderr utf-8, pref cp1252
|
||||||
|
|||||||
@@ -32,7 +32,7 @@ body:
|
|||||||
label: Example URLs
|
label: Example URLs
|
||||||
description: |
|
description: |
|
||||||
Example URLs that can be used to demonstrate the requested feature
|
Example URLs that can be used to demonstrate the requested feature
|
||||||
value: |
|
placeholder: |
|
||||||
https://www.youtube.com/watch?v=BaW_jenozKc
|
https://www.youtube.com/watch?v=BaW_jenozKc
|
||||||
validations:
|
validations:
|
||||||
required: true
|
required: true
|
||||||
@@ -53,10 +53,10 @@ body:
|
|||||||
label: Verbose log
|
label: Verbose log
|
||||||
description: |
|
description: |
|
||||||
Provide the complete verbose output of yt-dlp that demonstrates the need for the enhancement.
|
Provide the complete verbose output of yt-dlp that demonstrates the need for the enhancement.
|
||||||
Add the `-Uv` flag to your command line you run yt-dlp with (`yt-dlp -Uv <your command line>`), copy the WHOLE output and insert it below.
|
Add the `-vU` flag to your command line you run yt-dlp with (`yt-dlp -vU <your command line>`), copy the WHOLE output and insert it below.
|
||||||
It should look similar to this:
|
It should look similar to this:
|
||||||
placeholder: |
|
placeholder: |
|
||||||
[debug] Command-line config: ['-Uv', 'http://www.youtube.com/watch?v=BaW_jenozKc']
|
[debug] Command-line config: ['-vU', 'http://www.youtube.com/watch?v=BaW_jenozKc']
|
||||||
[debug] Portable config file: yt-dlp.conf
|
[debug] Portable config file: yt-dlp.conf
|
||||||
[debug] Portable config: ['-i']
|
[debug] Portable config: ['-i']
|
||||||
[debug] Encodings: locale cp1252, fs utf-8, stdout utf-8, stderr utf-8, pref cp1252
|
[debug] Encodings: locale cp1252, fs utf-8, stdout utf-8, stderr utf-8, pref cp1252
|
||||||
|
|||||||
@@ -38,10 +38,10 @@ body:
|
|||||||
label: Verbose log
|
label: Verbose log
|
||||||
description: |
|
description: |
|
||||||
Provide the complete verbose output of yt-dlp **that clearly demonstrates the problem**.
|
Provide the complete verbose output of yt-dlp **that clearly demonstrates the problem**.
|
||||||
Add the `-Uv` flag to **your** command line you run yt-dlp with (`yt-dlp -Uv <your command line>`), copy the WHOLE output and insert it below.
|
Add the `-vU` flag to **your** command line you run yt-dlp with (`yt-dlp -vU <your command line>`), copy the WHOLE output and insert it below.
|
||||||
It should look similar to this:
|
It should look similar to this:
|
||||||
placeholder: |
|
placeholder: |
|
||||||
[debug] Command-line config: ['-Uv', 'http://www.youtube.com/watch?v=BaW_jenozKc']
|
[debug] Command-line config: ['-vU', 'http://www.youtube.com/watch?v=BaW_jenozKc']
|
||||||
[debug] Portable config file: yt-dlp.conf
|
[debug] Portable config file: yt-dlp.conf
|
||||||
[debug] Portable config: ['-i']
|
[debug] Portable config: ['-i']
|
||||||
[debug] Encodings: locale cp1252, fs utf-8, stdout utf-8, stderr utf-8, pref cp1252
|
[debug] Encodings: locale cp1252, fs utf-8, stdout utf-8, stderr utf-8, pref cp1252
|
||||||
|
|||||||
@@ -35,10 +35,10 @@ body:
|
|||||||
label: Verbose log
|
label: Verbose log
|
||||||
description: |
|
description: |
|
||||||
If your question involes a yt-dlp command, provide the complete verbose output of that command.
|
If your question involes a yt-dlp command, provide the complete verbose output of that command.
|
||||||
Add the `-Uv` flag to **your** command line you run yt-dlp with (`yt-dlp -Uv <your command line>`), copy the WHOLE output and insert it below.
|
Add the `-vU` flag to **your** command line you run yt-dlp with (`yt-dlp -vU <your command line>`), copy the WHOLE output and insert it below.
|
||||||
It should look similar to this:
|
It should look similar to this:
|
||||||
placeholder: |
|
placeholder: |
|
||||||
[debug] Command-line config: ['-Uv', 'http://www.youtube.com/watch?v=BaW_jenozKc']
|
[debug] Command-line config: ['-vU', 'http://www.youtube.com/watch?v=BaW_jenozKc']
|
||||||
[debug] Portable config file: yt-dlp.conf
|
[debug] Portable config file: yt-dlp.conf
|
||||||
[debug] Portable config: ['-i']
|
[debug] Portable config: ['-i']
|
||||||
[debug] Encodings: locale cp1252, fs utf-8, stdout utf-8, stderr utf-8, pref cp1252
|
[debug] Encodings: locale cp1252, fs utf-8, stdout utf-8, stderr utf-8, pref cp1252
|
||||||
|
|||||||
+11
-15
@@ -96,7 +96,7 @@ jobs:
|
|||||||
env:
|
env:
|
||||||
BREW_TOKEN: ${{ secrets.BREW_TOKEN }}
|
BREW_TOKEN: ${{ secrets.BREW_TOKEN }}
|
||||||
if: "env.BREW_TOKEN != ''"
|
if: "env.BREW_TOKEN != ''"
|
||||||
uses: webfactory/ssh-agent@v0.5.3
|
uses: yt-dlp/ssh-agent@v0.5.3
|
||||||
with:
|
with:
|
||||||
ssh-private-key: ${{ env.BREW_TOKEN }}
|
ssh-private-key: ${{ env.BREW_TOKEN }}
|
||||||
- name: Update Homebrew Formulae
|
- name: Update Homebrew Formulae
|
||||||
@@ -165,7 +165,7 @@ jobs:
|
|||||||
- name: Install Requirements
|
- name: Install Requirements
|
||||||
run: |
|
run: |
|
||||||
brew install coreutils
|
brew install coreutils
|
||||||
/usr/bin/python3 -m pip install -U --user pip Pyinstaller==4.5.1 mutagen pycryptodomex websockets
|
/usr/bin/python3 -m pip install -U --user pip Pyinstaller==4.5.1 -r requirements.txt
|
||||||
- name: Bump version
|
- name: Bump version
|
||||||
id: bump_version
|
id: bump_version
|
||||||
run: /usr/bin/python3 devscripts/update-version.py
|
run: /usr/bin/python3 devscripts/update-version.py
|
||||||
@@ -192,11 +192,9 @@ jobs:
|
|||||||
run: echo "::set-output name=sha512_macos::$(sha512sum dist/yt-dlp_macos | awk '{print $1}')"
|
run: echo "::set-output name=sha512_macos::$(sha512sum dist/yt-dlp_macos | awk '{print $1}')"
|
||||||
|
|
||||||
- name: Run PyInstaller Script with --onedir
|
- name: Run PyInstaller Script with --onedir
|
||||||
run: /usr/bin/python3 pyinst.py --target-architecture universal2 --onedir
|
run: |
|
||||||
- uses: papeloto/action-zip@v1
|
/usr/bin/python3 pyinst.py --target-architecture universal2 --onedir
|
||||||
with:
|
zip ./dist/yt-dlp_macos.zip ./dist/yt-dlp_macos
|
||||||
files: ./dist/yt-dlp_macos
|
|
||||||
dest: ./dist/yt-dlp_macos.zip
|
|
||||||
- name: Upload yt-dlp MacOS onedir
|
- name: Upload yt-dlp MacOS onedir
|
||||||
id: upload-release-macos-zip
|
id: upload-release-macos-zip
|
||||||
uses: actions/upload-release-asset@v1
|
uses: actions/upload-release-asset@v1
|
||||||
@@ -210,7 +208,7 @@ jobs:
|
|||||||
- name: Get SHA2-256SUMS for yt-dlp_macos.zip
|
- name: Get SHA2-256SUMS for yt-dlp_macos.zip
|
||||||
id: sha256_macos_zip
|
id: sha256_macos_zip
|
||||||
run: echo "::set-output name=sha256_macos_zip::$(sha256sum dist/yt-dlp_macos.zip | awk '{print $1}')"
|
run: echo "::set-output name=sha256_macos_zip::$(sha256sum dist/yt-dlp_macos.zip | awk '{print $1}')"
|
||||||
- name: Get SHA2-512SUMS for yt-dlp_macos
|
- name: Get SHA2-512SUMS for yt-dlp_macos.zip
|
||||||
id: sha512_macos_zip
|
id: sha512_macos_zip
|
||||||
run: echo "::set-output name=sha512_macos_zip::$(sha512sum dist/yt-dlp_macos.zip | awk '{print $1}')"
|
run: echo "::set-output name=sha512_macos_zip::$(sha512sum dist/yt-dlp_macos.zip | awk '{print $1}')"
|
||||||
|
|
||||||
@@ -236,7 +234,7 @@ jobs:
|
|||||||
# Custom pyinstaller built with https://github.com/yt-dlp/pyinstaller-builds
|
# Custom pyinstaller built with https://github.com/yt-dlp/pyinstaller-builds
|
||||||
run: |
|
run: |
|
||||||
python -m pip install --upgrade pip setuptools wheel py2exe
|
python -m pip install --upgrade pip setuptools wheel py2exe
|
||||||
pip install "https://yt-dlp.github.io/Pyinstaller-Builds/x86_64/pyinstaller-4.5.1-py3-none-any.whl" mutagen pycryptodomex websockets
|
pip install "https://yt-dlp.github.io/Pyinstaller-Builds/x86_64/pyinstaller-4.5.1-py3-none-any.whl" -r requirements.txt
|
||||||
- name: Bump version
|
- name: Bump version
|
||||||
id: bump_version
|
id: bump_version
|
||||||
env:
|
env:
|
||||||
@@ -265,11 +263,9 @@ jobs:
|
|||||||
run: echo "::set-output name=sha512_win::$((Get-FileHash dist\yt-dlp.exe -Algorithm SHA512).Hash.ToLower())"
|
run: echo "::set-output name=sha512_win::$((Get-FileHash dist\yt-dlp.exe -Algorithm SHA512).Hash.ToLower())"
|
||||||
|
|
||||||
- name: Run PyInstaller Script with --onedir
|
- name: Run PyInstaller Script with --onedir
|
||||||
run: python pyinst.py --onedir
|
run: |
|
||||||
- uses: papeloto/action-zip@v1
|
python pyinst.py --onedir
|
||||||
with:
|
Compress-Archive -LiteralPath ./dist/yt-dlp -DestinationPath ./dist/yt-dlp_win.zip
|
||||||
files: ./dist/yt-dlp
|
|
||||||
dest: ./dist/yt-dlp_win.zip
|
|
||||||
- name: Upload yt-dlp Windows onedir
|
- name: Upload yt-dlp Windows onedir
|
||||||
id: upload-release-windows-zip
|
id: upload-release-windows-zip
|
||||||
uses: actions/upload-release-asset@v1
|
uses: actions/upload-release-asset@v1
|
||||||
@@ -325,7 +321,7 @@ jobs:
|
|||||||
- name: Install Requirements
|
- name: Install Requirements
|
||||||
run: |
|
run: |
|
||||||
python -m pip install --upgrade pip setuptools wheel
|
python -m pip install --upgrade pip setuptools wheel
|
||||||
pip install "https://yt-dlp.github.io/Pyinstaller-Builds/i686/pyinstaller-4.5.1-py3-none-any.whl" mutagen pycryptodomex websockets
|
pip install "https://yt-dlp.github.io/Pyinstaller-Builds/i686/pyinstaller-4.5.1-py3-none-any.whl" -r requirements.txt
|
||||||
- name: Bump version
|
- name: Bump version
|
||||||
id: bump_version
|
id: bump_version
|
||||||
env:
|
env:
|
||||||
|
|||||||
@@ -14,7 +14,10 @@ cookies
|
|||||||
*.frag.urls
|
*.frag.urls
|
||||||
*.info.json
|
*.info.json
|
||||||
*.live_chat.json
|
*.live_chat.json
|
||||||
|
*.meta
|
||||||
*.part*
|
*.part*
|
||||||
|
*.tmp
|
||||||
|
*.temp
|
||||||
*.unknown_video
|
*.unknown_video
|
||||||
*.ytdl
|
*.ytdl
|
||||||
.cache/
|
.cache/
|
||||||
|
|||||||
+95
-8
@@ -19,6 +19,7 @@
|
|||||||
- [Provide fallbacks](#provide-fallbacks)
|
- [Provide fallbacks](#provide-fallbacks)
|
||||||
- [Regular expressions](#regular-expressions)
|
- [Regular expressions](#regular-expressions)
|
||||||
- [Long lines policy](#long-lines-policy)
|
- [Long lines policy](#long-lines-policy)
|
||||||
|
- [Quotes](#quotes)
|
||||||
- [Inline values](#inline-values)
|
- [Inline values](#inline-values)
|
||||||
- [Collapse fallbacks](#collapse-fallbacks)
|
- [Collapse fallbacks](#collapse-fallbacks)
|
||||||
- [Trailing parentheses](#trailing-parentheses)
|
- [Trailing parentheses](#trailing-parentheses)
|
||||||
@@ -31,9 +32,9 @@
|
|||||||
|
|
||||||
Bugs and suggestions should be reported at: [yt-dlp/yt-dlp/issues](https://github.com/yt-dlp/yt-dlp/issues). Unless you were prompted to or there is another pertinent reason (e.g. GitHub fails to accept the bug report), please do not send bug reports via personal email. For discussions, join us in our [discord server](https://discord.gg/H5MNcFW63r).
|
Bugs and suggestions should be reported at: [yt-dlp/yt-dlp/issues](https://github.com/yt-dlp/yt-dlp/issues). Unless you were prompted to or there is another pertinent reason (e.g. GitHub fails to accept the bug report), please do not send bug reports via personal email. For discussions, join us in our [discord server](https://discord.gg/H5MNcFW63r).
|
||||||
|
|
||||||
**Please include the full output of yt-dlp when run with `-Uv`**, i.e. **add** `-Uv` flag to **your command line**, copy the **whole** output and post it in the issue body wrapped in \`\`\` for better formatting. It should look similar to this:
|
**Please include the full output of yt-dlp when run with `-vU`**, i.e. **add** `-vU` flag to **your command line**, copy the **whole** output and post it in the issue body wrapped in \`\`\` for better formatting. It should look similar to this:
|
||||||
```
|
```
|
||||||
$ yt-dlp -Uv <your command line>
|
$ yt-dlp -vU <your command line>
|
||||||
[debug] Command-line config: ['-v', 'demo.com']
|
[debug] Command-line config: ['-v', 'demo.com']
|
||||||
[debug] Encodings: locale UTF-8, fs utf-8, out utf-8, pref UTF-8
|
[debug] Encodings: locale UTF-8, fs utf-8, out utf-8, pref UTF-8
|
||||||
[debug] yt-dlp version 2021.09.25 (zip)
|
[debug] yt-dlp version 2021.09.25 (zip)
|
||||||
@@ -64,7 +65,7 @@ So please elaborate on what feature you are requesting, or what bug you want to
|
|||||||
|
|
||||||
If your report is shorter than two lines, it is almost certainly missing some of these, which makes it hard for us to respond to it. We're often too polite to close the issue outright, but the missing info makes misinterpretation likely. We often get frustrated by these issues, since the only possible way for us to move forward on them is to ask for clarification over and over.
|
If your report is shorter than two lines, it is almost certainly missing some of these, which makes it hard for us to respond to it. We're often too polite to close the issue outright, but the missing info makes misinterpretation likely. We often get frustrated by these issues, since the only possible way for us to move forward on them is to ask for clarification over and over.
|
||||||
|
|
||||||
For bug reports, this means that your report should contain the **complete** output of yt-dlp when called with the `-Uv` flag. The error message you get for (most) bugs even says so, but you would not believe how many of our bug reports do not contain this information.
|
For bug reports, this means that your report should contain the **complete** output of yt-dlp when called with the `-vU` flag. The error message you get for (most) bugs even says so, but you would not believe how many of our bug reports do not contain this information.
|
||||||
|
|
||||||
If the error is `ERROR: Unable to extract ...` and you cannot reproduce it from multiple countries, add `--write-pages` and upload the `.dump` files you get [somewhere](https://gist.github.com).
|
If the error is `ERROR: Unable to extract ...` and you cannot reproduce it from multiple countries, add `--write-pages` and upload the `.dump` files you get [somewhere](https://gist.github.com).
|
||||||
|
|
||||||
@@ -452,10 +453,14 @@ Here the presence or absence of other attributes including `style` is irrelevent
|
|||||||
|
|
||||||
### Long lines policy
|
### Long lines policy
|
||||||
|
|
||||||
There is a soft limit to keep lines of code under 100 characters long. This means it should be respected if possible and if it does not make readability and code maintenance worse. Sometimes, it may be reasonable to go upto 120 characters and sometimes even 80 can be unreadable. Keep in mind that this is not a hard limit and is just one of many tools to make the code more readable
|
There is a soft limit to keep lines of code under 100 characters long. This means it should be respected if possible and if it does not make readability and code maintenance worse. Sometimes, it may be reasonable to go upto 120 characters and sometimes even 80 can be unreadable. Keep in mind that this is not a hard limit and is just one of many tools to make the code more readable.
|
||||||
|
|
||||||
For example, you should **never** split long string literals like URLs or some other often copied entities over multiple lines to fit this limit:
|
For example, you should **never** split long string literals like URLs or some other often copied entities over multiple lines to fit this limit:
|
||||||
|
|
||||||
|
Conversely, don't unecessarily split small lines further. As a rule of thumb, if removing the line split keeps the code under 80 characters, it should be a single line.
|
||||||
|
|
||||||
|
##### Examples
|
||||||
|
|
||||||
Correct:
|
Correct:
|
||||||
|
|
||||||
```python
|
```python
|
||||||
@@ -469,6 +474,47 @@ Incorrect:
|
|||||||
'PLMYEtVRpaqY00V9W81Cwmzp6N6vZqfUKD4'
|
'PLMYEtVRpaqY00V9W81Cwmzp6N6vZqfUKD4'
|
||||||
```
|
```
|
||||||
|
|
||||||
|
Correct:
|
||||||
|
|
||||||
|
```python
|
||||||
|
uploader = traverse_obj(info, ('uploader', 'name'), ('author', 'fullname'))
|
||||||
|
```
|
||||||
|
|
||||||
|
Incorrect:
|
||||||
|
|
||||||
|
```python
|
||||||
|
uploader = traverse_obj(
|
||||||
|
info,
|
||||||
|
('uploader', 'name'),
|
||||||
|
('author', 'fullname'))
|
||||||
|
```
|
||||||
|
|
||||||
|
Correct:
|
||||||
|
|
||||||
|
```python
|
||||||
|
formats = self._extract_m3u8_formats(
|
||||||
|
m3u8_url, video_id, 'mp4', 'm3u8_native', m3u8_id='hls',
|
||||||
|
note='Downloading HD m3u8 information', errnote='Unable to download HD m3u8 information')
|
||||||
|
```
|
||||||
|
|
||||||
|
Incorrect:
|
||||||
|
|
||||||
|
```python
|
||||||
|
formats = self._extract_m3u8_formats(m3u8_url,
|
||||||
|
video_id,
|
||||||
|
'mp4',
|
||||||
|
'm3u8_native',
|
||||||
|
m3u8_id='hls',
|
||||||
|
note='Downloading HD m3u8 information',
|
||||||
|
errnote='Unable to download HD m3u8 information')
|
||||||
|
```
|
||||||
|
|
||||||
|
|
||||||
|
### Quotes
|
||||||
|
|
||||||
|
Always use single quotes for strings (even if the string has `'`) and double quotes for docstrings. Use `'''` only for multi-line strings. An exception can be made if a string has multiple single quotes in it and escaping makes it significantly harder to read. For f-strings, use you can use double quotes on the inside. But avoid f-strings that have too many quotes inside.
|
||||||
|
|
||||||
|
|
||||||
### Inline values
|
### Inline values
|
||||||
|
|
||||||
Extracting variables is acceptable for reducing code duplication and improving readability of complex expressions. However, you should avoid extracting variables used only once and moving them to opposite parts of the extractor file, which makes reading the linear flow difficult.
|
Extracting variables is acceptable for reducing code duplication and improving readability of complex expressions. However, you should avoid extracting variables used only once and moving them to opposite parts of the extractor file, which makes reading the linear flow difficult.
|
||||||
@@ -518,27 +564,68 @@ Methods supporting list of patterns are: `_search_regex`, `_html_search_regex`,
|
|||||||
|
|
||||||
### Trailing parentheses
|
### Trailing parentheses
|
||||||
|
|
||||||
Always move trailing parentheses after the last argument.
|
Always move trailing parentheses used for grouping/functions after the last argument. On the other hand, literal list/tuple/dict/set should closed be in a new line. Generators and list/dict comprehensions may use either style
|
||||||
|
|
||||||
Note that this *does not* apply to braces `}` or square brackets `]` both of which should closed be in a new line
|
#### Examples
|
||||||
|
|
||||||
#### Example
|
|
||||||
|
|
||||||
Correct:
|
Correct:
|
||||||
|
|
||||||
```python
|
```python
|
||||||
|
url = try_get(
|
||||||
|
info,
|
||||||
lambda x: x['ResultSet']['Result'][0]['VideoUrlSet']['VideoUrl'],
|
lambda x: x['ResultSet']['Result'][0]['VideoUrlSet']['VideoUrl'],
|
||||||
list)
|
list)
|
||||||
```
|
```
|
||||||
|
Correct:
|
||||||
|
|
||||||
|
```python
|
||||||
|
url = try_get(info,
|
||||||
|
lambda x: x['ResultSet']['Result'][0]['VideoUrlSet']['VideoUrl'],
|
||||||
|
list)
|
||||||
|
```
|
||||||
|
|
||||||
Incorrect:
|
Incorrect:
|
||||||
|
|
||||||
```python
|
```python
|
||||||
|
url = try_get(
|
||||||
|
info,
|
||||||
lambda x: x['ResultSet']['Result'][0]['VideoUrlSet']['VideoUrl'],
|
lambda x: x['ResultSet']['Result'][0]['VideoUrlSet']['VideoUrl'],
|
||||||
list,
|
list,
|
||||||
)
|
)
|
||||||
```
|
```
|
||||||
|
|
||||||
|
Correct:
|
||||||
|
|
||||||
|
```python
|
||||||
|
f = {
|
||||||
|
'url': url,
|
||||||
|
'format_id': format_id,
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Incorrect:
|
||||||
|
|
||||||
|
```python
|
||||||
|
f = {'url': url,
|
||||||
|
'format_id': format_id}
|
||||||
|
```
|
||||||
|
|
||||||
|
Correct:
|
||||||
|
|
||||||
|
```python
|
||||||
|
formats = [process_formats(f) for f in format_data
|
||||||
|
if f.get('type') in ('hls', 'dash', 'direct') and f.get('downloadable')]
|
||||||
|
```
|
||||||
|
|
||||||
|
Correct:
|
||||||
|
|
||||||
|
```python
|
||||||
|
formats = [
|
||||||
|
process_formats(f) for f in format_data
|
||||||
|
if f.get('type') in ('hls', 'dash', 'direct') and f.get('downloadable')
|
||||||
|
]
|
||||||
|
```
|
||||||
|
|
||||||
|
|
||||||
### Use convenience conversion and parsing functions
|
### Use convenience conversion and parsing functions
|
||||||
|
|
||||||
|
|||||||
+14
-1
@@ -2,6 +2,7 @@ pukkandan (owner)
|
|||||||
shirt-dev (collaborator)
|
shirt-dev (collaborator)
|
||||||
coletdjnz/colethedj (collaborator)
|
coletdjnz/colethedj (collaborator)
|
||||||
Ashish0804 (collaborator)
|
Ashish0804 (collaborator)
|
||||||
|
nao20010128nao/Lesmiscore (collaborator)
|
||||||
h-h-h-h
|
h-h-h-h
|
||||||
pauldubois98
|
pauldubois98
|
||||||
nixxo
|
nixxo
|
||||||
@@ -19,7 +20,6 @@ samiksome
|
|||||||
alxnull
|
alxnull
|
||||||
FelixFrog
|
FelixFrog
|
||||||
Zocker1999NET
|
Zocker1999NET
|
||||||
nao20010128nao
|
|
||||||
kurumigi
|
kurumigi
|
||||||
bbepis
|
bbepis
|
||||||
animelover1984/horahoradev
|
animelover1984/horahoradev
|
||||||
@@ -177,3 +177,16 @@ Sematre
|
|||||||
jaller94
|
jaller94
|
||||||
r5d
|
r5d
|
||||||
julien-hadleyjack
|
julien-hadleyjack
|
||||||
|
git-anony-mouse
|
||||||
|
mdawar
|
||||||
|
trassshhub
|
||||||
|
foghawk
|
||||||
|
k3ns1n
|
||||||
|
teridon
|
||||||
|
mozlima
|
||||||
|
timendum
|
||||||
|
ischmidt20
|
||||||
|
CreaValix
|
||||||
|
sian1468
|
||||||
|
arkamar
|
||||||
|
hyano
|
||||||
|
|||||||
+141
-1
@@ -11,6 +11,146 @@
|
|||||||
-->
|
-->
|
||||||
|
|
||||||
|
|
||||||
|
### 2021.01.21
|
||||||
|
|
||||||
|
* Add option `--concat-playlist` to **concat videos in a playlist**
|
||||||
|
* Allow **multiple and nested configuration files**
|
||||||
|
* Add more post-processing stages (`after_video`, `playlist`)
|
||||||
|
* Allow `--exec` to be run at any post-processing stage (Deprecates `--exec-before-download`)
|
||||||
|
* Allow `--print` to be run at any post-processing stage
|
||||||
|
* Allow listing formats, thumbnails, subtitles using `--print` by [pukkandan](https://github.com/pukkandan), [Zirro](https://github.com/Zirro)
|
||||||
|
* Add fields `video_autonumber`, `modified_date`, `modified_timestamp`, `playlist_count`, `channel_follower_count`
|
||||||
|
* Add key `requested_downloads` in the root `info_dict`
|
||||||
|
* Write `download_archive` only after all formats are downloaded
|
||||||
|
* [FfmpegMetadata] Allow setting metadata of individual streams using `meta<n>_` prefix
|
||||||
|
* Add option `--legacy-server-connect` by [xtkoba](https://github.com/xtkoba)
|
||||||
|
* Allow escaped `,` in `--extractor-args`
|
||||||
|
* Allow unicode characters in `info.json`
|
||||||
|
* Check for existing thumbnail/subtitle in final directory
|
||||||
|
* Don't treat empty containers as `None` in `sanitize_info`
|
||||||
|
* Fix `-s --ignore-no-formats --force-write-archive`
|
||||||
|
* Fix live title for multiple formats
|
||||||
|
* List playlist thumbnails in `--list-thumbnails`
|
||||||
|
* Raise error if subtitle download fails
|
||||||
|
* [cookies] Fix bug when keyring is unspecified
|
||||||
|
* [ffmpeg] Ignore unknown streams, standardize use of `-map 0`
|
||||||
|
* [outtmpl] Alternate form for `D` and fix suffix's case
|
||||||
|
* [utils] Add `Sec-Fetch-Mode` to `std_headers`
|
||||||
|
* [utils] Fix `format_bytes` output for Bytes by [pukkandan](https://github.com/pukkandan), [mdawar](https://github.com/mdawar)
|
||||||
|
* [utils] Handle `ss:xxx` in `parse_duration`
|
||||||
|
* [utils] Improve parsing for nested HTML elements by [zmousm](https://github.com/zmousm), [pukkandan](https://github.com/pukkandan)
|
||||||
|
* [utils] Use key `None` in `traverse_obj` to return as-is
|
||||||
|
* [extractor] Detect more subtitle codecs in MPD manifests by [fstirlitz](https://github.com/fstirlitz)
|
||||||
|
* [extractor] Extract chapters from JSON-LD by [iw0nderhow](https://github.com/iw0nderhow), [pukkandan](https://github.com/pukkandan)
|
||||||
|
* [extractor] Extract thumbnails from JSON-LD by [nixxo](https://github.com/nixxo)
|
||||||
|
* [extractor] Improve `url_result` and related
|
||||||
|
* [generic] Improve KVS player extraction by [trassshhub](https://github.com/trassshhub)
|
||||||
|
* [build] Reduce dependency on third party workflows
|
||||||
|
* [extractor,cleanup] Use `_search_nextjs_data`, `format_field`
|
||||||
|
* [cleanup] Minor fixes and cleanup
|
||||||
|
* [docs] Improvements
|
||||||
|
* [test] Fix TestVerboseOutput
|
||||||
|
* [afreecatv] Add livestreams extractor by [wlritchi](https://github.com/wlritchi)
|
||||||
|
* [callin] Add extractor by [foghawk](https://github.com/foghawk)
|
||||||
|
* [CrowdBunker] Add extractors by [Ashish0804](https://github.com/Ashish0804)
|
||||||
|
* [daftsex] Add extractors by [k3ns1n](https://github.com/k3ns1n)
|
||||||
|
* [digitalconcerthall] Add extractor by [teridon](https://github.com/teridon)
|
||||||
|
* [Drooble] Add extractor by [u-spec-png](https://github.com/u-spec-png)
|
||||||
|
* [EuropeanTour] Add extractor by [Ashish0804](https://github.com/Ashish0804)
|
||||||
|
* [iq.com] Add extractors by [MinePlayersPE](https://github.com/MinePlayersPE)
|
||||||
|
* [KelbyOne] Add extractor by [Ashish0804](https://github.com/Ashish0804)
|
||||||
|
* [LnkIE] Add extractor by [Ashish0804](https://github.com/Ashish0804)
|
||||||
|
* [MainStreaming] Add extractor by [coletdjnz](https://github.com/coletdjnz)
|
||||||
|
* [megatvcom] Add extractors by [zmousm](https://github.com/zmousm)
|
||||||
|
* [Newsy] Add extractor by [Ashish0804](https://github.com/Ashish0804)
|
||||||
|
* [noodlemagazine] Add extractor by [trassshhub](https://github.com/trassshhub)
|
||||||
|
* [PokerGo] Add extractors by [Ashish0804](https://github.com/Ashish0804)
|
||||||
|
* [Pornez] Add extractor by [mozlima](https://github.com/mozlima)
|
||||||
|
* [PRX] Add Extractors by [coletdjnz](https://github.com/coletdjnz)
|
||||||
|
* [RTNews] Add extractor by [Ashish0804](https://github.com/Ashish0804)
|
||||||
|
* [Rule34video] Add extractor by [trassshhub](https://github.com/trassshhub)
|
||||||
|
* [tvopengr] Add extractors by [zmousm](https://github.com/zmousm)
|
||||||
|
* [Vimm] Add extractor by [alerikaisattera](https://github.com/alerikaisattera)
|
||||||
|
* [glomex] Add extractors by [zmousm](https://github.com/zmousm)
|
||||||
|
* [instagram] Add story/highlight extractor by [u-spec-png](https://github.com/u-spec-png)
|
||||||
|
* [openrec] Add movie extractor by [Lesmiscore](https://github.com/Lesmiscore)
|
||||||
|
* [rai] Add Raiplaysound extractors by [nixxo](https://github.com/nixxo), [pukkandan](https://github.com/pukkandan)
|
||||||
|
* [aparat] Fix extractor
|
||||||
|
* [ard] Extract subtitles by [fstirlitz](https://github.com/fstirlitz)
|
||||||
|
* [BiliIntl] Add login by [MinePlayersPE](https://github.com/MinePlayersPE)
|
||||||
|
* [CeskaTelevize] Use `http` for manifests
|
||||||
|
* [CTVNewsIE] Add fallback for video search by [Ashish0804](https://github.com/Ashish0804)
|
||||||
|
* [dplay] Migrate DiscoveryPlusItaly to DiscoveryPlus by [timendum](https://github.com/timendum)
|
||||||
|
* [dplay] Re-structure DiscoveryPlus extractors
|
||||||
|
* [Dropbox] Support password protected files and more formats by [zenerdi0de](https://github.com/zenerdi0de)
|
||||||
|
* [facebook] Fix extraction from groups
|
||||||
|
* [facebook] Improve title and uploader extraction
|
||||||
|
* [facebook] Parse dash manifests
|
||||||
|
* [fox] Extract m3u8 from preview by [ischmidt20](https://github.com/ischmidt20)
|
||||||
|
* [funk] Support origin URLs
|
||||||
|
* [gfycat] Fix `uploader`
|
||||||
|
* [gfycat] Support embeds by [coletdjnz](https://github.com/coletdjnz)
|
||||||
|
* [hotstar] Add extractor args to ignore tags by [Ashish0804](https://github.com/Ashish0804)
|
||||||
|
* [hrfernsehen] Fix ardloader extraction by [CreaValix](https://github.com/CreaValix)
|
||||||
|
* [instagram] Fix username extraction for stories and highlights by [nyuszika7h](https://github.com/nyuszika7h)
|
||||||
|
* [kakao] Detect geo-restriction
|
||||||
|
* [line] Remove `tv.line.me` by [sian1468](https://github.com/sian1468)
|
||||||
|
* [mixch] Add `MixchArchiveIE` by [Lesmiscore](https://github.com/Lesmiscore)
|
||||||
|
* [mixcloud] Detect restrictions by [llacb47](https://github.com/llacb47)
|
||||||
|
* [NBCSports] Fix extraction of platform URLs by [ischmidt20](https://github.com/ischmidt20)
|
||||||
|
* [Nexx] Extract more metadata by [MinePlayersPE](https://github.com/MinePlayersPE)
|
||||||
|
* [Nexx] Support 3q CDN by [MinePlayersPE](https://github.com/MinePlayersPE)
|
||||||
|
* [pbs] de-prioritize AD formats
|
||||||
|
* [PornHub,YouTube] Refresh onion addresses by [unit193](https://github.com/unit193)
|
||||||
|
* [RedBullTV] Parse subtitles from manifest by [Ashish0804](https://github.com/Ashish0804)
|
||||||
|
* [streamcz] Fix extractor by [arkamar](https://github.com/arkamar), [pukkandan](https://github.com/pukkandan)
|
||||||
|
* [Ted] Rewrite extractor by [pukkandan](https://github.com/pukkandan), [trassshhub](https://github.com/trassshhub)
|
||||||
|
* [Theta] Fix valid URL by [alerikaisattera](https://github.com/alerikaisattera)
|
||||||
|
* [ThisOldHouseIE] Add support for premium videos by [Ashish0804](https://github.com/Ashish0804)
|
||||||
|
* [TikTok] Fix extraction for sigi-based webpages, add API fallback by [MinePlayersPE](https://github.com/MinePlayersPE)
|
||||||
|
* [TikTok] Pass cookies to formats, and misc fixes by [MinePlayersPE](https://github.com/MinePlayersPE)
|
||||||
|
* [TikTok] Extract captions, user thumbnail by [MinePlayersPE](https://github.com/MinePlayersPE)
|
||||||
|
* [TikTok] Change app version by [MinePlayersPE](https://github.com/MinePlayersPE), [llacb47](https://github.com/llacb47)
|
||||||
|
* [TVer] Extract message for unaired live by [Lesmiscore](https://github.com/Lesmiscore)
|
||||||
|
* [twitcasting] Refactor extractor by [Lesmiscore](https://github.com/Lesmiscore)
|
||||||
|
* [twitter] Fix video in quoted tweets
|
||||||
|
* [veoh] Improve extractor by [foghawk](https://github.com/foghawk)
|
||||||
|
* [vk] Capture `clip` URLs
|
||||||
|
* [vk] Fix VKUserVideosIE by [Ashish0804](https://github.com/Ashish0804)
|
||||||
|
* [vk] Improve `_VALID_URL` by [k3ns1n](https://github.com/k3ns1n)
|
||||||
|
* [VrtNU] Handle empty title by [pgaig](https://github.com/pgaig)
|
||||||
|
* [XVideos] Check HLS formats by [MinePlayersPE](https://github.com/MinePlayersPE)
|
||||||
|
* [yahoo:gyao] Improved playlist handling by [hyano](https://github.com/hyano)
|
||||||
|
* [youtube:tab] Extract more playlist metadata by [coletdjnz](https://github.com/coletdjnz), [pukkandan](https://github.com/pukkandan)
|
||||||
|
* [youtube:tab] Raise error on tab redirect by [krichbanana](https://github.com/krichbanana), [coletdjnz](https://github.com/coletdjnz)
|
||||||
|
* [youtube] Update Innertube clients by [coletdjnz](https://github.com/coletdjnz)
|
||||||
|
* [youtube] Detect live-stream embeds
|
||||||
|
* [youtube] Do not return `upload_date` for playlists
|
||||||
|
* [youtube] Extract channel subscriber count by [coletdjnz](https://github.com/coletdjnz)
|
||||||
|
* [youtube] Make invalid storyboard URL non-fatal
|
||||||
|
* [youtube] Enforce UTC, update innertube clients and tests by [coletdjnz](https://github.com/coletdjnz)
|
||||||
|
* [zdf] Add chapter extraction by [iw0nderhow](https://github.com/iw0nderhow)
|
||||||
|
* [zee5] Add geo-bypass
|
||||||
|
|
||||||
|
|
||||||
|
### 2021.12.27
|
||||||
|
|
||||||
|
* Avoid recursion error when re-extracting info
|
||||||
|
* [ffmpeg] Fix position of `--ppa`
|
||||||
|
* [aria2c] Don't show progress when `--no-progress`
|
||||||
|
* [cookies] Support other keyrings by [mbway](https://github.com/mbway)
|
||||||
|
* [EmbedThumbnail] Prefer AtomicParsley over ffmpeg if available
|
||||||
|
* [generic] Fix HTTP KVS Player by [git-anony-mouse](https://github.com/git-anony-mouse)
|
||||||
|
* [ThumbnailsConvertor] Fix for when there are no thumbnails
|
||||||
|
* [docs] Add examples for using `TYPES:` in `-P`/`-o`
|
||||||
|
* [PixivSketch] Add extractors by [nao20010128nao](https://github.com/nao20010128nao)
|
||||||
|
* [tiktok] Add music, sticker and tag IEs by [MinePlayersPE](https://github.com/MinePlayersPE)
|
||||||
|
* [BiliIntl] Fix extractor by [MinePlayersPE](https://github.com/MinePlayersPE)
|
||||||
|
* [CBC] Fix URL regex
|
||||||
|
* [tiktok] Fix `extractor_key` used in archive
|
||||||
|
* [youtube] **End `live-from-start` properly when stream ends with 403**
|
||||||
|
* [Zee5] Fix VALID_URL for tv-shows by [Ashish0804](https://github.com/Ashish0804)
|
||||||
|
|
||||||
### 2021.12.25
|
### 2021.12.25
|
||||||
|
|
||||||
* [dash,youtube] **Download live from start to end** by [nao20010128nao](https://github.com/nao20010128nao), [pukkandan](https://github.com/pukkandan)
|
* [dash,youtube] **Download live from start to end** by [nao20010128nao](https://github.com/nao20010128nao), [pukkandan](https://github.com/pukkandan)
|
||||||
@@ -104,7 +244,7 @@
|
|||||||
* [youtube:comments] Add more options for limiting number of comments extracted by [coletdjnz](https://github.com/coletdjnz)
|
* [youtube:comments] Add more options for limiting number of comments extracted by [coletdjnz](https://github.com/coletdjnz)
|
||||||
* [youtube:tab] Extract more metadata from feeds/channels/playlists by [coletdjnz](https://github.com/coletdjnz)
|
* [youtube:tab] Extract more metadata from feeds/channels/playlists by [coletdjnz](https://github.com/coletdjnz)
|
||||||
* [youtube:tab] Extract video thumbnails from playlist by [coletdjnz](https://github.com/coletdjnz), [pukkandan](https://github.com/pukkandan)
|
* [youtube:tab] Extract video thumbnails from playlist by [coletdjnz](https://github.com/coletdjnz), [pukkandan](https://github.com/pukkandan)
|
||||||
* [youtube:tab] Ignore query when redirecting channel to playlist and cleanup of related code Closes #2046
|
* [youtube:tab] Ignore query when redirecting channel to playlist and cleanup of related code
|
||||||
* [youtube] Fix `ytsearchdate`
|
* [youtube] Fix `ytsearchdate`
|
||||||
* [zdf] Support videos with different ptmd location by [iw0nderhow](https://github.com/iw0nderhow)
|
* [zdf] Support videos with different ptmd location by [iw0nderhow](https://github.com/iw0nderhow)
|
||||||
* [zee5] Support /episodes in URL
|
* [zee5] Support /episodes in URL
|
||||||
|
|||||||
+12
-2
@@ -36,5 +36,15 @@ You can also find lists of all [contributors of yt-dlp](CONTRIBUTORS) and [autho
|
|||||||
|
|
||||||
[](https://ko-fi.com/ashish0804)
|
[](https://ko-fi.com/ashish0804)
|
||||||
|
|
||||||
* Added support for new websites Zee5, MXPlayer, DiscoveryPlusIndia, ShemarooMe, Utreon etc
|
* Added support for new websites BiliIntl, DiscoveryPlusIndia, OlympicsReplay, PlanetMarathi, ShemarooMe, Utreon, Zee5 etc
|
||||||
* Added playlist/series downloads for TubiTv, SonyLIV, Voot, HotStar etc
|
* Added playlist/series downloads for Hotstar, ParamountPlus, Rumble, SonyLIV, Trovo, TubiTv, Voot etc
|
||||||
|
* Improved/fixed support for HiDive, HotStar, Hungama, LBRY, LinkedInLearning, Mxplayer, SonyLiv, TV2, Vimeo, VLive etc
|
||||||
|
|
||||||
|
|
||||||
|
## [Lesmiscore](https://github.com/Lesmiscore) (nao20010128nao)
|
||||||
|
|
||||||
|
**Bitcoin**: bc1qfd02r007cutfdjwjmyy9w23rjvtls6ncve7r3s
|
||||||
|
**Monacoin**: mona1q3tf7dzvshrhfe3md379xtvt2n22duhglv5dskr
|
||||||
|
|
||||||
|
* Download live from start to end for YouTube
|
||||||
|
* Added support for new websites mildom, PixivSketch, skeb, radiko, voicy, mirrativ, openrec, whowatch, damtomo, 17.live, mixch etc
|
||||||
|
|||||||
@@ -1,5 +1,6 @@
|
|||||||
all: lazy-extractors yt-dlp doc pypi-files
|
all: lazy-extractors yt-dlp doc pypi-files
|
||||||
clean: clean-test clean-dist clean-cache
|
clean: clean-test clean-dist
|
||||||
|
clean-all: clean clean-cache
|
||||||
completions: completion-bash completion-fish completion-zsh
|
completions: completion-bash completion-fish completion-zsh
|
||||||
doc: README.md CONTRIBUTING.md issuetemplates supportedsites
|
doc: README.md CONTRIBUTING.md issuetemplates supportedsites
|
||||||
ot: offlinetest
|
ot: offlinetest
|
||||||
@@ -14,14 +15,14 @@ pypi-files: AUTHORS Changelog.md LICENSE README.md README.txt supportedsites com
|
|||||||
|
|
||||||
clean-test:
|
clean-test:
|
||||||
rm -rf test/testdata/player-*.js tmp/ *.annotations.xml *.aria2 *.description *.dump *.frag \
|
rm -rf test/testdata/player-*.js tmp/ *.annotations.xml *.aria2 *.description *.dump *.frag \
|
||||||
*.frag.aria2 *.frag.urls *.info.json *.live_chat.json *.part* *.unknown_video *.ytdl \
|
*.frag.aria2 *.frag.urls *.info.json *.live_chat.json *.meta *.part* *.tmp *.temp *.unknown_video *.ytdl \
|
||||||
*.3gp *.ape *.avi *.desktop *.flac *.flv *.jpeg *.jpg *.m4a *.m4v *.mhtml *.mkv *.mov *.mp3 \
|
*.3gp *.ape *.avi *.desktop *.flac *.flv *.jpeg *.jpg *.m4a *.m4v *.mhtml *.mkv *.mov *.mp3 \
|
||||||
*.mp4 *.ogg *.opus *.png *.sbv *.srt *.swf *.swp *.ttml *.url *.vtt *.wav *.webloc *.webm *.webp
|
*.mp4 *.ogg *.opus *.png *.sbv *.srt *.swf *.swp *.ttml *.url *.vtt *.wav *.webloc *.webm *.webp
|
||||||
clean-dist:
|
clean-dist:
|
||||||
rm -rf yt-dlp.1.temp.md yt-dlp.1 README.txt MANIFEST build/ dist/ .coverage cover/ yt-dlp.tar.gz completions/ \
|
rm -rf yt-dlp.1.temp.md yt-dlp.1 README.txt MANIFEST build/ dist/ .coverage cover/ yt-dlp.tar.gz completions/ \
|
||||||
yt_dlp/extractor/lazy_extractors.py *.spec CONTRIBUTING.md.tmp yt-dlp yt-dlp.exe yt_dlp.egg-info/ AUTHORS .mailmap
|
yt_dlp/extractor/lazy_extractors.py *.spec CONTRIBUTING.md.tmp yt-dlp yt-dlp.exe yt_dlp.egg-info/ AUTHORS .mailmap
|
||||||
clean-cache:
|
clean-cache:
|
||||||
find . -name "*.pyc" -o -name "*.class" -delete
|
find . \( -name "*.pyc" -o -name "*.class" \) -delete
|
||||||
|
|
||||||
completion-bash: completions/bash/yt-dlp
|
completion-bash: completions/bash/yt-dlp
|
||||||
completion-fish: completions/fish/yt-dlp.fish
|
completion-fish: completions/fish/yt-dlp.fish
|
||||||
|
|||||||
@@ -3,17 +3,17 @@
|
|||||||
|
|
||||||
[](#readme)
|
[](#readme)
|
||||||
|
|
||||||
[](https://github.com/yt-dlp/yt-dlp/releases/latest)
|
[](#release-files "Release")
|
||||||
[](https://github.com/yt-dlp/yt-dlp/actions)
|
[](LICENSE "License")
|
||||||
[](LICENSE)
|
[](Collaborators.md#collaborators "Donate")
|
||||||
[](Collaborators.md#collaborators)
|
[](https://readthedocs.org/projects/yt-dlp/ "Docs")
|
||||||
[](supportedsites.md)
|
[](supportedsites.md "Supported Sites")
|
||||||
[](https://discord.gg/H5MNcFW63r)
|
[](https://pypi.org/project/yt-dlp "PyPi")
|
||||||
[](https://yt-dlp.readthedocs.io)
|
[](https://github.com/yt-dlp/yt-dlp/actions "CI Status")
|
||||||
[](https://github.com/yt-dlp/yt-dlp/commits)
|
[](https://discord.gg/H5MNcFW63r "Discord")
|
||||||
[](https://github.com/yt-dlp/yt-dlp/commits)
|
[](https://matrix.to/#/#yt-dlp:matrix.org "Matrix")
|
||||||
[](https://github.com/yt-dlp/yt-dlp/releases/latest)
|
[](https://github.com/yt-dlp/yt-dlp/commits "Commit History")
|
||||||
[](https://pypi.org/project/yt-dlp)
|
[](https://github.com/yt-dlp/yt-dlp/commits "Commit History")
|
||||||
|
|
||||||
</div>
|
</div>
|
||||||
<!-- MANPAGE: END EXCLUDED SECTION -->
|
<!-- MANPAGE: END EXCLUDED SECTION -->
|
||||||
@@ -88,9 +88,9 @@ yt-dlp is a [youtube-dl](https://github.com/ytdl-org/youtube-dl) fork based on t
|
|||||||
* Redirect channel's home URL automatically to `/video` to preserve the old behaviour
|
* Redirect channel's home URL automatically to `/video` to preserve the old behaviour
|
||||||
* `255kbps` audio is extracted (if available) from youtube music when premium cookies are given
|
* `255kbps` audio is extracted (if available) from youtube music when premium cookies are given
|
||||||
* Youtube music Albums, channels etc can be downloaded ([except self-uploaded music](https://github.com/yt-dlp/yt-dlp/issues/723))
|
* Youtube music Albums, channels etc can be downloaded ([except self-uploaded music](https://github.com/yt-dlp/yt-dlp/issues/723))
|
||||||
* Download livestreams from the start using `--live-from-start`
|
* Download livestreams from the start using `--live-from-start` (experimental)
|
||||||
|
|
||||||
* **Cookies from browser**: Cookies can be automatically extracted from all major web browsers using `--cookies-from-browser BROWSER[:PROFILE]`
|
* **Cookies from browser**: Cookies can be automatically extracted from all major web browsers using `--cookies-from-browser BROWSER[+KEYRING][:PROFILE]`
|
||||||
|
|
||||||
* **Split video by chapters**: Videos can be split into multiple files based on chapters using `--split-chapters`
|
* **Split video by chapters**: Videos can be split into multiple files based on chapters using `--split-chapters`
|
||||||
|
|
||||||
@@ -110,9 +110,9 @@ yt-dlp is a [youtube-dl](https://github.com/ytdl-org/youtube-dl) fork based on t
|
|||||||
|
|
||||||
* **Output template improvements**: Output templates can now have date-time formatting, numeric offsets, object traversal etc. See [output template](#output-template) for details. Even more advanced operations can also be done with the help of `--parse-metadata` and `--replace-in-metadata`
|
* **Output template improvements**: Output templates can now have date-time formatting, numeric offsets, object traversal etc. See [output template](#output-template) for details. Even more advanced operations can also be done with the help of `--parse-metadata` and `--replace-in-metadata`
|
||||||
|
|
||||||
* **Other new options**: Many new options have been added such as `--print`, `--wait-for-video`, `--sleep-requests`, `--convert-thumbnails`, `--write-link`, `--force-download-archive`, `--force-overwrites`, `--break-on-reject` etc
|
* **Other new options**: Many new options have been added such as `--concat-playlist`, `--print`, `--wait-for-video`, `--sleep-requests`, `--convert-thumbnails`, `--write-link`, `--force-download-archive`, `--force-overwrites`, `--break-on-reject` etc
|
||||||
|
|
||||||
* **Improvements**: Regex and other operators in `--match-filter`, multiple `--postprocessor-args` and `--downloader-args`, faster archive checking, more [format selection options](#format-selection), merge multi-video/audio etc
|
* **Improvements**: Regex and other operators in `--match-filter`, multiple `--postprocessor-args` and `--downloader-args`, faster archive checking, more [format selection options](#format-selection), merge multi-video/audio, multiple `--config-locations`, `--exec` at different stages, etc
|
||||||
|
|
||||||
* **Plugins**: Extractors and PostProcessors can be loaded from an external file. See [plugins](#plugins) for details
|
* **Plugins**: Extractors and PostProcessors can be loaded from an external file. See [plugins](#plugins) for details
|
||||||
|
|
||||||
@@ -133,12 +133,12 @@ Some of yt-dlp's default options are different from that of youtube-dl and youtu
|
|||||||
* `--ignore-errors` is enabled by default. Use `--abort-on-error` or `--compat-options abort-on-error` to abort on errors instead
|
* `--ignore-errors` is enabled by default. Use `--abort-on-error` or `--compat-options abort-on-error` to abort on errors instead
|
||||||
* When writing metadata files such as thumbnails, description or infojson, the same information (if available) is also written for playlists. Use `--no-write-playlist-metafiles` or `--compat-options no-playlist-metafiles` to not write these files
|
* When writing metadata files such as thumbnails, description or infojson, the same information (if available) is also written for playlists. Use `--no-write-playlist-metafiles` or `--compat-options no-playlist-metafiles` to not write these files
|
||||||
* `--add-metadata` attaches the `infojson` to `mkv` files in addition to writing the metadata when used with `--write-info-json`. Use `--no-embed-info-json` or `--compat-options no-attach-info-json` to revert this
|
* `--add-metadata` attaches the `infojson` to `mkv` files in addition to writing the metadata when used with `--write-info-json`. Use `--no-embed-info-json` or `--compat-options no-attach-info-json` to revert this
|
||||||
* Some metadata are embedded into different fields when using `--add-metadata` as compared to youtube-dl. Most notably, `comment` field contains the `webpage_url` and `synopsis` contains the `description`. You can [use `--parse-metadata`](https://github.com/yt-dlp/yt-dlp#modifying-metadata) to modify this to your liking or use `--compat-options embed-metadata` to revert this
|
* Some metadata are embedded into different fields when using `--add-metadata` as compared to youtube-dl. Most notably, `comment` field contains the `webpage_url` and `synopsis` contains the `description`. You can [use `--parse-metadata`](#modifying-metadata) to modify this to your liking or use `--compat-options embed-metadata` to revert this
|
||||||
* `playlist_index` behaves differently when used with options like `--playlist-reverse` and `--playlist-items`. See [#302](https://github.com/yt-dlp/yt-dlp/issues/302) for details. You can use `--compat-options playlist-index` if you want to keep the earlier behavior
|
* `playlist_index` behaves differently when used with options like `--playlist-reverse` and `--playlist-items`. See [#302](https://github.com/yt-dlp/yt-dlp/issues/302) for details. You can use `--compat-options playlist-index` if you want to keep the earlier behavior
|
||||||
* The output of `-F` is listed in a new format. Use `--compat-options list-formats` to revert this
|
* The output of `-F` is listed in a new format. Use `--compat-options list-formats` to revert this
|
||||||
* All *experiences* of a funimation episode are considered as a single video. This behavior breaks existing archives. Use `--compat-options seperate-video-versions` to extract information from only the default player
|
* All *experiences* of a funimation episode are considered as a single video. This behavior breaks existing archives. Use `--compat-options seperate-video-versions` to extract information from only the default player
|
||||||
* Youtube live chat (if available) is considered as a subtitle. Use `--sub-langs all,-live_chat` to download all subtitles except live chat. You can also use `--compat-options no-live-chat` to prevent live chat from downloading
|
* Youtube live chat (if available) is considered as a subtitle. Use `--sub-langs all,-live_chat` to download all subtitles except live chat. You can also use `--compat-options no-live-chat` to prevent live chat from downloading
|
||||||
* Youtube channel URLs are automatically redirected to `/video`. Append a `/featured` to the URL to download only the videos in the home page. If the channel does not have a videos tab, we try to download the equivalent `UU` playlist instead. Also, `/live` URLs raise an error if there are no live videos instead of silently downloading the entire channel. You may use `--compat-options no-youtube-channel-redirect` to revert all these redirections
|
* Youtube channel URLs are automatically redirected to `/video`. Append a `/featured` to the URL to download only the videos in the home page. If the channel does not have a videos tab, we try to download the equivalent `UU` playlist instead. For all other tabs, if the channel does not show the requested tab, an error will be raised. Also, `/live` URLs raise an error if there are no live videos instead of silently downloading the entire channel. You may use `--compat-options no-youtube-channel-redirect` to revert all these redirections
|
||||||
* Unavailable videos are also listed for youtube playlists. Use `--compat-options no-youtube-unavailable-videos` to remove this
|
* Unavailable videos are also listed for youtube playlists. Use `--compat-options no-youtube-unavailable-videos` to remove this
|
||||||
* If `ffmpeg` is used as the downloader, the downloading and merging of formats happen in a single step when possible. Use `--compat-options no-direct-merge` to revert this
|
* If `ffmpeg` is used as the downloader, the downloading and merging of formats happen in a single step when possible. Use `--compat-options no-direct-merge` to revert this
|
||||||
* Thumbnail embedding in `mp4` is done with mutagen if possible. Use `--compat-options embed-thumbnail-atomicparsley` to force the use of AtomicParsley instead
|
* Thumbnail embedding in `mp4` is done with mutagen if possible. Use `--compat-options embed-thumbnail-atomicparsley` to force the use of AtomicParsley instead
|
||||||
@@ -157,8 +157,19 @@ You can install yt-dlp using one of the following methods:
|
|||||||
|
|
||||||
### Using the release binary
|
### Using the release binary
|
||||||
|
|
||||||
You can simply download the [correct binary file](#release-files) for your OS: **[[Windows](https://github.com/yt-dlp/yt-dlp/releases/latest/download/yt-dlp.exe)] [[UNIX-like](https://github.com/yt-dlp/yt-dlp/releases/latest/download/yt-dlp)]**
|
You can simply download the [correct binary file](#release-files) for your OS
|
||||||
|
|
||||||
|
<!-- MANPAGE: BEGIN EXCLUDED SECTION -->
|
||||||
|
[](https://github.com/yt-dlp/yt-dlp/releases/latest/download/yt-dlp.exe)
|
||||||
|
[](https://github.com/yt-dlp/yt-dlp/releases/latest/download/yt-dlp)
|
||||||
|
[](https://github.com/yt-dlp/yt-dlp/releases/latest/download/yt-dlp.tar.gz)
|
||||||
|
[](#release-files)
|
||||||
|
[](https://github.com/yt-dlp/yt-dlp/releases)
|
||||||
|
<!-- MANPAGE: END EXCLUDED SECTION -->
|
||||||
|
|
||||||
|
Note: The manpages, shell completion files etc. are available in the [source tarball](https://github.com/yt-dlp/yt-dlp/releases/latest/download/yt-dlp.tar.gz)
|
||||||
|
|
||||||
|
<!-- TODO: Move to Wiki -->
|
||||||
In UNIX-like OSes (MacOS, Linux, BSD), you can also install the same in one of the following ways:
|
In UNIX-like OSes (MacOS, Linux, BSD), you can also install the same in one of the following ways:
|
||||||
|
|
||||||
```
|
```
|
||||||
@@ -176,7 +187,6 @@ sudo aria2c https://github.com/yt-dlp/yt-dlp/releases/latest/download/yt-dlp --d
|
|||||||
sudo chmod a+rx /usr/local/bin/yt-dlp
|
sudo chmod a+rx /usr/local/bin/yt-dlp
|
||||||
```
|
```
|
||||||
|
|
||||||
PS: The manpages, shell completion files etc. are available in [yt-dlp.tar.gz](https://github.com/yt-dlp/yt-dlp/releases/latest/download/yt-dlp.tar.gz)
|
|
||||||
|
|
||||||
### With [PIP](https://pypi.org/project/pip)
|
### With [PIP](https://pypi.org/project/pip)
|
||||||
|
|
||||||
@@ -197,6 +207,7 @@ python3 -m pip install --force-reinstall https://github.com/yt-dlp/yt-dlp/archiv
|
|||||||
|
|
||||||
Note that on some systems, you may need to use `py` or `python` instead of `python3`
|
Note that on some systems, you may need to use `py` or `python` instead of `python3`
|
||||||
|
|
||||||
|
<!-- TODO: Add to Wiki, Remove Taps -->
|
||||||
### With [Homebrew](https://brew.sh)
|
### With [Homebrew](https://brew.sh)
|
||||||
|
|
||||||
macOS or Linux users that are using Homebrew can also install it by:
|
macOS or Linux users that are using Homebrew can also install it by:
|
||||||
@@ -255,7 +266,7 @@ While all the other dependencies are optional, `ffmpeg` and `ffprobe` are highly
|
|||||||
* [**mutagen**](https://github.com/quodlibet/mutagen) - For embedding thumbnail in certain formats. Licensed under [GPLv2+](https://github.com/quodlibet/mutagen/blob/master/COPYING)
|
* [**mutagen**](https://github.com/quodlibet/mutagen) - For embedding thumbnail in certain formats. Licensed under [GPLv2+](https://github.com/quodlibet/mutagen/blob/master/COPYING)
|
||||||
* [**pycryptodomex**](https://github.com/Legrandin/pycryptodome) - For decrypting AES-128 HLS streams and various other data. Licensed under [BSD2](https://github.com/Legrandin/pycryptodome/blob/master/LICENSE.rst)
|
* [**pycryptodomex**](https://github.com/Legrandin/pycryptodome) - For decrypting AES-128 HLS streams and various other data. Licensed under [BSD2](https://github.com/Legrandin/pycryptodome/blob/master/LICENSE.rst)
|
||||||
* [**websockets**](https://github.com/aaugustin/websockets) - For downloading over websocket. Licensed under [BSD3](https://github.com/aaugustin/websockets/blob/main/LICENSE)
|
* [**websockets**](https://github.com/aaugustin/websockets) - For downloading over websocket. Licensed under [BSD3](https://github.com/aaugustin/websockets/blob/main/LICENSE)
|
||||||
* [**keyring**](https://github.com/jaraco/keyring) - For decrypting cookies of chromium-based browsers on Linux. Licensed under [MIT](https://github.com/jaraco/keyring/blob/main/LICENSE)
|
* [**secretstorage**](https://github.com/mitya57/secretstorage) - For accessing the Gnome keyring while decrypting cookies of Chromium-based browsers on Linux. Licensed under [BSD](https://github.com/mitya57/secretstorage/blob/master/LICENSE)
|
||||||
* [**AtomicParsley**](https://github.com/wez/atomicparsley) - For embedding thumbnail in mp4/m4a if mutagen is not present. Licensed under [GPLv2+](https://github.com/wez/atomicparsley/blob/master/COPYING)
|
* [**AtomicParsley**](https://github.com/wez/atomicparsley) - For embedding thumbnail in mp4/m4a if mutagen is not present. Licensed under [GPLv2+](https://github.com/wez/atomicparsley/blob/master/COPYING)
|
||||||
* [**rtmpdump**](http://rtmpdump.mplayerhq.hu) - For downloading `rtmp` streams. ffmpeg will be used as a fallback. Licensed under [GPLv2+](http://rtmpdump.mplayerhq.hu)
|
* [**rtmpdump**](http://rtmpdump.mplayerhq.hu) - For downloading `rtmp` streams. ffmpeg will be used as a fallback. Licensed under [GPLv2+](http://rtmpdump.mplayerhq.hu)
|
||||||
* [**mplayer**](http://mplayerhq.hu/design7/info.html) or [**mpv**](https://mpv.io) - For downloading `rstp` streams. ffmpeg will be used as a fallback. Licensed under [GPLv2+](https://github.com/mpv-player/mpv/blob/master/Copyright)
|
* [**mplayer**](http://mplayerhq.hu/design7/info.html) or [**mpv**](https://mpv.io) - For downloading `rstp` streams. ffmpeg will be used as a fallback. Licensed under [GPLv2+](https://github.com/mpv-player/mpv/blob/master/Copyright)
|
||||||
@@ -267,7 +278,7 @@ To use or redistribute the dependencies, you must agree to their respective lice
|
|||||||
|
|
||||||
The Windows and MacOS standalone release binaries are already built with the python interpreter, mutagen, pycryptodomex and websockets included.
|
The Windows and MacOS standalone release binaries are already built with the python interpreter, mutagen, pycryptodomex and websockets included.
|
||||||
|
|
||||||
**Note**: There are some regressions in newer ffmpeg versions that causes various issues when used alongside yt-dlp. Since ffmpeg is such an important dependency, we provide [custom builds](https://github.com/yt-dlp/FFmpeg-Builds/wiki/Latest#latest-autobuilds) with patches for these issues at [yt-dlp/FFmpeg-Builds](https://github.com/yt-dlp/FFmpeg-Builds). See [the readme](https://github.com/yt-dlp/FFmpeg-Builds#patches-applied) for details on the specific issues solved by these builds
|
**Note**: There are some regressions in newer ffmpeg versions that causes various issues when used alongside yt-dlp. Since ffmpeg is such an important dependency, we provide [custom builds](https://github.com/yt-dlp/FFmpeg-Builds#ffmpeg-static-auto-builds) with patches for these issues at [yt-dlp/FFmpeg-Builds](https://github.com/yt-dlp/FFmpeg-Builds). See [the readme](https://github.com/yt-dlp/FFmpeg-Builds#patches-applied) for details on the specific issues solved by these builds
|
||||||
|
|
||||||
|
|
||||||
## COMPILE
|
## COMPILE
|
||||||
@@ -327,22 +338,27 @@ You can also fork the project on github and run your fork's [build workflow](.gi
|
|||||||
an error. The default value "fixup_error"
|
an error. The default value "fixup_error"
|
||||||
repairs broken URLs, but emits an error if
|
repairs broken URLs, but emits an error if
|
||||||
this is not possible instead of searching
|
this is not possible instead of searching
|
||||||
--ignore-config, --no-config Disable loading any configuration files
|
--ignore-config Don't load any more configuration files
|
||||||
except the one provided by --config-location.
|
except those given by --config-locations.
|
||||||
When given inside a configuration
|
For backward compatibility, if this option
|
||||||
file, no further configuration files are
|
is found inside the system configuration
|
||||||
loaded. Additionally, (for backward
|
file, the user configuration is not loaded.
|
||||||
compatibility) if this option is found
|
(Alias: --no-config)
|
||||||
inside the system configuration file, the
|
--no-config-locations Do not load any custom configuration files
|
||||||
user configuration is not loaded
|
(default). When given inside a
|
||||||
--config-location PATH Location of the main configuration file;
|
configuration file, ignore all previous
|
||||||
|
--config-locations defined in the current
|
||||||
|
file
|
||||||
|
--config-locations PATH Location of the main configuration file;
|
||||||
either the path to the config or its
|
either the path to the config or its
|
||||||
containing directory
|
containing directory. Can be used multiple
|
||||||
|
times and inside other configuration files
|
||||||
--flat-playlist Do not extract the videos of a playlist,
|
--flat-playlist Do not extract the videos of a playlist,
|
||||||
only list them
|
only list them
|
||||||
--no-flat-playlist Extract the videos of a playlist
|
--no-flat-playlist Extract the videos of a playlist
|
||||||
--live-from-start Download livestreams from the start.
|
--live-from-start Download livestreams from the start.
|
||||||
Currently only supported for YouTube
|
Currently only supported for YouTube
|
||||||
|
(Experimental)
|
||||||
--no-live-from-start Download livestreams from the current time
|
--no-live-from-start Download livestreams from the current time
|
||||||
(default)
|
(default)
|
||||||
--wait-for-video MIN[-MAX] Wait for scheduled streams to become
|
--wait-for-video MIN[-MAX] Wait for scheduled streams to become
|
||||||
@@ -514,8 +530,8 @@ You can also fork the project on github and run your fork's [build workflow](.gi
|
|||||||
example, --downloader aria2c --downloader
|
example, --downloader aria2c --downloader
|
||||||
"dash,m3u8:native" will use aria2c for
|
"dash,m3u8:native" will use aria2c for
|
||||||
http/ftp downloads, and the native
|
http/ftp downloads, and the native
|
||||||
downloader for dash/m3u8 downloads
|
downloader for dash/m3u8 downloads (Alias:
|
||||||
(Alias: --external-downloader)
|
--external-downloader)
|
||||||
--downloader-args NAME:ARGS Give these arguments to the external
|
--downloader-args NAME:ARGS Give these arguments to the external
|
||||||
downloader. Specify the downloader name and
|
downloader. Specify the downloader name and
|
||||||
the arguments separated by a colon ":". For
|
the arguments separated by a colon ":". For
|
||||||
@@ -523,8 +539,8 @@ You can also fork the project on github and run your fork's [build workflow](.gi
|
|||||||
different positions using the same syntax
|
different positions using the same syntax
|
||||||
as --postprocessor-args. You can use this
|
as --postprocessor-args. You can use this
|
||||||
option multiple times to give different
|
option multiple times to give different
|
||||||
arguments to different downloaders
|
arguments to different downloaders (Alias:
|
||||||
(Alias: --external-downloader-args)
|
--external-downloader-args)
|
||||||
|
|
||||||
## Filesystem Options:
|
## Filesystem Options:
|
||||||
-a, --batch-file FILE File containing URLs to download ("-" for
|
-a, --batch-file FILE File containing URLs to download ("-" for
|
||||||
@@ -535,7 +551,7 @@ You can also fork the project on github and run your fork's [build workflow](.gi
|
|||||||
-P, --paths [TYPES:]PATH The paths where the files should be
|
-P, --paths [TYPES:]PATH The paths where the files should be
|
||||||
downloaded. Specify the type of file and
|
downloaded. Specify the type of file and
|
||||||
the path separated by a colon ":". All the
|
the path separated by a colon ":". All the
|
||||||
same types as --output are supported.
|
same TYPES as --output are supported.
|
||||||
Additionally, you can also provide "home"
|
Additionally, you can also provide "home"
|
||||||
(default) and "temp" paths. All
|
(default) and "temp" paths. All
|
||||||
intermediary files are first downloaded to
|
intermediary files are first downloaded to
|
||||||
@@ -598,8 +614,8 @@ You can also fork the project on github and run your fork's [build workflow](.gi
|
|||||||
without this option if the extraction is
|
without this option if the extraction is
|
||||||
known to be quick (Alias: --get-comments)
|
known to be quick (Alias: --get-comments)
|
||||||
--no-write-comments Do not retrieve video comments unless the
|
--no-write-comments Do not retrieve video comments unless the
|
||||||
extraction is known to be quick
|
extraction is known to be quick (Alias:
|
||||||
(Alias: --no-get-comments)
|
--no-get-comments)
|
||||||
--load-info-json FILE JSON file containing the video information
|
--load-info-json FILE JSON file containing the video information
|
||||||
(created with the "--write-info-json"
|
(created with the "--write-info-json"
|
||||||
option)
|
option)
|
||||||
@@ -607,16 +623,19 @@ You can also fork the project on github and run your fork's [build workflow](.gi
|
|||||||
from and dump cookie jar in
|
from and dump cookie jar in
|
||||||
--no-cookies Do not read/dump cookies from/to file
|
--no-cookies Do not read/dump cookies from/to file
|
||||||
(default)
|
(default)
|
||||||
--cookies-from-browser BROWSER[:PROFILE]
|
--cookies-from-browser BROWSER[+KEYRING][:PROFILE]
|
||||||
Load cookies from a user profile of the
|
The name of the browser and (optionally)
|
||||||
given web browser. Currently supported
|
the name/path of the profile to load
|
||||||
browsers are: brave, chrome, chromium,
|
cookies from, separated by a ":". Currently
|
||||||
edge, firefox, opera, safari, vivaldi. You
|
supported browsers are: brave, chrome,
|
||||||
can specify the user profile name or
|
chromium, edge, firefox, opera, safari,
|
||||||
directory using "BROWSER:PROFILE_NAME" or
|
vivaldi. By default, the most recently
|
||||||
"BROWSER:PROFILE_PATH". If no profile is
|
accessed profile is used. The keyring used
|
||||||
given, the most recently accessed one is
|
for decrypting Chromium cookies on Linux
|
||||||
used
|
can be (optionally) specified after the
|
||||||
|
browser name separated by a "+". Currently
|
||||||
|
supported keyrings are: basictext,
|
||||||
|
gnomekeyring, kwallet
|
||||||
--no-cookies-from-browser Do not load cookies from browser (default)
|
--no-cookies-from-browser Do not load cookies from browser (default)
|
||||||
--cache-dir DIR Location in the filesystem where youtube-dl
|
--cache-dir DIR Location in the filesystem where youtube-dl
|
||||||
can store some downloaded information (such
|
can store some downloaded information (such
|
||||||
@@ -659,10 +678,14 @@ You can also fork the project on github and run your fork's [build workflow](.gi
|
|||||||
formats are found (default)
|
formats are found (default)
|
||||||
--skip-download Do not download the video but write all
|
--skip-download Do not download the video but write all
|
||||||
related files (Alias: --no-download)
|
related files (Alias: --no-download)
|
||||||
-O, --print TEMPLATE Quiet, but print the given fields for each
|
-O, --print [WHEN:]TEMPLATE Field name or output template to print to
|
||||||
video. Simulate unless --no-simulate is
|
screen, optionally prefixed with when to
|
||||||
used. Either a field name or same syntax as
|
print it, separated by a ":". Supported
|
||||||
the output template can be used
|
values of "WHEN" are the same as that of
|
||||||
|
--use-postprocessor, and "video" (default).
|
||||||
|
Implies --quiet and --simulate (unless
|
||||||
|
--no-simulate is used). This option can be
|
||||||
|
used multiple times
|
||||||
-j, --dump-json Quiet, but print JSON information for each
|
-j, --dump-json Quiet, but print JSON information for each
|
||||||
video. Simulate unless --no-simulate is
|
video. Simulate unless --no-simulate is
|
||||||
used. See "OUTPUT TEMPLATE" for a
|
used. See "OUTPUT TEMPLATE" for a
|
||||||
@@ -700,6 +723,9 @@ You can also fork the project on github and run your fork's [build workflow](.gi
|
|||||||
|
|
||||||
## Workarounds:
|
## Workarounds:
|
||||||
--encoding ENCODING Force the specified encoding (experimental)
|
--encoding ENCODING Force the specified encoding (experimental)
|
||||||
|
--legacy-server-connect Explicitly allow HTTPS connection to
|
||||||
|
servers that do not support RFC 5746 secure
|
||||||
|
renegotiation
|
||||||
--no-check-certificates Suppress HTTPS certificate validation
|
--no-check-certificates Suppress HTTPS certificate validation
|
||||||
--prefer-insecure Use an unencrypted connection to retrieve
|
--prefer-insecure Use an unencrypted connection to retrieve
|
||||||
information about the video (Currently
|
information about the video (Currently
|
||||||
@@ -778,9 +804,9 @@ You can also fork the project on github and run your fork's [build workflow](.gi
|
|||||||
be regex) or "all" separated by commas.
|
be regex) or "all" separated by commas.
|
||||||
(Eg: --sub-langs "en.*,ja") You can prefix
|
(Eg: --sub-langs "en.*,ja") You can prefix
|
||||||
the language code with a "-" to exempt it
|
the language code with a "-" to exempt it
|
||||||
from the requested languages. (Eg: --sub-
|
from the requested languages. (Eg:
|
||||||
langs all,-live_chat) Use --list-subs for a
|
--sub-langs all,-live_chat) Use --list-subs
|
||||||
list of available language tags
|
for a list of available language tags
|
||||||
|
|
||||||
## Authentication Options:
|
## Authentication Options:
|
||||||
-u, --username USERNAME Login with this account ID
|
-u, --username USERNAME Login with this account ID
|
||||||
@@ -882,6 +908,15 @@ You can also fork the project on github and run your fork's [build workflow](.gi
|
|||||||
multiple times
|
multiple times
|
||||||
--xattrs Write metadata to the video file's xattrs
|
--xattrs Write metadata to the video file's xattrs
|
||||||
(using dublin core and xdg standards)
|
(using dublin core and xdg standards)
|
||||||
|
--concat-playlist POLICY Concatenate videos in a playlist. One of
|
||||||
|
"never", "always", or "multi_video"
|
||||||
|
(default; only when the videos form a
|
||||||
|
single show). All the video files must have
|
||||||
|
same codecs and number of streams to be
|
||||||
|
concatable. The "pl_video:" prefix can be
|
||||||
|
used with "--paths" and "--output" to set
|
||||||
|
the output filename for the split files.
|
||||||
|
See "OUTPUT TEMPLATE" for details
|
||||||
--fixup POLICY Automatically correct known faults of the
|
--fixup POLICY Automatically correct known faults of the
|
||||||
file. One of never (do nothing), warn (only
|
file. One of never (do nothing), warn (only
|
||||||
emit a warning), detect_or_warn (the
|
emit a warning), detect_or_warn (the
|
||||||
@@ -891,23 +926,20 @@ You can also fork the project on github and run your fork's [build workflow](.gi
|
|||||||
--ffmpeg-location PATH Location of the ffmpeg binary; either the
|
--ffmpeg-location PATH Location of the ffmpeg binary; either the
|
||||||
path to the binary or its containing
|
path to the binary or its containing
|
||||||
directory
|
directory
|
||||||
--exec CMD Execute a command on the file after
|
--exec [WHEN:]CMD Execute a command, optionally prefixed with
|
||||||
downloading and post-processing. Same
|
when to execute it (after_move if
|
||||||
syntax as the output template can be used
|
unspecified), separated by a ":". Supported
|
||||||
to pass any field as arguments to the
|
values of "WHEN" are the same as that of
|
||||||
command. An additional field "filepath"
|
--use-postprocessor. Same syntax as the
|
||||||
|
output template can be used to pass any
|
||||||
|
field as arguments to the command. After
|
||||||
|
download, an additional field "filepath"
|
||||||
that contains the final path of the
|
that contains the final path of the
|
||||||
downloaded file is also available. If no
|
downloaded file is also available, and if
|
||||||
fields are passed, %(filepath)q is appended
|
no fields are passed, %(filepath)q is
|
||||||
to the end of the command. This option can
|
appended to the end of the command. This
|
||||||
be used multiple times
|
|
||||||
--no-exec Remove any previously defined --exec
|
|
||||||
--exec-before-download CMD Execute a command before the actual
|
|
||||||
download. The syntax is the same as --exec
|
|
||||||
but "filepath" is not available. This
|
|
||||||
option can be used multiple times
|
option can be used multiple times
|
||||||
--no-exec-before-download Remove any previously defined
|
--no-exec Remove any previously defined --exec
|
||||||
--exec-before-download
|
|
||||||
--convert-subs FORMAT Convert the subtitles to another format
|
--convert-subs FORMAT Convert the subtitles to another format
|
||||||
(currently supported: srt|vtt|ass|lrc)
|
(currently supported: srt|vtt|ass|lrc)
|
||||||
(Alias: --convert-subtitles)
|
(Alias: --convert-subtitles)
|
||||||
@@ -946,10 +978,12 @@ You can also fork the project on github and run your fork's [build workflow](.gi
|
|||||||
"pre_process" (after extraction),
|
"pre_process" (after extraction),
|
||||||
"before_dl" (before video download),
|
"before_dl" (before video download),
|
||||||
"post_process" (after video download;
|
"post_process" (after video download;
|
||||||
default) or "after_move" (after moving file
|
default), "after_move" (after moving file
|
||||||
to their final locations). This option can
|
to their final locations), "after_video"
|
||||||
be used multiple times to add different
|
(after downloading and processing all
|
||||||
postprocessors
|
formats of a video), or "playlist" (end of
|
||||||
|
playlist). This option can be used multiple
|
||||||
|
times to add different postprocessors
|
||||||
|
|
||||||
## SponsorBlock Options:
|
## SponsorBlock Options:
|
||||||
Make chapter entries for, or remove various segments (sponsor,
|
Make chapter entries for, or remove various segments (sponsor,
|
||||||
@@ -1087,7 +1121,7 @@ The field names themselves (the part inside the parenthesis) can also have some
|
|||||||
|
|
||||||
1. **Default**: A literal default value can be specified for when the field is empty using a `|` separator. This overrides `--output-na-template`. Eg: `%(uploader|Unknown)s`
|
1. **Default**: A literal default value can be specified for when the field is empty using a `|` separator. This overrides `--output-na-template`. Eg: `%(uploader|Unknown)s`
|
||||||
|
|
||||||
1. **More Conversions**: In addition to the normal format types `diouxXeEfFgGcrs`, `B`, `j`, `l`, `q`, `D`, `S` can be used for converting to **B**ytes, **j**son (flag `#` for pretty-printing), a comma separated **l**ist (flag `#` for `\n` newline-separated), a string **q**uoted for the terminal (flag `#` to split a list into different arguments), to add **D**ecimal suffixes (Eg: 10M), and to **S**anitize as filename (flag `#` for restricted), respectively
|
1. **More Conversions**: In addition to the normal format types `diouxXeEfFgGcrs`, `B`, `j`, `l`, `q`, `D`, `S` can be used for converting to **B**ytes, **j**son (flag `#` for pretty-printing), a comma separated **l**ist (flag `#` for `\n` newline-separated), a string **q**uoted for the terminal (flag `#` to split a list into different arguments), to add **D**ecimal suffixes (Eg: 10M) (flag `#` to use 1024 as factor), and to **S**anitize as filename (flag `#` for restricted), respectively
|
||||||
|
|
||||||
1. **Unicode normalization**: The format type `U` can be used for NFC [unicode normalization](https://docs.python.org/3/library/unicodedata.html#unicodedata.normalize). The alternate form flag (`#`) changes the normalization to NFD and the conversion flag `+` can be used for NFKC/NFKD compatibility equivalence normalization. Eg: `%(title)+.100U` is NFKC
|
1. **Unicode normalization**: The format type `U` can be used for NFC [unicode normalization](https://docs.python.org/3/library/unicodedata.html#unicodedata.normalize). The alternate form flag (`#`) changes the normalization to NFD and the conversion flag `+` can be used for NFKC/NFKD compatibility equivalence normalization. Eg: `%(title)+.100U` is NFKC
|
||||||
|
|
||||||
@@ -1096,7 +1130,7 @@ To summarize, the general syntax for a field is:
|
|||||||
%(name[.keys][addition][>strf][,alternate][&replacement][|default])[flags][width][.precision][length]type
|
%(name[.keys][addition][>strf][,alternate][&replacement][|default])[flags][width][.precision][length]type
|
||||||
```
|
```
|
||||||
|
|
||||||
Additionally, you can set different output templates for the various metadata files separately from the general output template by specifying the type of file followed by the template separated by a colon `:`. The different file types supported are `subtitle`, `thumbnail`, `description`, `annotation` (deprecated), `infojson`, `link`, `pl_thumbnail`, `pl_description`, `pl_infojson`, `chapter`. For example, `-o "%(title)s.%(ext)s" -o "thumbnail:%(title)s\%(title)s.%(ext)s"` will put the thumbnails in a folder with the same name as the video. If any of the templates (except default) is empty, that type of file will not be written. Eg: `--write-thumbnail -o "thumbnail:"` will write thumbnails only for playlists and not for video.
|
Additionally, you can set different output templates for the various metadata files separately from the general output template by specifying the type of file followed by the template separated by a colon `:`. The different file types supported are `subtitle`, `thumbnail`, `description`, `annotation` (deprecated), `infojson`, `link`, `pl_thumbnail`, `pl_description`, `pl_infojson`, `chapter`, `pl_video`. For example, `-o "%(title)s.%(ext)s" -o "thumbnail:%(title)s\%(title)s.%(ext)s"` will put the thumbnails in a folder with the same name as the video. If any of the templates (except default) is empty, that type of file will not be written. Eg: `--write-thumbnail -o "thumbnail:"` will write thumbnails only for playlists and not for video.
|
||||||
|
|
||||||
The available fields are:
|
The available fields are:
|
||||||
|
|
||||||
@@ -1112,11 +1146,14 @@ The available fields are:
|
|||||||
- `creator` (string): The creator of the video
|
- `creator` (string): The creator of the video
|
||||||
- `timestamp` (numeric): UNIX timestamp of the moment the video became available
|
- `timestamp` (numeric): UNIX timestamp of the moment the video became available
|
||||||
- `upload_date` (string): Video upload date (YYYYMMDD)
|
- `upload_date` (string): Video upload date (YYYYMMDD)
|
||||||
- `release_date` (string): The date (YYYYMMDD) when the video was released
|
|
||||||
- `release_timestamp` (numeric): UNIX timestamp of the moment the video was released
|
- `release_timestamp` (numeric): UNIX timestamp of the moment the video was released
|
||||||
|
- `release_date` (string): The date (YYYYMMDD) when the video was released
|
||||||
|
- `modified_timestamp` (numeric): UNIX timestamp of the moment the video was last modified
|
||||||
|
- `modified_date` (string): The date (YYYYMMDD) when the video was last modified
|
||||||
- `uploader_id` (string): Nickname or id of the video uploader
|
- `uploader_id` (string): Nickname or id of the video uploader
|
||||||
- `channel` (string): Full name of the channel the video is uploaded on
|
- `channel` (string): Full name of the channel the video is uploaded on
|
||||||
- `channel_id` (string): Id of the channel
|
- `channel_id` (string): Id of the channel
|
||||||
|
- `channel_follower_count` (numeric): Number of followers of the channel
|
||||||
- `location` (string): Physical location where the video was filmed
|
- `location` (string): Physical location where the video was filmed
|
||||||
- `duration` (numeric): Length of the video in seconds
|
- `duration` (numeric): Length of the video in seconds
|
||||||
- `duration_string` (string): Length of the video (HH:mm:ss)
|
- `duration_string` (string): Length of the video (HH:mm:ss)
|
||||||
@@ -1156,8 +1193,10 @@ The available fields are:
|
|||||||
- `extractor_key` (string): Key name of the extractor
|
- `extractor_key` (string): Key name of the extractor
|
||||||
- `epoch` (numeric): Unix epoch when creating the file
|
- `epoch` (numeric): Unix epoch when creating the file
|
||||||
- `autonumber` (numeric): Number that will be increased with each download, starting at `--autonumber-start`
|
- `autonumber` (numeric): Number that will be increased with each download, starting at `--autonumber-start`
|
||||||
|
- `video_autonumber` (numeric): Number that will be increased with each video
|
||||||
- `n_entries` (numeric): Total number of extracted items in the playlist
|
- `n_entries` (numeric): Total number of extracted items in the playlist
|
||||||
- `playlist` (string): Name or id of the playlist that contains the video
|
- `playlist` (string): Name or id of the playlist that contains the video
|
||||||
|
- `playlist_count` (numeric): Total number of items in the playlist. May not be known if entire playlist is not extracted
|
||||||
- `playlist_index` (numeric): Index of the video in the playlist padded with leading zeros according the final index
|
- `playlist_index` (numeric): Index of the video in the playlist padded with leading zeros according the final index
|
||||||
- `playlist_autonumber` (numeric): Position of the video in the playlist download queue padded with leading zeros according to the total length of the playlist
|
- `playlist_autonumber` (numeric): Position of the video in the playlist download queue padded with leading zeros according to the total length of the playlist
|
||||||
- `playlist_id` (string): Playlist identifier
|
- `playlist_id` (string): Playlist identifier
|
||||||
@@ -1209,6 +1248,11 @@ Available only when used in `--print`:
|
|||||||
|
|
||||||
- `urls` (string): The URLs of all requested formats, one in each line
|
- `urls` (string): The URLs of all requested formats, one in each line
|
||||||
- `filename` (string): Name of the video file. Note that the actual filename may be different due to post-processing. Use `--exec echo` to get the name after all postprocessing is complete
|
- `filename` (string): Name of the video file. Note that the actual filename may be different due to post-processing. Use `--exec echo` to get the name after all postprocessing is complete
|
||||||
|
- `formats_table` (table): The video format table as printed by `--list-formats`
|
||||||
|
- `thumbnails_table` (table): The thumbnail format table as printed by `--list-thumbnails`
|
||||||
|
- `subtitles_table` (table): The subtitle format table as printed by `--list-subs`
|
||||||
|
- `automatic_captions_table` (table): The automatic subtitle format table as printed by `--list-subs`
|
||||||
|
|
||||||
|
|
||||||
Available only in `--sponsorblock-chapter-title`:
|
Available only in `--sponsorblock-chapter-title`:
|
||||||
|
|
||||||
@@ -1260,7 +1304,7 @@ $ yt-dlp -o "%(playlist)s/%(playlist_index)s - %(title)s.%(ext)s" "https://www.y
|
|||||||
$ yt-dlp -o "%(upload_date>%Y)s/%(title)s.%(ext)s" "https://www.youtube.com/playlist?list=PLwiyx1dc3P2JR9N8gQaQN_BCvlSlap7re"
|
$ yt-dlp -o "%(upload_date>%Y)s/%(title)s.%(ext)s" "https://www.youtube.com/playlist?list=PLwiyx1dc3P2JR9N8gQaQN_BCvlSlap7re"
|
||||||
|
|
||||||
# Prefix playlist index with " - " separator, but only if it is available
|
# Prefix playlist index with " - " separator, but only if it is available
|
||||||
$ yt-dlp -o '%(playlist_index|)s%(playlist_index& - |)s%(title)s.%(ext)s' BaW_jenozKc https://www.youtube.com/user/TheLinuxFoundation/playlists
|
$ yt-dlp -o '%(playlist_index|)s%(playlist_index& - |)s%(title)s.%(ext)s' BaW_jenozKc "https://www.youtube.com/user/TheLinuxFoundation/playlists"
|
||||||
|
|
||||||
# Download all playlists of YouTube channel/user keeping each playlist in separate directory:
|
# Download all playlists of YouTube channel/user keeping each playlist in separate directory:
|
||||||
$ yt-dlp -o "%(uploader)s/%(playlist)s/%(playlist_index)s - %(title)s.%(ext)s" "https://www.youtube.com/user/TheLinuxFoundation/playlists"
|
$ yt-dlp -o "%(uploader)s/%(playlist)s/%(playlist_index)s - %(title)s.%(ext)s" "https://www.youtube.com/user/TheLinuxFoundation/playlists"
|
||||||
@@ -1271,6 +1315,13 @@ $ yt-dlp -u user -p password -P "~/MyVideos" -o "%(playlist)s/%(chapter_number)s
|
|||||||
# Download entire series season keeping each series and each season in separate directory under C:/MyVideos
|
# Download entire series season keeping each series and each season in separate directory under C:/MyVideos
|
||||||
$ yt-dlp -P "C:/MyVideos" -o "%(series)s/%(season_number)s - %(season)s/%(episode_number)s - %(episode)s.%(ext)s" "https://videomore.ru/kino_v_detalayah/5_sezon/367617"
|
$ yt-dlp -P "C:/MyVideos" -o "%(series)s/%(season_number)s - %(season)s/%(episode_number)s - %(episode)s.%(ext)s" "https://videomore.ru/kino_v_detalayah/5_sezon/367617"
|
||||||
|
|
||||||
|
# Download video as "C:\MyVideos\uploader\title.ext", subtitles as "C:\MyVideos\subs\uploader\title.ext"
|
||||||
|
# and put all temporary files in "C:\MyVideos\tmp"
|
||||||
|
$ yt-dlp -P "C:/MyVideos" -P "temp:tmp" -P "subtitle:subs" -o "%(uploader)s/%(title)s.%(ext)s" BaW_jenoz --write-subs
|
||||||
|
|
||||||
|
# Download video as "C:\MyVideos\uploader\title.ext" and subtitles as "C:\MyVideos\uploader\subs\title.ext"
|
||||||
|
$ yt-dlp -P "C:/MyVideos" -o "%(uploader)s/%(title)s.%(ext)s" -o "subtitle:%(uploader)s/subs/%(title)s.%(ext)s" BaW_jenozKc --write-subs
|
||||||
|
|
||||||
# Stream the video being downloaded to stdout
|
# Stream the video being downloaded to stdout
|
||||||
$ yt-dlp -o - BaW_jenozKc
|
$ yt-dlp -o - BaW_jenozKc
|
||||||
```
|
```
|
||||||
@@ -1366,10 +1417,10 @@ The available fields are:
|
|||||||
|
|
||||||
- `hasvid`: Gives priority to formats that has a video stream
|
- `hasvid`: Gives priority to formats that has a video stream
|
||||||
- `hasaud`: Gives priority to formats that has a audio stream
|
- `hasaud`: Gives priority to formats that has a audio stream
|
||||||
- `ie_pref`: The format preference as given by the extractor
|
- `ie_pref`: The format preference
|
||||||
- `lang`: Language preference as given by the extractor
|
- `lang`: The language preference
|
||||||
- `quality`: The quality of the format as given by the extractor
|
- `quality`: The quality of the format
|
||||||
- `source`: Preference of the source as given by the extractor
|
- `source`: The preference of the source
|
||||||
- `proto`: Protocol used for download (`https`/`ftps` > `http`/`ftp` > `m3u8_native`/`m3u8` > `http_dash_segments`> `websocket_frag` > `mms`/`rtsp` > `f4f`/`f4m`)
|
- `proto`: Protocol used for download (`https`/`ftps` > `http`/`ftp` > `m3u8_native`/`m3u8` > `http_dash_segments`> `websocket_frag` > `mms`/`rtsp` > `f4f`/`f4m`)
|
||||||
- `vcodec`: Video Codec (`av01` > `vp9.2` > `vp9` > `h265` > `h264` > `vp8` > `h263` > `theora` > other)
|
- `vcodec`: Video Codec (`av01` > `vp9.2` > `vp9` > `h265` > `h264` > `vp8` > `h263` > `theora` > other)
|
||||||
- `acodec`: Audio Codec (`flac`/`alac` > `wav`/`aiff` > `opus` > `vorbis` > `aac` > `mp4a` > `mp3` > `eac3` > `ac3` > `dts` > other)
|
- `acodec`: Audio Codec (`flac`/`alac` > `wav`/`aiff` > `opus` > `vorbis` > `aac` > `mp4a` > `mp3` > `eac3` > `ac3` > `dts` > other)
|
||||||
@@ -1537,7 +1588,7 @@ Note that any field created by this can be used in the [output template](#output
|
|||||||
|
|
||||||
This option also has a few special uses:
|
This option also has a few special uses:
|
||||||
* You can download an additional URL based on the metadata of the currently downloaded video. To do this, set the field `additional_urls` to the URL that you want to download. Eg: `--parse-metadata "description:(?P<additional_urls>https?://www\.vimeo\.com/\d+)` will download the first vimeo video found in the description
|
* You can download an additional URL based on the metadata of the currently downloaded video. To do this, set the field `additional_urls` to the URL that you want to download. Eg: `--parse-metadata "description:(?P<additional_urls>https?://www\.vimeo\.com/\d+)` will download the first vimeo video found in the description
|
||||||
* You can use this to change the metadata that is embedded in the media file. To do this, set the value of the corresponding field with a `meta_` prefix. For example, any value you set to `meta_description` field will be added to the `description` field in the file. For example, you can use this to set a different "description" and "synopsis". Any value set to the `meta_` field will overwrite all default values.
|
* You can use this to change the metadata that is embedded in the media file. To do this, set the value of the corresponding field with a `meta_` prefix. For example, any value you set to `meta_description` field will be added to the `description` field in the file. For example, you can use this to set a different "description" and "synopsis". To modify the metadata of individual streams, use the `meta<n>_` prefix (Eg: `meta1_language`). Any value set to the `meta_` field will overwrite all default values.
|
||||||
|
|
||||||
For reference, these are the fields yt-dlp adds by default to the file metadata:
|
For reference, these are the fields yt-dlp adds by default to the file metadata:
|
||||||
|
|
||||||
@@ -1621,6 +1672,11 @@ The following extractors use this feature:
|
|||||||
#### gamejolt
|
#### gamejolt
|
||||||
* `comment_sort`: `hot` (default), `you` (cookies needed), `top`, `new` - choose comment sorting mode (on GameJolt's side)
|
* `comment_sort`: `hot` (default), `you` (cookies needed), `top`, `new` - choose comment sorting mode (on GameJolt's side)
|
||||||
|
|
||||||
|
#### hotstar
|
||||||
|
* `res`: resolution to ignore - one or more of `sd`, `hd`, `fhd`
|
||||||
|
* `vcodec`: vcodec to ignore - one or more of `h264`, `h265`, `dvh265`
|
||||||
|
* `dr`: dynamic range to ignore - one or more of `sdr`, `hdr10`, `dv`
|
||||||
|
|
||||||
NOTE: These options may be changed/removed in the future without concern for backward compatibility
|
NOTE: These options may be changed/removed in the future without concern for backward compatibility
|
||||||
|
|
||||||
<!-- MANPAGE: MOVE "INSTALLATION" SECTION HERE -->
|
<!-- MANPAGE: MOVE "INSTALLATION" SECTION HERE -->
|
||||||
@@ -1656,7 +1712,7 @@ with YoutubeDL(ydl_opts) as ydl:
|
|||||||
ydl.download(['https://www.youtube.com/watch?v=BaW_jenozKc'])
|
ydl.download(['https://www.youtube.com/watch?v=BaW_jenozKc'])
|
||||||
```
|
```
|
||||||
|
|
||||||
Most likely, you'll want to use various options. For a list of options available, have a look at [`yt_dlp/YoutubeDL.py`](yt_dlp/YoutubeDL.py#L162).
|
Most likely, you'll want to use various options. For a list of options available, have a look at [`yt_dlp/YoutubeDL.py`](yt_dlp/YoutubeDL.py#L191).
|
||||||
|
|
||||||
Here's a more complete example demonstrating various functionality:
|
Here's a more complete example demonstrating various functionality:
|
||||||
|
|
||||||
@@ -1762,6 +1818,14 @@ with yt_dlp.YoutubeDL(ydl_opts) as ydl:
|
|||||||
|
|
||||||
These are all the deprecated options and the current alternative to achieve the same effect
|
These are all the deprecated options and the current alternative to achieve the same effect
|
||||||
|
|
||||||
|
#### Almost redundant options
|
||||||
|
While these options are almost the same as their new counterparts, there are some differences that prevents them being redundant
|
||||||
|
|
||||||
|
-j, --dump-json --print "%()j"
|
||||||
|
-F, --list-formats --print formats_table
|
||||||
|
--list-thumbnails --print thumbnails_table --print playlist:thumbnails_table
|
||||||
|
--list-subs --print automatic_captions_table --print subtitles_table
|
||||||
|
|
||||||
#### Redundant options
|
#### Redundant options
|
||||||
While these options are redundant, they are still expected to be used due to their ease of use
|
While these options are redundant, they are still expected to be used due to their ease of use
|
||||||
|
|
||||||
@@ -1773,7 +1837,6 @@ While these options are redundant, they are still expected to be used due to the
|
|||||||
--get-thumbnail --print thumbnail
|
--get-thumbnail --print thumbnail
|
||||||
-e, --get-title --print title
|
-e, --get-title --print title
|
||||||
-g, --get-url --print urls
|
-g, --get-url --print urls
|
||||||
-j, --dump-json --print "%()j"
|
|
||||||
--match-title REGEX --match-filter "title ~= (?i)REGEX"
|
--match-title REGEX --match-filter "title ~= (?i)REGEX"
|
||||||
--reject-title REGEX --match-filter "title !~= (?i)REGEX"
|
--reject-title REGEX --match-filter "title !~= (?i)REGEX"
|
||||||
--min-views COUNT --match-filter "view_count >=? COUNT"
|
--min-views COUNT --match-filter "view_count >=? COUNT"
|
||||||
@@ -1783,6 +1846,8 @@ While these options are redundant, they are still expected to be used due to the
|
|||||||
#### Not recommended
|
#### Not recommended
|
||||||
While these options still work, their use is not recommended since there are other alternatives to achieve the same
|
While these options still work, their use is not recommended since there are other alternatives to achieve the same
|
||||||
|
|
||||||
|
--exec-before-download CMD --exec "before_dl:CMD"
|
||||||
|
--no-exec-before-download --no-exec
|
||||||
--all-formats -f all
|
--all-formats -f all
|
||||||
--all-subs --sub-langs all --write-subs
|
--all-subs --sub-langs all --write-subs
|
||||||
--print-json -j --no-simulate
|
--print-json -j --no-simulate
|
||||||
|
|||||||
+52
-5
@@ -41,6 +41,7 @@
|
|||||||
- **aenetworks:collection**
|
- **aenetworks:collection**
|
||||||
- **aenetworks:show**
|
- **aenetworks:show**
|
||||||
- **afreecatv**: afreecatv.com
|
- **afreecatv**: afreecatv.com
|
||||||
|
- **afreecatv:live**: afreecatv.com
|
||||||
- **AirMozilla**
|
- **AirMozilla**
|
||||||
- **AliExpressLive**
|
- **AliExpressLive**
|
||||||
- **AlJazeera**
|
- **AlJazeera**
|
||||||
@@ -162,6 +163,7 @@
|
|||||||
- **BuzzFeed**
|
- **BuzzFeed**
|
||||||
- **BYUtv**
|
- **BYUtv**
|
||||||
- **CableAV**
|
- **CableAV**
|
||||||
|
- **Callin**
|
||||||
- **CAM4**
|
- **CAM4**
|
||||||
- **Camdemy**
|
- **Camdemy**
|
||||||
- **CamdemyFolder**
|
- **CamdemyFolder**
|
||||||
@@ -232,6 +234,8 @@
|
|||||||
- **Cracked**
|
- **Cracked**
|
||||||
- **Crackle**
|
- **Crackle**
|
||||||
- **CrooksAndLiars**
|
- **CrooksAndLiars**
|
||||||
|
- **CrowdBunker**
|
||||||
|
- **CrowdBunkerChannel**
|
||||||
- **crunchyroll**
|
- **crunchyroll**
|
||||||
- **crunchyroll:beta**
|
- **crunchyroll:beta**
|
||||||
- **crunchyroll:playlist**
|
- **crunchyroll:playlist**
|
||||||
@@ -246,6 +250,7 @@
|
|||||||
- **curiositystream:collections**
|
- **curiositystream:collections**
|
||||||
- **curiositystream:series**
|
- **curiositystream:series**
|
||||||
- **CWTV**
|
- **CWTV**
|
||||||
|
- **Daftsex**
|
||||||
- **DagelijkseKost**: dagelijksekost.een.be
|
- **DagelijkseKost**: dagelijksekost.een.be
|
||||||
- **DailyMail**
|
- **DailyMail**
|
||||||
- **dailymotion**
|
- **dailymotion**
|
||||||
@@ -265,6 +270,7 @@
|
|||||||
- **democracynow**
|
- **democracynow**
|
||||||
- **DHM**: Filmarchiv - Deutsches Historisches Museum
|
- **DHM**: Filmarchiv - Deutsches Historisches Museum
|
||||||
- **Digg**
|
- **Digg**
|
||||||
|
- **DigitalConcertHall**: DigitalConcertHall extractor
|
||||||
- **DigitallySpeaking**
|
- **DigitallySpeaking**
|
||||||
- **Digiteka**
|
- **Digiteka**
|
||||||
- **Discovery**
|
- **Discovery**
|
||||||
@@ -288,6 +294,7 @@
|
|||||||
- **DouyuTV**: 斗鱼
|
- **DouyuTV**: 斗鱼
|
||||||
- **DPlay**
|
- **DPlay**
|
||||||
- **DRBonanza**
|
- **DRBonanza**
|
||||||
|
- **Drooble**
|
||||||
- **Dropbox**
|
- **Dropbox**
|
||||||
- **Dropout**
|
- **Dropout**
|
||||||
- **DropoutSeason**
|
- **DropoutSeason**
|
||||||
@@ -330,6 +337,7 @@
|
|||||||
- **ESPNCricInfo**
|
- **ESPNCricInfo**
|
||||||
- **EsriVideo**
|
- **EsriVideo**
|
||||||
- **Europa**
|
- **Europa**
|
||||||
|
- **EuropeanTour**
|
||||||
- **EUScreen**
|
- **EUScreen**
|
||||||
- **EWETV**
|
- **EWETV**
|
||||||
- **ExpoTV**
|
- **ExpoTV**
|
||||||
@@ -407,6 +415,8 @@
|
|||||||
- **Glide**: Glide mobile video messages (glide.me)
|
- **Glide**: Glide mobile video messages (glide.me)
|
||||||
- **Globo**
|
- **Globo**
|
||||||
- **GloboArticle**
|
- **GloboArticle**
|
||||||
|
- **glomex**: Glomex videos
|
||||||
|
- **glomex:embed**: Glomex embedded videos
|
||||||
- **Go**
|
- **Go**
|
||||||
- **GodTube**
|
- **GodTube**
|
||||||
- **Gofile**
|
- **Gofile**
|
||||||
@@ -470,6 +480,7 @@
|
|||||||
- **IndavideoEmbed**
|
- **IndavideoEmbed**
|
||||||
- **InfoQ**
|
- **InfoQ**
|
||||||
- **Instagram**
|
- **Instagram**
|
||||||
|
- **instagram:story**
|
||||||
- **instagram:tag**: Instagram hashtag search URLs
|
- **instagram:tag**: Instagram hashtag search URLs
|
||||||
- **instagram:user**: Instagram user profile
|
- **instagram:user**: Instagram user profile
|
||||||
- **InstagramIOS**: IOS instagram:// URL
|
- **InstagramIOS**: IOS instagram:// URL
|
||||||
@@ -477,6 +488,8 @@
|
|||||||
- **InternetVideoArchive**
|
- **InternetVideoArchive**
|
||||||
- **IPrima**
|
- **IPrima**
|
||||||
- **IPrimaCNN**
|
- **IPrimaCNN**
|
||||||
|
- **iq.com**: International version of iQiyi
|
||||||
|
- **iq.com:album**
|
||||||
- **iqiyi**: 爱奇艺
|
- **iqiyi**: 爱奇艺
|
||||||
- **Ir90Tv**
|
- **Ir90Tv**
|
||||||
- **ITTF**
|
- **ITTF**
|
||||||
@@ -500,6 +513,7 @@
|
|||||||
- **KarriereVideos**
|
- **KarriereVideos**
|
||||||
- **Katsomo**
|
- **Katsomo**
|
||||||
- **KeezMovies**
|
- **KeezMovies**
|
||||||
|
- **KelbyOne**
|
||||||
- **Ketnet**
|
- **Ketnet**
|
||||||
- **khanacademy**
|
- **khanacademy**
|
||||||
- **khanacademy:unit**
|
- **khanacademy:unit**
|
||||||
@@ -545,7 +559,6 @@
|
|||||||
- **limelight:channel_list**
|
- **limelight:channel_list**
|
||||||
- **LineLive**
|
- **LineLive**
|
||||||
- **LineLiveChannel**
|
- **LineLiveChannel**
|
||||||
- **LineTV**
|
|
||||||
- **LinkedIn**
|
- **LinkedIn**
|
||||||
- **linkedin:learning**
|
- **linkedin:learning**
|
||||||
- **linkedin:learning:course**
|
- **linkedin:learning:course**
|
||||||
@@ -554,6 +567,7 @@
|
|||||||
- **LiveJournal**
|
- **LiveJournal**
|
||||||
- **livestream**
|
- **livestream**
|
||||||
- **livestream:original**
|
- **livestream:original**
|
||||||
|
- **Lnk**
|
||||||
- **LnkGo**
|
- **LnkGo**
|
||||||
- **loc**: Library of Congress
|
- **loc**: Library of Congress
|
||||||
- **LocalNews8**
|
- **LocalNews8**
|
||||||
@@ -566,6 +580,7 @@
|
|||||||
- **mailru**: Видео@Mail.Ru
|
- **mailru**: Видео@Mail.Ru
|
||||||
- **mailru:music**: Музыка@Mail.Ru
|
- **mailru:music**: Музыка@Mail.Ru
|
||||||
- **mailru:music:search**: Музыка@Mail.Ru
|
- **mailru:music:search**: Музыка@Mail.Ru
|
||||||
|
- **MainStreaming**: MainStreaming Player
|
||||||
- **MallTV**
|
- **MallTV**
|
||||||
- **mangomolo:live**
|
- **mangomolo:live**
|
||||||
- **mangomolo:video**
|
- **mangomolo:video**
|
||||||
@@ -592,6 +607,8 @@
|
|||||||
- **MediasiteNamedCatalog**
|
- **MediasiteNamedCatalog**
|
||||||
- **Medici**
|
- **Medici**
|
||||||
- **megaphone.fm**: megaphone.fm embedded players
|
- **megaphone.fm**: megaphone.fm embedded players
|
||||||
|
- **megatvcom**: megatv.com videos
|
||||||
|
- **megatvcom:embed**: megatv.com embedded videos
|
||||||
- **Meipai**: 美拍
|
- **Meipai**: 美拍
|
||||||
- **MelonVOD**
|
- **MelonVOD**
|
||||||
- **META**
|
- **META**
|
||||||
@@ -615,6 +632,7 @@
|
|||||||
- **mirrativ:user**
|
- **mirrativ:user**
|
||||||
- **MiTele**: mitele.es
|
- **MiTele**: mitele.es
|
||||||
- **mixch**
|
- **mixch**
|
||||||
|
- **mixch:archive**
|
||||||
- **mixcloud**
|
- **mixcloud**
|
||||||
- **mixcloud:playlist**
|
- **mixcloud:playlist**
|
||||||
- **mixcloud:user**
|
- **mixcloud:user**
|
||||||
@@ -704,6 +722,7 @@
|
|||||||
- **Newgrounds:playlist**
|
- **Newgrounds:playlist**
|
||||||
- **Newgrounds:user**
|
- **Newgrounds:user**
|
||||||
- **Newstube**
|
- **Newstube**
|
||||||
|
- **Newsy**
|
||||||
- **NextMedia**: 蘋果日報
|
- **NextMedia**: 蘋果日報
|
||||||
- **NextMediaActionNews**: 蘋果日報 - 動新聞
|
- **NextMediaActionNews**: 蘋果日報 - 動新聞
|
||||||
- **NextTV**: 壹電視
|
- **NextTV**: 壹電視
|
||||||
@@ -733,6 +752,7 @@
|
|||||||
- **NJPWWorld**: 新日本プロレスワールド
|
- **NJPWWorld**: 新日本プロレスワールド
|
||||||
- **NobelPrize**
|
- **NobelPrize**
|
||||||
- **NonkTube**
|
- **NonkTube**
|
||||||
|
- **NoodleMagazine**
|
||||||
- **Noovo**
|
- **Noovo**
|
||||||
- **Normalboots**
|
- **Normalboots**
|
||||||
- **NosVideo**
|
- **NosVideo**
|
||||||
@@ -785,6 +805,7 @@
|
|||||||
- **OpencastPlaylist**
|
- **OpencastPlaylist**
|
||||||
- **openrec**
|
- **openrec**
|
||||||
- **openrec:capture**
|
- **openrec:capture**
|
||||||
|
- **openrec:movie**
|
||||||
- **OraTV**
|
- **OraTV**
|
||||||
- **orf:burgenland**: Radio Burgenland
|
- **orf:burgenland**: Radio Burgenland
|
||||||
- **orf:fm4**: radio FM4
|
- **orf:fm4**: radio FM4
|
||||||
@@ -836,6 +857,8 @@
|
|||||||
- **Pinkbike**
|
- **Pinkbike**
|
||||||
- **Pinterest**
|
- **Pinterest**
|
||||||
- **PinterestCollection**
|
- **PinterestCollection**
|
||||||
|
- **pixiv:sketch**
|
||||||
|
- **pixiv:sketch:user**
|
||||||
- **Pladform**
|
- **Pladform**
|
||||||
- **PlanetMarathi**
|
- **PlanetMarathi**
|
||||||
- **Platzi**
|
- **Platzi**
|
||||||
@@ -854,6 +877,8 @@
|
|||||||
- **podomatic**
|
- **podomatic**
|
||||||
- **Pokemon**
|
- **Pokemon**
|
||||||
- **PokemonWatch**
|
- **PokemonWatch**
|
||||||
|
- **PokerGo**
|
||||||
|
- **PokerGoCollection**
|
||||||
- **PolsatGo**
|
- **PolsatGo**
|
||||||
- **PolskieRadio**
|
- **PolskieRadio**
|
||||||
- **polskieradio:kierowcow**
|
- **polskieradio:kierowcow**
|
||||||
@@ -865,6 +890,7 @@
|
|||||||
- **PopcornTV**
|
- **PopcornTV**
|
||||||
- **PornCom**
|
- **PornCom**
|
||||||
- **PornerBros**
|
- **PornerBros**
|
||||||
|
- **Pornez**
|
||||||
- **PornFlip**
|
- **PornFlip**
|
||||||
- **PornHd**
|
- **PornHd**
|
||||||
- **PornHub**: PornHub and Thumbzilla
|
- **PornHub**: PornHub and Thumbzilla
|
||||||
@@ -879,6 +905,11 @@
|
|||||||
- **PressTV**
|
- **PressTV**
|
||||||
- **ProjectVeritas**
|
- **ProjectVeritas**
|
||||||
- **prosiebensat1**: ProSiebenSat.1 Digital
|
- **prosiebensat1**: ProSiebenSat.1 Digital
|
||||||
|
- **PRXAccount**
|
||||||
|
- **PRXSeries**
|
||||||
|
- **prxseries:search**: PRX Series Search; "prxseries:" prefix
|
||||||
|
- **prxstories:search**: PRX Stories Search; "prxstories:" prefix
|
||||||
|
- **PRXStory**
|
||||||
- **puhutv**
|
- **puhutv**
|
||||||
- **puhutv:serie**
|
- **puhutv:serie**
|
||||||
- **Puls4**
|
- **Puls4**
|
||||||
@@ -912,8 +943,9 @@
|
|||||||
- **RaiPlay**
|
- **RaiPlay**
|
||||||
- **RaiPlayLive**
|
- **RaiPlayLive**
|
||||||
- **RaiPlayPlaylist**
|
- **RaiPlayPlaylist**
|
||||||
- **RaiPlayRadio**
|
- **RaiPlaySound**
|
||||||
- **RaiPlayRadioPlaylist**
|
- **RaiPlaySoundLive**
|
||||||
|
- **RaiPlaySoundPlaylist**
|
||||||
- **RayWenderlich**
|
- **RayWenderlich**
|
||||||
- **RayWenderlichCourse**
|
- **RayWenderlichCourse**
|
||||||
- **RBMARadio**
|
- **RBMARadio**
|
||||||
@@ -948,12 +980,15 @@
|
|||||||
- **Roxwel**
|
- **Roxwel**
|
||||||
- **Rozhlas**
|
- **Rozhlas**
|
||||||
- **RTBF**
|
- **RTBF**
|
||||||
|
- **RTDocumentry**
|
||||||
|
- **RTDocumentryPlaylist**
|
||||||
- **rte**: Raidió Teilifís Éireann TV
|
- **rte**: Raidió Teilifís Éireann TV
|
||||||
- **rte:radio**: Raidió Teilifís Éireann radio
|
- **rte:radio**: Raidió Teilifís Éireann radio
|
||||||
- **rtl.nl**: rtl.nl and rtlxl.nl
|
- **rtl.nl**: rtl.nl and rtlxl.nl
|
||||||
- **rtl2**
|
- **rtl2**
|
||||||
- **rtl2:you**
|
- **rtl2:you**
|
||||||
- **rtl2:you:series**
|
- **rtl2:you:series**
|
||||||
|
- **RTNews**
|
||||||
- **RTP**
|
- **RTP**
|
||||||
- **RTRFM**
|
- **RTRFM**
|
||||||
- **RTS**: RTS.ch
|
- **RTS**: RTS.ch
|
||||||
@@ -965,8 +1000,10 @@
|
|||||||
- **RTVNH**
|
- **RTVNH**
|
||||||
- **RTVS**
|
- **RTVS**
|
||||||
- **RUHD**
|
- **RUHD**
|
||||||
|
- **Rule34Video**
|
||||||
- **RumbleChannel**
|
- **RumbleChannel**
|
||||||
- **RumbleEmbed**
|
- **RumbleEmbed**
|
||||||
|
- **Ruptly**
|
||||||
- **rutube**: Rutube videos
|
- **rutube**: Rutube videos
|
||||||
- **rutube:channel**: Rutube channel
|
- **rutube:channel**: Rutube channel
|
||||||
- **rutube:embed**: Rutube embedded videos
|
- **rutube:embed**: Rutube embedded videos
|
||||||
@@ -1107,7 +1144,10 @@
|
|||||||
- **TeamTreeHouse**
|
- **TeamTreeHouse**
|
||||||
- **TechTalks**
|
- **TechTalks**
|
||||||
- **techtv.mit.edu**
|
- **techtv.mit.edu**
|
||||||
- **ted**
|
- **TedEmbed**
|
||||||
|
- **TedPlaylist**
|
||||||
|
- **TedSeries**
|
||||||
|
- **TedTalk**
|
||||||
- **Tele13**
|
- **Tele13**
|
||||||
- **Tele5**
|
- **Tele5**
|
||||||
- **TeleBruxelles**
|
- **TeleBruxelles**
|
||||||
@@ -1141,6 +1181,9 @@
|
|||||||
- **ThreeSpeak**
|
- **ThreeSpeak**
|
||||||
- **ThreeSpeakUser**
|
- **ThreeSpeakUser**
|
||||||
- **TikTok**
|
- **TikTok**
|
||||||
|
- **tiktok:effect**
|
||||||
|
- **tiktok:sound**
|
||||||
|
- **tiktok:tag**
|
||||||
- **tiktok:user**
|
- **tiktok:user**
|
||||||
- **tinypic**: tinypic.com videos
|
- **tinypic**: tinypic.com videos
|
||||||
- **TMZ**
|
- **TMZ**
|
||||||
@@ -1202,6 +1245,8 @@
|
|||||||
- **TVNowNew**
|
- **TVNowNew**
|
||||||
- **TVNowSeason**
|
- **TVNowSeason**
|
||||||
- **TVNowShow**
|
- **TVNowShow**
|
||||||
|
- **tvopengr:embed**: tvopen.gr embedded videos
|
||||||
|
- **tvopengr:watch**: tvopen.gr (and ethnos.gr) videos
|
||||||
- **tvp**: Telewizja Polska
|
- **tvp**: Telewizja Polska
|
||||||
- **tvp:embed**: Telewizja Polska
|
- **tvp:embed**: Telewizja Polska
|
||||||
- **tvp:series**
|
- **tvp:series**
|
||||||
@@ -1294,6 +1339,7 @@
|
|||||||
- **vimeo:review**: Review pages on vimeo
|
- **vimeo:review**: Review pages on vimeo
|
||||||
- **vimeo:user**
|
- **vimeo:user**
|
||||||
- **vimeo:watchlater**: Vimeo watch later list, "vimeowatchlater" keyword (requires authentication)
|
- **vimeo:watchlater**: Vimeo watch later list, "vimeowatchlater" keyword (requires authentication)
|
||||||
|
- **Vimm**
|
||||||
- **Vimple**: Vimple - one-click video hosting
|
- **Vimple**: Vimple - one-click video hosting
|
||||||
- **Vine**
|
- **Vine**
|
||||||
- **vine:user**
|
- **vine:user**
|
||||||
@@ -1420,9 +1466,10 @@
|
|||||||
- **youtube:search_url**: YouTube search URLs with sorting and filter support
|
- **youtube:search_url**: YouTube search URLs with sorting and filter support
|
||||||
- **youtube:subscriptions**: YouTube subscriptions feed; ":ytsubs" keyword (requires cookies)
|
- **youtube:subscriptions**: YouTube subscriptions feed; ":ytsubs" keyword (requires cookies)
|
||||||
- **youtube:tab**: YouTube Tabs
|
- **youtube:tab**: YouTube Tabs
|
||||||
|
- **youtube:user**: YouTube user videos; "ytuser:" prefix
|
||||||
- **youtube:watchlater**: Youtube watch later list; ":ytwatchlater" keyword (requires cookies)
|
- **youtube:watchlater**: Youtube watch later list; ":ytwatchlater" keyword (requires cookies)
|
||||||
|
- **YoutubeLivestreamEmbed**: YouTube livestream embeds
|
||||||
- **YoutubeYtBe**: youtu.be
|
- **YoutubeYtBe**: youtu.be
|
||||||
- **YoutubeYtUser**: YouTube user videos; "ytuser:" prefix
|
|
||||||
- **Zapiks**
|
- **Zapiks**
|
||||||
- **Zattoo**
|
- **Zattoo**
|
||||||
- **ZattooLive**
|
- **ZattooLive**
|
||||||
|
|||||||
+6
-2
@@ -211,7 +211,7 @@ def sanitize_got_info_dict(got_dict):
|
|||||||
|
|
||||||
# Auto-generated
|
# Auto-generated
|
||||||
'autonumber', 'playlist', 'format_index', 'video_ext', 'audio_ext', 'duration_string', 'epoch',
|
'autonumber', 'playlist', 'format_index', 'video_ext', 'audio_ext', 'duration_string', 'epoch',
|
||||||
'fulltitle', 'extractor', 'extractor_key', 'filepath', 'infojson_filename', 'original_url',
|
'fulltitle', 'extractor', 'extractor_key', 'filepath', 'infojson_filename', 'original_url', 'n_entries',
|
||||||
|
|
||||||
# Only live_status needs to be checked
|
# Only live_status needs to be checked
|
||||||
'is_live', 'was_live',
|
'is_live', 'was_live',
|
||||||
@@ -224,6 +224,8 @@ def sanitize_got_info_dict(got_dict):
|
|||||||
return f'md5:{md5(value)}'
|
return f'md5:{md5(value)}'
|
||||||
elif isinstance(value, list) and len(value) > 10:
|
elif isinstance(value, list) and len(value) > 10:
|
||||||
return f'count:{len(value)}'
|
return f'count:{len(value)}'
|
||||||
|
elif key.endswith('_count') and isinstance(value, int):
|
||||||
|
return int
|
||||||
return value
|
return value
|
||||||
|
|
||||||
test_info_dict = {
|
test_info_dict = {
|
||||||
@@ -233,7 +235,7 @@ def sanitize_got_info_dict(got_dict):
|
|||||||
}
|
}
|
||||||
|
|
||||||
# display_id may be generated from id
|
# display_id may be generated from id
|
||||||
if test_info_dict.get('display_id') == test_info_dict['id']:
|
if test_info_dict.get('display_id') == test_info_dict.get('id'):
|
||||||
test_info_dict.pop('display_id')
|
test_info_dict.pop('display_id')
|
||||||
|
|
||||||
return test_info_dict
|
return test_info_dict
|
||||||
@@ -259,6 +261,8 @@ def expect_info_dict(self, got_dict, expected_dict):
|
|||||||
def _repr(v):
|
def _repr(v):
|
||||||
if isinstance(v, compat_str):
|
if isinstance(v, compat_str):
|
||||||
return "'%s'" % v.replace('\\', '\\\\').replace("'", "\\'").replace('\n', '\\n')
|
return "'%s'" % v.replace('\\', '\\\\').replace("'", "\\'").replace('\n', '\\n')
|
||||||
|
elif isinstance(v, type):
|
||||||
|
return v.__name__
|
||||||
else:
|
else:
|
||||||
return repr(v)
|
return repr(v)
|
||||||
info_dict_str = ''
|
info_dict_str = ''
|
||||||
|
|||||||
@@ -208,6 +208,91 @@ class TestInfoExtractor(unittest.TestCase):
|
|||||||
},
|
},
|
||||||
{'expected_type': 'NewsArticle'},
|
{'expected_type': 'NewsArticle'},
|
||||||
),
|
),
|
||||||
|
(
|
||||||
|
r'''<script type="application/ld+json">
|
||||||
|
{"url":"/vrtnu/a-z/het-journaal/2021/het-journaal-het-journaal-19u-20211231/",
|
||||||
|
"name":"Het journaal 19u",
|
||||||
|
"description":"Het journaal 19u van vrijdag 31 december 2021.",
|
||||||
|
"potentialAction":{"url":"https://vrtnu.page.link/pfVy6ihgCAJKgHqe8","@type":"ShareAction"},
|
||||||
|
"mainEntityOfPage":{"@id":"1640092242445","@type":"WebPage"},
|
||||||
|
"publication":[{
|
||||||
|
"startDate":"2021-12-31T19:00:00.000+01:00",
|
||||||
|
"endDate":"2022-01-30T23:55:00.000+01:00",
|
||||||
|
"publishedBy":{"name":"een","@type":"Organization"},
|
||||||
|
"publishedOn":{"url":"https://www.vrt.be/vrtnu/","name":"VRT NU","@type":"BroadcastService"},
|
||||||
|
"@id":"pbs-pub-3a7ec233-da95-4c1e-9b2b-cf5fdfebcbe8",
|
||||||
|
"@type":"BroadcastEvent"
|
||||||
|
}],
|
||||||
|
"video":{
|
||||||
|
"name":"Het journaal - Aflevering 365 (Seizoen 2021)",
|
||||||
|
"description":"Het journaal 19u van vrijdag 31 december 2021. Bekijk aflevering 365 van seizoen 2021 met VRT NU via de site of app.",
|
||||||
|
"thumbnailUrl":"//images.vrt.be/width1280/2021/12/31/80d5ed00-6a64-11ec-b07d-02b7b76bf47f.jpg",
|
||||||
|
"expires":"2022-01-30T23:55:00.000+01:00",
|
||||||
|
"hasPart":[
|
||||||
|
{"name":"Explosie Turnhout","startOffset":70,"@type":"Clip"},
|
||||||
|
{"name":"Jaarwisseling","startOffset":440,"@type":"Clip"},
|
||||||
|
{"name":"Natuurbranden Colorado","startOffset":1179,"@type":"Clip"},
|
||||||
|
{"name":"Klimaatverandering","startOffset":1263,"@type":"Clip"},
|
||||||
|
{"name":"Zacht weer","startOffset":1367,"@type":"Clip"},
|
||||||
|
{"name":"Financiële balans","startOffset":1383,"@type":"Clip"},
|
||||||
|
{"name":"Club Brugge","startOffset":1484,"@type":"Clip"},
|
||||||
|
{"name":"Mentale gezondheid bij topsporters","startOffset":1575,"@type":"Clip"},
|
||||||
|
{"name":"Olympische Winterspelen","startOffset":1728,"@type":"Clip"},
|
||||||
|
{"name":"Sober oudjaar in Nederland","startOffset":1873,"@type":"Clip"}
|
||||||
|
],
|
||||||
|
"duration":"PT34M39.23S",
|
||||||
|
"uploadDate":"2021-12-31T19:00:00.000+01:00",
|
||||||
|
"@id":"vid-9457d0c6-b8ac-4aba-b5e1-15aa3a3295b5",
|
||||||
|
"@type":"VideoObject"
|
||||||
|
},
|
||||||
|
"genre":["Nieuws en actua"],
|
||||||
|
"episodeNumber":365,
|
||||||
|
"partOfSeries":{"name":"Het journaal","@id":"222831405527","@type":"TVSeries"},
|
||||||
|
"partOfSeason":{"name":"Seizoen 2021","@id":"961809365527","@type":"TVSeason"},
|
||||||
|
"@context":"https://schema.org","@id":"961685295527","@type":"TVEpisode"}</script>
|
||||||
|
''',
|
||||||
|
{
|
||||||
|
'chapters': [
|
||||||
|
{"title": "Explosie Turnhout", "start_time": 70, "end_time": 440},
|
||||||
|
{"title": "Jaarwisseling", "start_time": 440, "end_time": 1179},
|
||||||
|
{"title": "Natuurbranden Colorado", "start_time": 1179, "end_time": 1263},
|
||||||
|
{"title": "Klimaatverandering", "start_time": 1263, "end_time": 1367},
|
||||||
|
{"title": "Zacht weer", "start_time": 1367, "end_time": 1383},
|
||||||
|
{"title": "Financiële balans", "start_time": 1383, "end_time": 1484},
|
||||||
|
{"title": "Club Brugge", "start_time": 1484, "end_time": 1575},
|
||||||
|
{"title": "Mentale gezondheid bij topsporters", "start_time": 1575, "end_time": 1728},
|
||||||
|
{"title": "Olympische Winterspelen", "start_time": 1728, "end_time": 1873},
|
||||||
|
{"title": "Sober oudjaar in Nederland", "start_time": 1873, "end_time": 2079.23}
|
||||||
|
],
|
||||||
|
'title': 'Het journaal - Aflevering 365 (Seizoen 2021)'
|
||||||
|
}, {}
|
||||||
|
),
|
||||||
|
(
|
||||||
|
# test multiple thumbnails in a list
|
||||||
|
r'''
|
||||||
|
<script type="application/ld+json">
|
||||||
|
{"@context":"https://schema.org",
|
||||||
|
"@type":"VideoObject",
|
||||||
|
"thumbnailUrl":["https://www.rainews.it/cropgd/640x360/dl/img/2021/12/30/1640886376927_GettyImages.jpg"]}
|
||||||
|
</script>''',
|
||||||
|
{
|
||||||
|
'thumbnails': [{'url': 'https://www.rainews.it/cropgd/640x360/dl/img/2021/12/30/1640886376927_GettyImages.jpg'}],
|
||||||
|
},
|
||||||
|
{},
|
||||||
|
),
|
||||||
|
(
|
||||||
|
# test single thumbnail
|
||||||
|
r'''
|
||||||
|
<script type="application/ld+json">
|
||||||
|
{"@context":"https://schema.org",
|
||||||
|
"@type":"VideoObject",
|
||||||
|
"thumbnailUrl":"https://www.rainews.it/cropgd/640x360/dl/img/2021/12/30/1640886376927_GettyImages.jpg"}
|
||||||
|
</script>''',
|
||||||
|
{
|
||||||
|
'thumbnails': [{'url': 'https://www.rainews.it/cropgd/640x360/dl/img/2021/12/30/1640886376927_GettyImages.jpg'}],
|
||||||
|
},
|
||||||
|
{},
|
||||||
|
)
|
||||||
]
|
]
|
||||||
for html, expected_dict, search_json_ld_kwargs in _TESTS:
|
for html, expected_dict, search_json_ld_kwargs in _TESTS:
|
||||||
expect_dict(
|
expect_dict(
|
||||||
|
|||||||
@@ -30,6 +30,7 @@ class YDL(FakeYDL):
|
|||||||
self.msgs = []
|
self.msgs = []
|
||||||
|
|
||||||
def process_info(self, info_dict):
|
def process_info(self, info_dict):
|
||||||
|
info_dict = info_dict.copy()
|
||||||
info_dict.pop('__original_infodict', None)
|
info_dict.pop('__original_infodict', None)
|
||||||
self.downloaded_info_dicts.append(info_dict)
|
self.downloaded_info_dicts.append(info_dict)
|
||||||
|
|
||||||
@@ -645,6 +646,7 @@ class TestYoutubeDL(unittest.TestCase):
|
|||||||
'ext': 'mp4',
|
'ext': 'mp4',
|
||||||
'width': None,
|
'width': None,
|
||||||
'height': 1080,
|
'height': 1080,
|
||||||
|
'filesize': 1024,
|
||||||
'title1': '$PATH',
|
'title1': '$PATH',
|
||||||
'title2': '%PATH%',
|
'title2': '%PATH%',
|
||||||
'title3': 'foo/bar\\test',
|
'title3': 'foo/bar\\test',
|
||||||
@@ -778,8 +780,9 @@ class TestYoutubeDL(unittest.TestCase):
|
|||||||
test('%(title5)#U', 'a\u0301e\u0301i\u0301 𝐀')
|
test('%(title5)#U', 'a\u0301e\u0301i\u0301 𝐀')
|
||||||
test('%(title5)+U', 'áéí A')
|
test('%(title5)+U', 'áéí A')
|
||||||
test('%(title5)+#U', 'a\u0301e\u0301i\u0301 A')
|
test('%(title5)+#U', 'a\u0301e\u0301i\u0301 A')
|
||||||
test('%(height)D', '1K')
|
test('%(height)D', '1k')
|
||||||
test('%(height)5.2D', ' 1.08K')
|
test('%(filesize)#D', '1Ki')
|
||||||
|
test('%(height)5.2D', ' 1.08k')
|
||||||
test('%(title4)#S', 'foo_bar_test')
|
test('%(title4)#S', 'foo_bar_test')
|
||||||
test('%(title4).10S', ('foo \'bar\' ', 'foo \'bar\'' + ('#' if compat_os_name == 'nt' else ' ')))
|
test('%(title4).10S', ('foo \'bar\' ', 'foo \'bar\'' + ('#' if compat_os_name == 'nt' else ' ')))
|
||||||
if compat_os_name == 'nt':
|
if compat_os_name == 'nt':
|
||||||
@@ -906,7 +909,7 @@ class TestYoutubeDL(unittest.TestCase):
|
|||||||
def _match_entry(self, info_dict, incomplete=False):
|
def _match_entry(self, info_dict, incomplete=False):
|
||||||
res = super(FilterYDL, self)._match_entry(info_dict, incomplete)
|
res = super(FilterYDL, self)._match_entry(info_dict, incomplete)
|
||||||
if res is None:
|
if res is None:
|
||||||
self.downloaded_info_dicts.append(info_dict)
|
self.downloaded_info_dicts.append(info_dict.copy())
|
||||||
return res
|
return res
|
||||||
|
|
||||||
first = {
|
first = {
|
||||||
@@ -1151,6 +1154,7 @@ class TestYoutubeDL(unittest.TestCase):
|
|||||||
self.assertTrue(entries[1] is None)
|
self.assertTrue(entries[1] is None)
|
||||||
self.assertEqual(len(ydl.downloaded_info_dicts), 1)
|
self.assertEqual(len(ydl.downloaded_info_dicts), 1)
|
||||||
downloaded = ydl.downloaded_info_dicts[0]
|
downloaded = ydl.downloaded_info_dicts[0]
|
||||||
|
entries[2].pop('requested_downloads', None)
|
||||||
self.assertEqual(entries[2], downloaded)
|
self.assertEqual(entries[2], downloaded)
|
||||||
self.assertEqual(downloaded['url'], TEST_URL)
|
self.assertEqual(downloaded['url'], TEST_URL)
|
||||||
self.assertEqual(downloaded['title'], 'Video Transparent 2')
|
self.assertEqual(downloaded['title'], 'Video Transparent 2')
|
||||||
|
|||||||
+34
-2
@@ -8,6 +8,8 @@ from yt_dlp.cookies import (
|
|||||||
WindowsChromeCookieDecryptor,
|
WindowsChromeCookieDecryptor,
|
||||||
parse_safari_cookies,
|
parse_safari_cookies,
|
||||||
pbkdf2_sha1,
|
pbkdf2_sha1,
|
||||||
|
_get_linux_desktop_environment,
|
||||||
|
_LinuxDesktopEnvironment,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -42,6 +44,37 @@ class MonkeyPatch:
|
|||||||
|
|
||||||
|
|
||||||
class TestCookies(unittest.TestCase):
|
class TestCookies(unittest.TestCase):
|
||||||
|
def test_get_desktop_environment(self):
|
||||||
|
""" based on https://chromium.googlesource.com/chromium/src/+/refs/heads/main/base/nix/xdg_util_unittest.cc """
|
||||||
|
test_cases = [
|
||||||
|
({}, _LinuxDesktopEnvironment.OTHER),
|
||||||
|
|
||||||
|
({'DESKTOP_SESSION': 'gnome'}, _LinuxDesktopEnvironment.GNOME),
|
||||||
|
({'DESKTOP_SESSION': 'mate'}, _LinuxDesktopEnvironment.GNOME),
|
||||||
|
({'DESKTOP_SESSION': 'kde4'}, _LinuxDesktopEnvironment.KDE),
|
||||||
|
({'DESKTOP_SESSION': 'kde'}, _LinuxDesktopEnvironment.KDE),
|
||||||
|
({'DESKTOP_SESSION': 'xfce'}, _LinuxDesktopEnvironment.XFCE),
|
||||||
|
|
||||||
|
({'GNOME_DESKTOP_SESSION_ID': 1}, _LinuxDesktopEnvironment.GNOME),
|
||||||
|
({'KDE_FULL_SESSION': 1}, _LinuxDesktopEnvironment.KDE),
|
||||||
|
|
||||||
|
({'XDG_CURRENT_DESKTOP': 'X-Cinnamon'}, _LinuxDesktopEnvironment.CINNAMON),
|
||||||
|
({'XDG_CURRENT_DESKTOP': 'GNOME'}, _LinuxDesktopEnvironment.GNOME),
|
||||||
|
({'XDG_CURRENT_DESKTOP': 'GNOME:GNOME-Classic'}, _LinuxDesktopEnvironment.GNOME),
|
||||||
|
({'XDG_CURRENT_DESKTOP': 'GNOME : GNOME-Classic'}, _LinuxDesktopEnvironment.GNOME),
|
||||||
|
|
||||||
|
({'XDG_CURRENT_DESKTOP': 'Unity', 'DESKTOP_SESSION': 'gnome-fallback'}, _LinuxDesktopEnvironment.GNOME),
|
||||||
|
({'XDG_CURRENT_DESKTOP': 'KDE', 'KDE_SESSION_VERSION': '5'}, _LinuxDesktopEnvironment.KDE),
|
||||||
|
({'XDG_CURRENT_DESKTOP': 'KDE'}, _LinuxDesktopEnvironment.KDE),
|
||||||
|
({'XDG_CURRENT_DESKTOP': 'Pantheon'}, _LinuxDesktopEnvironment.PANTHEON),
|
||||||
|
({'XDG_CURRENT_DESKTOP': 'Unity'}, _LinuxDesktopEnvironment.UNITY),
|
||||||
|
({'XDG_CURRENT_DESKTOP': 'Unity:Unity7'}, _LinuxDesktopEnvironment.UNITY),
|
||||||
|
({'XDG_CURRENT_DESKTOP': 'Unity:Unity8'}, _LinuxDesktopEnvironment.UNITY),
|
||||||
|
]
|
||||||
|
|
||||||
|
for env, expected_desktop_environment in test_cases:
|
||||||
|
self.assertEqual(_get_linux_desktop_environment(env), expected_desktop_environment)
|
||||||
|
|
||||||
def test_chrome_cookie_decryptor_linux_derive_key(self):
|
def test_chrome_cookie_decryptor_linux_derive_key(self):
|
||||||
key = LinuxChromeCookieDecryptor.derive_key(b'abc')
|
key = LinuxChromeCookieDecryptor.derive_key(b'abc')
|
||||||
self.assertEqual(key, b'7\xa1\xec\xd4m\xfcA\xc7\xb19Z\xd0\x19\xdcM\x17')
|
self.assertEqual(key, b'7\xa1\xec\xd4m\xfcA\xc7\xb19Z\xd0\x19\xdcM\x17')
|
||||||
@@ -58,8 +91,7 @@ class TestCookies(unittest.TestCase):
|
|||||||
self.assertEqual(decryptor.decrypt(encrypted_value), value)
|
self.assertEqual(decryptor.decrypt(encrypted_value), value)
|
||||||
|
|
||||||
def test_chrome_cookie_decryptor_linux_v11(self):
|
def test_chrome_cookie_decryptor_linux_v11(self):
|
||||||
with MonkeyPatch(cookies, {'_get_linux_keyring_password': lambda *args, **kwargs: b'',
|
with MonkeyPatch(cookies, {'_get_linux_keyring_password': lambda *args, **kwargs: b''}):
|
||||||
'KEYRING_AVAILABLE': True}):
|
|
||||||
encrypted_value = b'v11#\x81\x10>`w\x8f)\xc0\xb2\xc1\r\xf4\x1al\xdd\x93\xfd\xf8\xf8N\xf2\xa9\x83\xf1\xe9o\x0elVQd'
|
encrypted_value = b'v11#\x81\x10>`w\x8f)\xc0\xb2\xc1\r\xf4\x1al\xdd\x93\xfd\xf8\xf8N\xf2\xa9\x83\xf1\xe9o\x0elVQd'
|
||||||
value = 'tz=Europe.London'
|
value = 'tz=Europe.London'
|
||||||
decryptor = LinuxChromeCookieDecryptor('Chrome', Logger())
|
decryptor = LinuxChromeCookieDecryptor('Chrome', Logger())
|
||||||
|
|||||||
@@ -53,7 +53,7 @@ class YoutubeDL(yt_dlp.YoutubeDL):
|
|||||||
raise ExtractorError(message)
|
raise ExtractorError(message)
|
||||||
|
|
||||||
def process_info(self, info_dict):
|
def process_info(self, info_dict):
|
||||||
self.processed_info_dicts.append(info_dict)
|
self.processed_info_dicts.append(info_dict.copy())
|
||||||
return super(YoutubeDL, self).process_info(info_dict)
|
return super(YoutubeDL, self).process_info(info_dict)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -1,26 +0,0 @@
|
|||||||
# coding: utf-8
|
|
||||||
|
|
||||||
from __future__ import unicode_literals
|
|
||||||
|
|
||||||
# Allow direct execution
|
|
||||||
import os
|
|
||||||
import sys
|
|
||||||
import unittest
|
|
||||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
|
||||||
|
|
||||||
from yt_dlp.options import _hide_login_info
|
|
||||||
|
|
||||||
|
|
||||||
class TestOptions(unittest.TestCase):
|
|
||||||
def test_hide_login_info(self):
|
|
||||||
self.assertEqual(_hide_login_info(['-u', 'foo', '-p', 'bar']),
|
|
||||||
['-u', 'PRIVATE', '-p', 'PRIVATE'])
|
|
||||||
self.assertEqual(_hide_login_info(['-u']), ['-u'])
|
|
||||||
self.assertEqual(_hide_login_info(['-u', 'foo', '-u', 'bar']),
|
|
||||||
['-u', 'PRIVATE', '-u', 'PRIVATE'])
|
|
||||||
self.assertEqual(_hide_login_info(['--username=foo']),
|
|
||||||
['--username=PRIVATE'])
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == '__main__':
|
|
||||||
unittest.main()
|
|
||||||
@@ -13,7 +13,7 @@ from test.helper import FakeYDL, md5, is_download_test
|
|||||||
from yt_dlp.extractor import (
|
from yt_dlp.extractor import (
|
||||||
YoutubeIE,
|
YoutubeIE,
|
||||||
DailymotionIE,
|
DailymotionIE,
|
||||||
TEDIE,
|
TedTalkIE,
|
||||||
VimeoIE,
|
VimeoIE,
|
||||||
WallaIE,
|
WallaIE,
|
||||||
CeskaTelevizeIE,
|
CeskaTelevizeIE,
|
||||||
@@ -141,7 +141,7 @@ class TestDailymotionSubtitles(BaseTestSubtitles):
|
|||||||
@is_download_test
|
@is_download_test
|
||||||
class TestTedSubtitles(BaseTestSubtitles):
|
class TestTedSubtitles(BaseTestSubtitles):
|
||||||
url = 'http://www.ted.com/talks/dan_dennett_on_our_consciousness.html'
|
url = 'http://www.ted.com/talks/dan_dennett_on_our_consciousness.html'
|
||||||
IE = TEDIE
|
IE = TedTalkIE
|
||||||
|
|
||||||
def test_allsubtitles(self):
|
def test_allsubtitles(self):
|
||||||
self.DL.params['writesubtitles'] = True
|
self.DL.params['writesubtitles'] = True
|
||||||
|
|||||||
+117
-15
@@ -23,6 +23,7 @@ from yt_dlp.utils import (
|
|||||||
caesar,
|
caesar,
|
||||||
clean_html,
|
clean_html,
|
||||||
clean_podcast_url,
|
clean_podcast_url,
|
||||||
|
Config,
|
||||||
date_from_str,
|
date_from_str,
|
||||||
datetime_from_str,
|
datetime_from_str,
|
||||||
DateRange,
|
DateRange,
|
||||||
@@ -37,11 +38,18 @@ from yt_dlp.utils import (
|
|||||||
ExtractorError,
|
ExtractorError,
|
||||||
find_xpath_attr,
|
find_xpath_attr,
|
||||||
fix_xml_ampersands,
|
fix_xml_ampersands,
|
||||||
|
format_bytes,
|
||||||
float_or_none,
|
float_or_none,
|
||||||
get_element_by_class,
|
get_element_by_class,
|
||||||
get_element_by_attribute,
|
get_element_by_attribute,
|
||||||
get_elements_by_class,
|
get_elements_by_class,
|
||||||
get_elements_by_attribute,
|
get_elements_by_attribute,
|
||||||
|
get_element_html_by_class,
|
||||||
|
get_element_html_by_attribute,
|
||||||
|
get_elements_html_by_class,
|
||||||
|
get_elements_html_by_attribute,
|
||||||
|
get_elements_text_and_html_by_attribute,
|
||||||
|
get_element_text_and_html_by_tag,
|
||||||
InAdvancePagedList,
|
InAdvancePagedList,
|
||||||
int_or_none,
|
int_or_none,
|
||||||
intlist_to_bytes,
|
intlist_to_bytes,
|
||||||
@@ -116,6 +124,7 @@ from yt_dlp.compat import (
|
|||||||
compat_chr,
|
compat_chr,
|
||||||
compat_etree_fromstring,
|
compat_etree_fromstring,
|
||||||
compat_getenv,
|
compat_getenv,
|
||||||
|
compat_HTMLParseError,
|
||||||
compat_os_name,
|
compat_os_name,
|
||||||
compat_setenv,
|
compat_setenv,
|
||||||
)
|
)
|
||||||
@@ -634,6 +643,8 @@ class TestUtil(unittest.TestCase):
|
|||||||
self.assertEqual(parse_duration('PT1H0.040S'), 3600.04)
|
self.assertEqual(parse_duration('PT1H0.040S'), 3600.04)
|
||||||
self.assertEqual(parse_duration('PT00H03M30SZ'), 210)
|
self.assertEqual(parse_duration('PT00H03M30SZ'), 210)
|
||||||
self.assertEqual(parse_duration('P0Y0M0DT0H4M20.880S'), 260.88)
|
self.assertEqual(parse_duration('P0Y0M0DT0H4M20.880S'), 260.88)
|
||||||
|
self.assertEqual(parse_duration('01:02:03:050'), 3723.05)
|
||||||
|
self.assertEqual(parse_duration('103:050'), 103.05)
|
||||||
|
|
||||||
def test_fix_xml_ampersands(self):
|
def test_fix_xml_ampersands(self):
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
@@ -1573,46 +1584,116 @@ Line 1
|
|||||||
self.assertEqual(urshift(3, 1), 1)
|
self.assertEqual(urshift(3, 1), 1)
|
||||||
self.assertEqual(urshift(-3, 1), 2147483646)
|
self.assertEqual(urshift(-3, 1), 2147483646)
|
||||||
|
|
||||||
|
GET_ELEMENT_BY_CLASS_TEST_STRING = '''
|
||||||
|
<span class="foo bar">nice</span>
|
||||||
|
'''
|
||||||
|
|
||||||
def test_get_element_by_class(self):
|
def test_get_element_by_class(self):
|
||||||
html = '''
|
html = self.GET_ELEMENT_BY_CLASS_TEST_STRING
|
||||||
<span class="foo bar">nice</span>
|
|
||||||
'''
|
|
||||||
|
|
||||||
self.assertEqual(get_element_by_class('foo', html), 'nice')
|
self.assertEqual(get_element_by_class('foo', html), 'nice')
|
||||||
self.assertEqual(get_element_by_class('no-such-class', html), None)
|
self.assertEqual(get_element_by_class('no-such-class', html), None)
|
||||||
|
|
||||||
|
def test_get_element_html_by_class(self):
|
||||||
|
html = self.GET_ELEMENT_BY_CLASS_TEST_STRING
|
||||||
|
|
||||||
|
self.assertEqual(get_element_html_by_class('foo', html), html.strip())
|
||||||
|
self.assertEqual(get_element_by_class('no-such-class', html), None)
|
||||||
|
|
||||||
|
GET_ELEMENT_BY_ATTRIBUTE_TEST_STRING = '''
|
||||||
|
<div itemprop="author" itemscope>foo</div>
|
||||||
|
'''
|
||||||
|
|
||||||
def test_get_element_by_attribute(self):
|
def test_get_element_by_attribute(self):
|
||||||
html = '''
|
html = self.GET_ELEMENT_BY_CLASS_TEST_STRING
|
||||||
<span class="foo bar">nice</span>
|
|
||||||
'''
|
|
||||||
|
|
||||||
self.assertEqual(get_element_by_attribute('class', 'foo bar', html), 'nice')
|
self.assertEqual(get_element_by_attribute('class', 'foo bar', html), 'nice')
|
||||||
self.assertEqual(get_element_by_attribute('class', 'foo', html), None)
|
self.assertEqual(get_element_by_attribute('class', 'foo', html), None)
|
||||||
self.assertEqual(get_element_by_attribute('class', 'no-such-foo', html), None)
|
self.assertEqual(get_element_by_attribute('class', 'no-such-foo', html), None)
|
||||||
|
|
||||||
html = '''
|
html = self.GET_ELEMENT_BY_ATTRIBUTE_TEST_STRING
|
||||||
<div itemprop="author" itemscope>foo</div>
|
|
||||||
'''
|
|
||||||
|
|
||||||
self.assertEqual(get_element_by_attribute('itemprop', 'author', html), 'foo')
|
self.assertEqual(get_element_by_attribute('itemprop', 'author', html), 'foo')
|
||||||
|
|
||||||
|
def test_get_element_html_by_attribute(self):
|
||||||
|
html = self.GET_ELEMENT_BY_CLASS_TEST_STRING
|
||||||
|
|
||||||
|
self.assertEqual(get_element_html_by_attribute('class', 'foo bar', html), html.strip())
|
||||||
|
self.assertEqual(get_element_html_by_attribute('class', 'foo', html), None)
|
||||||
|
self.assertEqual(get_element_html_by_attribute('class', 'no-such-foo', html), None)
|
||||||
|
|
||||||
|
html = self.GET_ELEMENT_BY_ATTRIBUTE_TEST_STRING
|
||||||
|
|
||||||
|
self.assertEqual(get_element_html_by_attribute('itemprop', 'author', html), html.strip())
|
||||||
|
|
||||||
|
GET_ELEMENTS_BY_CLASS_TEST_STRING = '''
|
||||||
|
<span class="foo bar">nice</span><span class="foo bar">also nice</span>
|
||||||
|
'''
|
||||||
|
GET_ELEMENTS_BY_CLASS_RES = ['<span class="foo bar">nice</span>', '<span class="foo bar">also nice</span>']
|
||||||
|
|
||||||
def test_get_elements_by_class(self):
|
def test_get_elements_by_class(self):
|
||||||
html = '''
|
html = self.GET_ELEMENTS_BY_CLASS_TEST_STRING
|
||||||
<span class="foo bar">nice</span><span class="foo bar">also nice</span>
|
|
||||||
'''
|
|
||||||
|
|
||||||
self.assertEqual(get_elements_by_class('foo', html), ['nice', 'also nice'])
|
self.assertEqual(get_elements_by_class('foo', html), ['nice', 'also nice'])
|
||||||
self.assertEqual(get_elements_by_class('no-such-class', html), [])
|
self.assertEqual(get_elements_by_class('no-such-class', html), [])
|
||||||
|
|
||||||
|
def test_get_elements_html_by_class(self):
|
||||||
|
html = self.GET_ELEMENTS_BY_CLASS_TEST_STRING
|
||||||
|
|
||||||
|
self.assertEqual(get_elements_html_by_class('foo', html), self.GET_ELEMENTS_BY_CLASS_RES)
|
||||||
|
self.assertEqual(get_elements_html_by_class('no-such-class', html), [])
|
||||||
|
|
||||||
def test_get_elements_by_attribute(self):
|
def test_get_elements_by_attribute(self):
|
||||||
html = '''
|
html = self.GET_ELEMENTS_BY_CLASS_TEST_STRING
|
||||||
<span class="foo bar">nice</span><span class="foo bar">also nice</span>
|
|
||||||
'''
|
|
||||||
|
|
||||||
self.assertEqual(get_elements_by_attribute('class', 'foo bar', html), ['nice', 'also nice'])
|
self.assertEqual(get_elements_by_attribute('class', 'foo bar', html), ['nice', 'also nice'])
|
||||||
self.assertEqual(get_elements_by_attribute('class', 'foo', html), [])
|
self.assertEqual(get_elements_by_attribute('class', 'foo', html), [])
|
||||||
self.assertEqual(get_elements_by_attribute('class', 'no-such-foo', html), [])
|
self.assertEqual(get_elements_by_attribute('class', 'no-such-foo', html), [])
|
||||||
|
|
||||||
|
def test_get_elements_html_by_attribute(self):
|
||||||
|
html = self.GET_ELEMENTS_BY_CLASS_TEST_STRING
|
||||||
|
|
||||||
|
self.assertEqual(get_elements_html_by_attribute('class', 'foo bar', html), self.GET_ELEMENTS_BY_CLASS_RES)
|
||||||
|
self.assertEqual(get_elements_html_by_attribute('class', 'foo', html), [])
|
||||||
|
self.assertEqual(get_elements_html_by_attribute('class', 'no-such-foo', html), [])
|
||||||
|
|
||||||
|
def test_get_elements_text_and_html_by_attribute(self):
|
||||||
|
html = self.GET_ELEMENTS_BY_CLASS_TEST_STRING
|
||||||
|
|
||||||
|
self.assertEqual(
|
||||||
|
list(get_elements_text_and_html_by_attribute('class', 'foo bar', html)),
|
||||||
|
list(zip(['nice', 'also nice'], self.GET_ELEMENTS_BY_CLASS_RES)))
|
||||||
|
self.assertEqual(list(get_elements_text_and_html_by_attribute('class', 'foo', html)), [])
|
||||||
|
self.assertEqual(list(get_elements_text_and_html_by_attribute('class', 'no-such-foo', html)), [])
|
||||||
|
|
||||||
|
GET_ELEMENT_BY_TAG_TEST_STRING = '''
|
||||||
|
random text lorem ipsum</p>
|
||||||
|
<div>
|
||||||
|
this should be returned
|
||||||
|
<span>this should also be returned</span>
|
||||||
|
<div>
|
||||||
|
this should also be returned
|
||||||
|
</div>
|
||||||
|
closing tag above should not trick, so this should also be returned
|
||||||
|
</div>
|
||||||
|
but this text should not be returned
|
||||||
|
'''
|
||||||
|
GET_ELEMENT_BY_TAG_RES_OUTERDIV_HTML = GET_ELEMENT_BY_TAG_TEST_STRING.strip()[32:276]
|
||||||
|
GET_ELEMENT_BY_TAG_RES_OUTERDIV_TEXT = GET_ELEMENT_BY_TAG_RES_OUTERDIV_HTML[5:-6]
|
||||||
|
GET_ELEMENT_BY_TAG_RES_INNERSPAN_HTML = GET_ELEMENT_BY_TAG_TEST_STRING.strip()[78:119]
|
||||||
|
GET_ELEMENT_BY_TAG_RES_INNERSPAN_TEXT = GET_ELEMENT_BY_TAG_RES_INNERSPAN_HTML[6:-7]
|
||||||
|
|
||||||
|
def test_get_element_text_and_html_by_tag(self):
|
||||||
|
html = self.GET_ELEMENT_BY_TAG_TEST_STRING
|
||||||
|
|
||||||
|
self.assertEqual(
|
||||||
|
get_element_text_and_html_by_tag('div', html),
|
||||||
|
(self.GET_ELEMENT_BY_TAG_RES_OUTERDIV_TEXT, self.GET_ELEMENT_BY_TAG_RES_OUTERDIV_HTML))
|
||||||
|
self.assertEqual(
|
||||||
|
get_element_text_and_html_by_tag('span', html),
|
||||||
|
(self.GET_ELEMENT_BY_TAG_RES_INNERSPAN_TEXT, self.GET_ELEMENT_BY_TAG_RES_INNERSPAN_HTML))
|
||||||
|
self.assertRaises(compat_HTMLParseError, get_element_text_and_html_by_tag, 'article', html)
|
||||||
|
|
||||||
def test_iri_to_uri(self):
|
def test_iri_to_uri(self):
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
iri_to_uri('https://www.google.com/search?q=foo&ie=utf-8&oe=utf-8&client=firefox-b'),
|
iri_to_uri('https://www.google.com/search?q=foo&ie=utf-8&oe=utf-8&client=firefox-b'),
|
||||||
@@ -1688,6 +1769,27 @@ Line 1
|
|||||||
ll = reversed(ll)
|
ll = reversed(ll)
|
||||||
test(ll, -15, 14, range(15))
|
test(ll, -15, 14, range(15))
|
||||||
|
|
||||||
|
def test_format_bytes(self):
|
||||||
|
self.assertEqual(format_bytes(0), '0.00B')
|
||||||
|
self.assertEqual(format_bytes(1000), '1000.00B')
|
||||||
|
self.assertEqual(format_bytes(1024), '1.00KiB')
|
||||||
|
self.assertEqual(format_bytes(1024**2), '1.00MiB')
|
||||||
|
self.assertEqual(format_bytes(1024**3), '1.00GiB')
|
||||||
|
self.assertEqual(format_bytes(1024**4), '1.00TiB')
|
||||||
|
self.assertEqual(format_bytes(1024**5), '1.00PiB')
|
||||||
|
self.assertEqual(format_bytes(1024**6), '1.00EiB')
|
||||||
|
self.assertEqual(format_bytes(1024**7), '1.00ZiB')
|
||||||
|
self.assertEqual(format_bytes(1024**8), '1.00YiB')
|
||||||
|
|
||||||
|
def test_hide_login_info(self):
|
||||||
|
self.assertEqual(Config.hide_login_info(['-u', 'foo', '-p', 'bar']),
|
||||||
|
['-u', 'PRIVATE', '-p', 'PRIVATE'])
|
||||||
|
self.assertEqual(Config.hide_login_info(['-u']), ['-u'])
|
||||||
|
self.assertEqual(Config.hide_login_info(['-u', 'foo', '-u', 'bar']),
|
||||||
|
['-u', 'PRIVATE', '-u', 'PRIVATE'])
|
||||||
|
self.assertEqual(Config.hide_login_info(['--username=foo']),
|
||||||
|
['--username=PRIVATE'])
|
||||||
|
|
||||||
|
|
||||||
if __name__ == '__main__':
|
if __name__ == '__main__':
|
||||||
unittest.main()
|
unittest.main()
|
||||||
|
|||||||
@@ -19,52 +19,52 @@ class TestVerboseOutput(unittest.TestCase):
|
|||||||
[
|
[
|
||||||
sys.executable, 'yt_dlp/__main__.py', '-v',
|
sys.executable, 'yt_dlp/__main__.py', '-v',
|
||||||
'--username', 'johnsmith@gmail.com',
|
'--username', 'johnsmith@gmail.com',
|
||||||
'--password', 'secret',
|
'--password', 'my_secret_password',
|
||||||
], cwd=rootDir, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
|
], cwd=rootDir, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
|
||||||
sout, serr = outp.communicate()
|
sout, serr = outp.communicate()
|
||||||
self.assertTrue(b'--username' in serr)
|
self.assertTrue(b'--username' in serr)
|
||||||
self.assertTrue(b'johnsmith' not in serr)
|
self.assertTrue(b'johnsmith' not in serr)
|
||||||
self.assertTrue(b'--password' in serr)
|
self.assertTrue(b'--password' in serr)
|
||||||
self.assertTrue(b'secret' not in serr)
|
self.assertTrue(b'my_secret_password' not in serr)
|
||||||
|
|
||||||
def test_private_info_shortarg(self):
|
def test_private_info_shortarg(self):
|
||||||
outp = subprocess.Popen(
|
outp = subprocess.Popen(
|
||||||
[
|
[
|
||||||
sys.executable, 'yt_dlp/__main__.py', '-v',
|
sys.executable, 'yt_dlp/__main__.py', '-v',
|
||||||
'-u', 'johnsmith@gmail.com',
|
'-u', 'johnsmith@gmail.com',
|
||||||
'-p', 'secret',
|
'-p', 'my_secret_password',
|
||||||
], cwd=rootDir, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
|
], cwd=rootDir, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
|
||||||
sout, serr = outp.communicate()
|
sout, serr = outp.communicate()
|
||||||
self.assertTrue(b'-u' in serr)
|
self.assertTrue(b'-u' in serr)
|
||||||
self.assertTrue(b'johnsmith' not in serr)
|
self.assertTrue(b'johnsmith' not in serr)
|
||||||
self.assertTrue(b'-p' in serr)
|
self.assertTrue(b'-p' in serr)
|
||||||
self.assertTrue(b'secret' not in serr)
|
self.assertTrue(b'my_secret_password' not in serr)
|
||||||
|
|
||||||
def test_private_info_eq(self):
|
def test_private_info_eq(self):
|
||||||
outp = subprocess.Popen(
|
outp = subprocess.Popen(
|
||||||
[
|
[
|
||||||
sys.executable, 'yt_dlp/__main__.py', '-v',
|
sys.executable, 'yt_dlp/__main__.py', '-v',
|
||||||
'--username=johnsmith@gmail.com',
|
'--username=johnsmith@gmail.com',
|
||||||
'--password=secret',
|
'--password=my_secret_password',
|
||||||
], cwd=rootDir, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
|
], cwd=rootDir, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
|
||||||
sout, serr = outp.communicate()
|
sout, serr = outp.communicate()
|
||||||
self.assertTrue(b'--username' in serr)
|
self.assertTrue(b'--username' in serr)
|
||||||
self.assertTrue(b'johnsmith' not in serr)
|
self.assertTrue(b'johnsmith' not in serr)
|
||||||
self.assertTrue(b'--password' in serr)
|
self.assertTrue(b'--password' in serr)
|
||||||
self.assertTrue(b'secret' not in serr)
|
self.assertTrue(b'my_secret_password' not in serr)
|
||||||
|
|
||||||
def test_private_info_shortarg_eq(self):
|
def test_private_info_shortarg_eq(self):
|
||||||
outp = subprocess.Popen(
|
outp = subprocess.Popen(
|
||||||
[
|
[
|
||||||
sys.executable, 'yt_dlp/__main__.py', '-v',
|
sys.executable, 'yt_dlp/__main__.py', '-v',
|
||||||
'-u=johnsmith@gmail.com',
|
'-u=johnsmith@gmail.com',
|
||||||
'-p=secret',
|
'-p=my_secret_password',
|
||||||
], cwd=rootDir, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
|
], cwd=rootDir, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
|
||||||
sout, serr = outp.communicate()
|
sout, serr = outp.communicate()
|
||||||
self.assertTrue(b'-u' in serr)
|
self.assertTrue(b'-u' in serr)
|
||||||
self.assertTrue(b'johnsmith' not in serr)
|
self.assertTrue(b'johnsmith' not in serr)
|
||||||
self.assertTrue(b'-p' in serr)
|
self.assertTrue(b'-p' in serr)
|
||||||
self.assertTrue(b'secret' not in serr)
|
self.assertTrue(b'my_secret_password' not in serr)
|
||||||
|
|
||||||
|
|
||||||
if __name__ == '__main__':
|
if __name__ == '__main__':
|
||||||
|
|||||||
+308
-235
@@ -91,6 +91,7 @@ from .utils import (
|
|||||||
PerRequestProxyHandler,
|
PerRequestProxyHandler,
|
||||||
platform_name,
|
platform_name,
|
||||||
Popen,
|
Popen,
|
||||||
|
POSTPROCESS_WHEN,
|
||||||
PostProcessingError,
|
PostProcessingError,
|
||||||
preferredencoding,
|
preferredencoding,
|
||||||
prepend_extension,
|
prepend_extension,
|
||||||
@@ -199,7 +200,9 @@ class YoutubeDL(object):
|
|||||||
verbose: Print additional info to stdout.
|
verbose: Print additional info to stdout.
|
||||||
quiet: Do not print messages to stdout.
|
quiet: Do not print messages to stdout.
|
||||||
no_warnings: Do not print out anything for warnings.
|
no_warnings: Do not print out anything for warnings.
|
||||||
forceprint: A list of templates to force print
|
forceprint: A dict with keys video/playlist mapped to
|
||||||
|
a list of templates to force print to stdout
|
||||||
|
For compatibility, a single list is also accepted
|
||||||
forceurl: Force printing final URL. (Deprecated)
|
forceurl: Force printing final URL. (Deprecated)
|
||||||
forcetitle: Force printing title. (Deprecated)
|
forcetitle: Force printing title. (Deprecated)
|
||||||
forceid: Force printing ID. (Deprecated)
|
forceid: Force printing ID. (Deprecated)
|
||||||
@@ -317,10 +320,12 @@ class YoutubeDL(object):
|
|||||||
break_per_url: Whether break_on_reject and break_on_existing
|
break_per_url: Whether break_on_reject and break_on_existing
|
||||||
should act on each input URL as opposed to for the entire queue
|
should act on each input URL as opposed to for the entire queue
|
||||||
cookiefile: File name where cookies should be read from and dumped to
|
cookiefile: File name where cookies should be read from and dumped to
|
||||||
cookiesfrombrowser: A tuple containing the name of the browser and the profile
|
cookiesfrombrowser: A tuple containing the name of the browser, the profile
|
||||||
name/path from where cookies are loaded.
|
name/pathfrom where cookies are loaded, and the name of the
|
||||||
Eg: ('chrome', ) or ('vivaldi', 'default')
|
keyring. Eg: ('chrome', ) or ('vivaldi', 'default', 'BASICTEXT')
|
||||||
nocheckcertificate:Do not verify SSL certificates
|
legacyserverconnect: Explicitly allow HTTPS connection to servers that do not
|
||||||
|
support RFC 5746 secure renegotiation
|
||||||
|
nocheckcertificate: Do not verify SSL certificates
|
||||||
prefer_insecure: Use HTTP instead of HTTPS to retrieve information.
|
prefer_insecure: Use HTTP instead of HTTPS to retrieve information.
|
||||||
At the moment, this is only supported by YouTube.
|
At the moment, this is only supported by YouTube.
|
||||||
proxy: URL of the proxy server to use
|
proxy: URL of the proxy server to use
|
||||||
@@ -505,7 +510,7 @@ class YoutubeDL(object):
|
|||||||
|
|
||||||
params = None
|
params = None
|
||||||
_ies = {}
|
_ies = {}
|
||||||
_pps = {'pre_process': [], 'before_dl': [], 'after_move': [], 'post_process': []}
|
_pps = {k: [] for k in POSTPROCESS_WHEN}
|
||||||
_printed_messages = set()
|
_printed_messages = set()
|
||||||
_first_webpage_request = True
|
_first_webpage_request = True
|
||||||
_download_retcode = None
|
_download_retcode = None
|
||||||
@@ -523,7 +528,7 @@ class YoutubeDL(object):
|
|||||||
params = {}
|
params = {}
|
||||||
self._ies = {}
|
self._ies = {}
|
||||||
self._ies_instances = {}
|
self._ies_instances = {}
|
||||||
self._pps = {'pre_process': [], 'before_dl': [], 'after_move': [], 'post_process': []}
|
self._pps = {k: [] for k in POSTPROCESS_WHEN}
|
||||||
self._printed_messages = set()
|
self._printed_messages = set()
|
||||||
self._first_webpage_request = True
|
self._first_webpage_request = True
|
||||||
self._post_hooks = []
|
self._post_hooks = []
|
||||||
@@ -531,6 +536,7 @@ class YoutubeDL(object):
|
|||||||
self._postprocessor_hooks = []
|
self._postprocessor_hooks = []
|
||||||
self._download_retcode = 0
|
self._download_retcode = 0
|
||||||
self._num_downloads = 0
|
self._num_downloads = 0
|
||||||
|
self._num_videos = 0
|
||||||
self._screen_file = [sys.stdout, sys.stderr][params.get('logtostderr', False)]
|
self._screen_file = [sys.stdout, sys.stderr][params.get('logtostderr', False)]
|
||||||
self._err_file = sys.stderr
|
self._err_file = sys.stderr
|
||||||
self.params = params
|
self.params = params
|
||||||
@@ -585,6 +591,11 @@ class YoutubeDL(object):
|
|||||||
else:
|
else:
|
||||||
self.params['nooverwrites'] = not self.params['overwrites']
|
self.params['nooverwrites'] = not self.params['overwrites']
|
||||||
|
|
||||||
|
# Compatibility with older syntax
|
||||||
|
params.setdefault('forceprint', {})
|
||||||
|
if not isinstance(params['forceprint'], dict):
|
||||||
|
params['forceprint'] = {'video': params['forceprint']}
|
||||||
|
|
||||||
if params.get('bidi_workaround', False):
|
if params.get('bidi_workaround', False):
|
||||||
try:
|
try:
|
||||||
import pty
|
import pty
|
||||||
@@ -1036,6 +1047,7 @@ class YoutubeDL(object):
|
|||||||
if info_dict.get('duration', None) is not None
|
if info_dict.get('duration', None) is not None
|
||||||
else None)
|
else None)
|
||||||
info_dict['autonumber'] = self.params.get('autonumber_start', 1) - 1 + self._num_downloads
|
info_dict['autonumber'] = self.params.get('autonumber_start', 1) - 1 + self._num_downloads
|
||||||
|
info_dict['video_autonumber'] = self._num_videos
|
||||||
if info_dict.get('resolution') is None:
|
if info_dict.get('resolution') is None:
|
||||||
info_dict['resolution'] = self.format_resolution(info_dict, default=None)
|
info_dict['resolution'] = self.format_resolution(info_dict, default=None)
|
||||||
|
|
||||||
@@ -1151,7 +1163,7 @@ class YoutubeDL(object):
|
|||||||
str_fmt = f'{fmt[:-1]}s'
|
str_fmt = f'{fmt[:-1]}s'
|
||||||
if fmt[-1] == 'l': # list
|
if fmt[-1] == 'l': # list
|
||||||
delim = '\n' if '#' in flags else ', '
|
delim = '\n' if '#' in flags else ', '
|
||||||
value, fmt = delim.join(variadic(value, allowed_types=(str, bytes))), str_fmt
|
value, fmt = delim.join(map(str, variadic(value, allowed_types=(str, bytes)))), str_fmt
|
||||||
elif fmt[-1] == 'j': # json
|
elif fmt[-1] == 'j': # json
|
||||||
value, fmt = json.dumps(value, default=_dumpjson_default, indent=4 if '#' in flags else None), str_fmt
|
value, fmt = json.dumps(value, default=_dumpjson_default, indent=4 if '#' in flags else None), str_fmt
|
||||||
elif fmt[-1] == 'q': # quoted
|
elif fmt[-1] == 'q': # quoted
|
||||||
@@ -1166,7 +1178,9 @@ class YoutubeDL(object):
|
|||||||
'NF%s%s' % ('K' if '+' in flags else '', 'D' if '#' in flags else 'C'),
|
'NF%s%s' % ('K' if '+' in flags else '', 'D' if '#' in flags else 'C'),
|
||||||
value), str_fmt
|
value), str_fmt
|
||||||
elif fmt[-1] == 'D': # decimal suffix
|
elif fmt[-1] == 'D': # decimal suffix
|
||||||
value, fmt = format_decimal_suffix(value, f'%{fmt[:-1]}f%s' if fmt[:-1] else '%d%s'), 's'
|
num_fmt, fmt = fmt[:-1].replace('#', ''), 's'
|
||||||
|
value = format_decimal_suffix(value, f'%{num_fmt}f%s' if num_fmt else '%d%s',
|
||||||
|
factor=1024 if '#' in flags else 1000)
|
||||||
elif fmt[-1] == 'S': # filename sanitization
|
elif fmt[-1] == 'S': # filename sanitization
|
||||||
value, fmt = filename_sanitizer(initial_field, value, restricted='#' in flags), str_fmt
|
value, fmt = filename_sanitizer(initial_field, value, restricted='#' in flags), str_fmt
|
||||||
elif fmt[-1] == 'c':
|
elif fmt[-1] == 'c':
|
||||||
@@ -1348,31 +1362,33 @@ class YoutubeDL(object):
|
|||||||
def __handle_extraction_exceptions(func):
|
def __handle_extraction_exceptions(func):
|
||||||
@functools.wraps(func)
|
@functools.wraps(func)
|
||||||
def wrapper(self, *args, **kwargs):
|
def wrapper(self, *args, **kwargs):
|
||||||
try:
|
while True:
|
||||||
return func(self, *args, **kwargs)
|
try:
|
||||||
except GeoRestrictedError as e:
|
return func(self, *args, **kwargs)
|
||||||
msg = e.msg
|
except (DownloadCancelled, LazyList.IndexError, PagedList.IndexError):
|
||||||
if e.countries:
|
|
||||||
msg += '\nThis video is available in %s.' % ', '.join(
|
|
||||||
map(ISO3166Utils.short2full, e.countries))
|
|
||||||
msg += '\nYou might want to use a VPN or a proxy server (with --proxy) to workaround.'
|
|
||||||
self.report_error(msg)
|
|
||||||
except ExtractorError as e: # An error we somewhat expected
|
|
||||||
self.report_error(compat_str(e), e.format_traceback())
|
|
||||||
except ReExtractInfo as e:
|
|
||||||
if e.expected:
|
|
||||||
self.to_screen(f'{e}; Re-extracting data')
|
|
||||||
else:
|
|
||||||
self.to_stderr('\r')
|
|
||||||
self.report_warning(f'{e}; Re-extracting data')
|
|
||||||
return wrapper(self, *args, **kwargs)
|
|
||||||
except (DownloadCancelled, LazyList.IndexError, PagedList.IndexError):
|
|
||||||
raise
|
|
||||||
except Exception as e:
|
|
||||||
if self.params.get('ignoreerrors'):
|
|
||||||
self.report_error(error_to_compat_str(e), tb=encode_compat_str(traceback.format_exc()))
|
|
||||||
else:
|
|
||||||
raise
|
raise
|
||||||
|
except ReExtractInfo as e:
|
||||||
|
if e.expected:
|
||||||
|
self.to_screen(f'{e}; Re-extracting data')
|
||||||
|
else:
|
||||||
|
self.to_stderr('\r')
|
||||||
|
self.report_warning(f'{e}; Re-extracting data')
|
||||||
|
continue
|
||||||
|
except GeoRestrictedError as e:
|
||||||
|
msg = e.msg
|
||||||
|
if e.countries:
|
||||||
|
msg += '\nThis video is available in %s.' % ', '.join(
|
||||||
|
map(ISO3166Utils.short2full, e.countries))
|
||||||
|
msg += '\nYou might want to use a VPN or a proxy server (with --proxy) to workaround.'
|
||||||
|
self.report_error(msg)
|
||||||
|
except ExtractorError as e: # An error we somewhat expected
|
||||||
|
self.report_error(str(e), e.format_traceback())
|
||||||
|
except Exception as e:
|
||||||
|
if self.params.get('ignoreerrors'):
|
||||||
|
self.report_error(str(e), tb=encode_compat_str(traceback.format_exc()))
|
||||||
|
else:
|
||||||
|
raise
|
||||||
|
break
|
||||||
return wrapper
|
return wrapper
|
||||||
|
|
||||||
def _wait_for_video(self, ie_result):
|
def _wait_for_video(self, ie_result):
|
||||||
@@ -1582,6 +1598,19 @@ class YoutubeDL(object):
|
|||||||
def _ensure_dir_exists(self, path):
|
def _ensure_dir_exists(self, path):
|
||||||
return make_dir(path, self.report_error)
|
return make_dir(path, self.report_error)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _playlist_infodict(ie_result, **kwargs):
|
||||||
|
return {
|
||||||
|
**ie_result,
|
||||||
|
'playlist': ie_result.get('title') or ie_result.get('id'),
|
||||||
|
'playlist_id': ie_result.get('id'),
|
||||||
|
'playlist_title': ie_result.get('title'),
|
||||||
|
'playlist_uploader': ie_result.get('uploader'),
|
||||||
|
'playlist_uploader_id': ie_result.get('uploader_id'),
|
||||||
|
'playlist_index': 0,
|
||||||
|
**kwargs,
|
||||||
|
}
|
||||||
|
|
||||||
def __process_playlist(self, ie_result, download):
|
def __process_playlist(self, ie_result, download):
|
||||||
# We process each entry in the playlist
|
# We process each entry in the playlist
|
||||||
playlist = ie_result.get('title') or ie_result.get('id')
|
playlist = ie_result.get('title') or ie_result.get('id')
|
||||||
@@ -1622,14 +1651,15 @@ class YoutubeDL(object):
|
|||||||
playlistitems = orderedSet(iter_playlistitems(playlistitems_str))
|
playlistitems = orderedSet(iter_playlistitems(playlistitems_str))
|
||||||
|
|
||||||
ie_entries = ie_result['entries']
|
ie_entries = ie_result['entries']
|
||||||
msg = (
|
|
||||||
'Downloading %d videos' if not isinstance(ie_entries, list)
|
|
||||||
else 'Collected %d videos; downloading %%d of them' % len(ie_entries))
|
|
||||||
|
|
||||||
if isinstance(ie_entries, list):
|
if isinstance(ie_entries, list):
|
||||||
|
playlist_count = len(ie_entries)
|
||||||
|
msg = f'Collected {playlist_count} videos; downloading %d of them'
|
||||||
|
ie_result['playlist_count'] = ie_result.get('playlist_count') or playlist_count
|
||||||
|
|
||||||
def get_entry(i):
|
def get_entry(i):
|
||||||
return ie_entries[i - 1]
|
return ie_entries[i - 1]
|
||||||
else:
|
else:
|
||||||
|
msg = 'Downloading %d videos'
|
||||||
if not isinstance(ie_entries, (PagedList, LazyList)):
|
if not isinstance(ie_entries, (PagedList, LazyList)):
|
||||||
ie_entries = LazyList(ie_entries)
|
ie_entries = LazyList(ie_entries)
|
||||||
|
|
||||||
@@ -1638,7 +1668,7 @@ class YoutubeDL(object):
|
|||||||
lambda self, i: ie_entries[i - 1]
|
lambda self, i: ie_entries[i - 1]
|
||||||
)(self, i)
|
)(self, i)
|
||||||
|
|
||||||
entries = []
|
entries, broken = [], False
|
||||||
items = playlistitems if playlistitems is not None else itertools.count(playliststart)
|
items = playlistitems if playlistitems is not None else itertools.count(playliststart)
|
||||||
for i in items:
|
for i in items:
|
||||||
if i == 0:
|
if i == 0:
|
||||||
@@ -1660,6 +1690,7 @@ class YoutubeDL(object):
|
|||||||
if entry is not None:
|
if entry is not None:
|
||||||
self._match_entry(entry, incomplete=True, silent=True)
|
self._match_entry(entry, incomplete=True, silent=True)
|
||||||
except (ExistingVideoReached, RejectedVideoReached):
|
except (ExistingVideoReached, RejectedVideoReached):
|
||||||
|
broken = True
|
||||||
break
|
break
|
||||||
ie_result['entries'] = entries
|
ie_result['entries'] = entries
|
||||||
|
|
||||||
@@ -1670,23 +1701,19 @@ class YoutubeDL(object):
|
|||||||
if entry is not None]
|
if entry is not None]
|
||||||
n_entries = len(entries)
|
n_entries = len(entries)
|
||||||
|
|
||||||
|
if not (ie_result.get('playlist_count') or broken or playlistitems or playlistend):
|
||||||
|
ie_result['playlist_count'] = n_entries
|
||||||
|
|
||||||
if not playlistitems and (playliststart != 1 or playlistend):
|
if not playlistitems and (playliststart != 1 or playlistend):
|
||||||
playlistitems = list(range(playliststart, playliststart + n_entries))
|
playlistitems = list(range(playliststart, playliststart + n_entries))
|
||||||
ie_result['requested_entries'] = playlistitems
|
ie_result['requested_entries'] = playlistitems
|
||||||
|
|
||||||
_infojson_written = False
|
_infojson_written = False
|
||||||
if not self.params.get('simulate') and self.params.get('allow_playlist_files', True):
|
write_playlist_files = self.params.get('allow_playlist_files', True)
|
||||||
ie_copy = {
|
if write_playlist_files and self.params.get('list_thumbnails'):
|
||||||
'playlist': playlist,
|
self.list_thumbnails(ie_result)
|
||||||
'playlist_id': ie_result.get('id'),
|
if write_playlist_files and not self.params.get('simulate'):
|
||||||
'playlist_title': ie_result.get('title'),
|
ie_copy = self._playlist_infodict(ie_result, n_entries=n_entries)
|
||||||
'playlist_uploader': ie_result.get('uploader'),
|
|
||||||
'playlist_uploader_id': ie_result.get('uploader_id'),
|
|
||||||
'playlist_index': 0,
|
|
||||||
'n_entries': n_entries,
|
|
||||||
}
|
|
||||||
ie_copy.update(dict(ie_result))
|
|
||||||
|
|
||||||
_infojson_written = self._write_info_json(
|
_infojson_written = self._write_info_json(
|
||||||
'playlist', ie_result, self.prepare_filename(ie_copy, 'pl_infojson'))
|
'playlist', ie_result, self.prepare_filename(ie_copy, 'pl_infojson'))
|
||||||
if _infojson_written is None:
|
if _infojson_written is None:
|
||||||
@@ -1719,6 +1746,7 @@ class YoutubeDL(object):
|
|||||||
extra = {
|
extra = {
|
||||||
'n_entries': n_entries,
|
'n_entries': n_entries,
|
||||||
'_last_playlist_index': max(playlistitems) if playlistitems else (playlistend or n_entries),
|
'_last_playlist_index': max(playlistitems) if playlistitems else (playlistend or n_entries),
|
||||||
|
'playlist_count': ie_result.get('playlist_count'),
|
||||||
'playlist_index': playlist_index,
|
'playlist_index': playlist_index,
|
||||||
'playlist_autonumber': i,
|
'playlist_autonumber': i,
|
||||||
'playlist': playlist,
|
'playlist': playlist,
|
||||||
@@ -1751,7 +1779,9 @@ class YoutubeDL(object):
|
|||||||
'updated playlist', ie_result,
|
'updated playlist', ie_result,
|
||||||
self.prepare_filename(ie_copy, 'pl_infojson'), overwrite=True) is None:
|
self.prepare_filename(ie_copy, 'pl_infojson'), overwrite=True) is None:
|
||||||
return
|
return
|
||||||
self.to_screen('[download] Finished downloading playlist: %s' % playlist)
|
|
||||||
|
ie_result = self.run_all_pps('playlist', ie_result)
|
||||||
|
self.to_screen(f'[download] Finished downloading playlist: {playlist}')
|
||||||
return ie_result
|
return ie_result
|
||||||
|
|
||||||
@__handle_extraction_exceptions
|
@__handle_extraction_exceptions
|
||||||
@@ -2256,6 +2286,7 @@ class YoutubeDL(object):
|
|||||||
|
|
||||||
def process_video_result(self, info_dict, download=True):
|
def process_video_result(self, info_dict, download=True):
|
||||||
assert info_dict.get('_type', 'video') == 'video'
|
assert info_dict.get('_type', 'video') == 'video'
|
||||||
|
self._num_videos += 1
|
||||||
|
|
||||||
if 'id' not in info_dict:
|
if 'id' not in info_dict:
|
||||||
raise ExtractorError('Missing "id" field in extractor result')
|
raise ExtractorError('Missing "id" field in extractor result')
|
||||||
@@ -2309,6 +2340,7 @@ class YoutubeDL(object):
|
|||||||
for ts_key, date_key in (
|
for ts_key, date_key in (
|
||||||
('timestamp', 'upload_date'),
|
('timestamp', 'upload_date'),
|
||||||
('release_timestamp', 'release_date'),
|
('release_timestamp', 'release_date'),
|
||||||
|
('modified_timestamp', 'modified_date'),
|
||||||
):
|
):
|
||||||
if info_dict.get(date_key) is None and info_dict.get(ts_key) is not None:
|
if info_dict.get(date_key) is None and info_dict.get(ts_key) is not None:
|
||||||
# Working around out-of-range timestamp values (e.g. negative ones on Windows,
|
# Working around out-of-range timestamp values (e.g. negative ones on Windows,
|
||||||
@@ -2368,9 +2400,14 @@ class YoutubeDL(object):
|
|||||||
if not self.params.get('allow_unplayable_formats'):
|
if not self.params.get('allow_unplayable_formats'):
|
||||||
formats = [f for f in formats if not f.get('has_drm')]
|
formats = [f for f in formats if not f.get('has_drm')]
|
||||||
|
|
||||||
|
# backward compatibility
|
||||||
|
info_dict['fulltitle'] = info_dict['title']
|
||||||
|
|
||||||
if info_dict.get('is_live'):
|
if info_dict.get('is_live'):
|
||||||
get_from_start = bool(self.params.get('live_from_start'))
|
get_from_start = bool(self.params.get('live_from_start'))
|
||||||
formats = [f for f in formats if bool(f.get('is_from_start')) == get_from_start]
|
formats = [f for f in formats if bool(f.get('is_from_start')) == get_from_start]
|
||||||
|
if not get_from_start:
|
||||||
|
info_dict['title'] += ' ' + datetime.datetime.now().strftime('%Y-%m-%d %H:%M')
|
||||||
|
|
||||||
if not formats:
|
if not formats:
|
||||||
self.raise_no_formats(info_dict)
|
self.raise_no_formats(info_dict)
|
||||||
@@ -2532,24 +2569,46 @@ class YoutubeDL(object):
|
|||||||
if not self.params.get('ignore_no_formats_error'):
|
if not self.params.get('ignore_no_formats_error'):
|
||||||
raise ExtractorError('Requested format is not available', expected=True,
|
raise ExtractorError('Requested format is not available', expected=True,
|
||||||
video_id=info_dict['id'], ie=info_dict['extractor'])
|
video_id=info_dict['id'], ie=info_dict['extractor'])
|
||||||
else:
|
self.report_warning('Requested format is not available')
|
||||||
self.report_warning('Requested format is not available')
|
# Process what we can, even without any available formats.
|
||||||
# Process what we can, even without any available formats.
|
formats_to_download = [{}]
|
||||||
self.process_info(dict(info_dict))
|
|
||||||
elif download:
|
best_format = formats_to_download[-1]
|
||||||
self.to_screen(
|
if download:
|
||||||
'[info] %s: Downloading %d format(s): %s' % (
|
if best_format:
|
||||||
info_dict['id'], len(formats_to_download),
|
self.to_screen(
|
||||||
", ".join([f['format_id'] for f in formats_to_download])))
|
f'[info] {info_dict["id"]}: Downloading {len(formats_to_download)} format(s): '
|
||||||
for fmt in formats_to_download:
|
+ ', '.join([f['format_id'] for f in formats_to_download]))
|
||||||
new_info = dict(info_dict)
|
max_downloads_reached = False
|
||||||
|
for i, fmt in enumerate(formats_to_download):
|
||||||
|
formats_to_download[i] = new_info = dict(info_dict)
|
||||||
# Save a reference to the original info_dict so that it can be modified in process_info if needed
|
# Save a reference to the original info_dict so that it can be modified in process_info if needed
|
||||||
new_info['__original_infodict'] = info_dict
|
|
||||||
new_info.update(fmt)
|
new_info.update(fmt)
|
||||||
self.process_info(new_info)
|
new_info['__original_infodict'] = info_dict
|
||||||
|
try:
|
||||||
|
self.process_info(new_info)
|
||||||
|
except MaxDownloadsReached:
|
||||||
|
max_downloads_reached = True
|
||||||
|
new_info.pop('__original_infodict')
|
||||||
|
# Remove copied info
|
||||||
|
for key, val in tuple(new_info.items()):
|
||||||
|
if info_dict.get(key) == val:
|
||||||
|
new_info.pop(key)
|
||||||
|
if max_downloads_reached:
|
||||||
|
break
|
||||||
|
|
||||||
|
write_archive = set(f.get('__write_download_archive', False) for f in formats_to_download)
|
||||||
|
assert write_archive.issubset({True, False, 'ignore'})
|
||||||
|
if True in write_archive and False not in write_archive:
|
||||||
|
self.record_download_archive(info_dict)
|
||||||
|
|
||||||
|
info_dict['requested_downloads'] = formats_to_download
|
||||||
|
info_dict = self.run_all_pps('after_video', info_dict)
|
||||||
|
if max_downloads_reached:
|
||||||
|
raise MaxDownloadsReached()
|
||||||
|
|
||||||
# We update the info dict with the selected best quality format (backwards compatibility)
|
# We update the info dict with the selected best quality format (backwards compatibility)
|
||||||
if formats_to_download:
|
info_dict.update(best_format)
|
||||||
info_dict.update(formats_to_download[-1])
|
|
||||||
return info_dict
|
return info_dict
|
||||||
|
|
||||||
def process_subtitles(self, video_id, normal_subtitles, automatic_captions):
|
def process_subtitles(self, video_id, normal_subtitles, automatic_captions):
|
||||||
@@ -2620,6 +2679,20 @@ class YoutubeDL(object):
|
|||||||
subs[lang] = f
|
subs[lang] = f
|
||||||
return subs
|
return subs
|
||||||
|
|
||||||
|
def _forceprint(self, tmpl, info_dict):
|
||||||
|
mobj = re.match(r'\w+(=?)$', tmpl)
|
||||||
|
if mobj and mobj.group(1):
|
||||||
|
tmpl = f'{tmpl[:-1]} = %({tmpl[:-1]})r'
|
||||||
|
elif mobj:
|
||||||
|
tmpl = '%({})s'.format(tmpl)
|
||||||
|
|
||||||
|
info_dict = info_dict.copy()
|
||||||
|
info_dict['formats_table'] = self.render_formats_table(info_dict)
|
||||||
|
info_dict['thumbnails_table'] = self.render_thumbnails_table(info_dict)
|
||||||
|
info_dict['subtitles_table'] = self.render_subtitles_table(info_dict.get('id'), info_dict.get('subtitles'))
|
||||||
|
info_dict['automatic_captions_table'] = self.render_subtitles_table(info_dict.get('id'), info_dict.get('automatic_captions'))
|
||||||
|
self.to_stdout(self.evaluate_outtmpl(tmpl, info_dict))
|
||||||
|
|
||||||
def __forced_printings(self, info_dict, filename, incomplete):
|
def __forced_printings(self, info_dict, filename, incomplete):
|
||||||
def print_mandatory(field, actual_field=None):
|
def print_mandatory(field, actual_field=None):
|
||||||
if actual_field is None:
|
if actual_field is None:
|
||||||
@@ -2642,15 +2715,10 @@ class YoutubeDL(object):
|
|||||||
elif 'url' in info_dict:
|
elif 'url' in info_dict:
|
||||||
info_dict['urls'] = info_dict['url'] + info_dict.get('play_path', '')
|
info_dict['urls'] = info_dict['url'] + info_dict.get('play_path', '')
|
||||||
|
|
||||||
if self.params.get('forceprint') or self.params.get('forcejson'):
|
if self.params['forceprint'].get('video') or self.params.get('forcejson'):
|
||||||
self.post_extract(info_dict)
|
self.post_extract(info_dict)
|
||||||
for tmpl in self.params.get('forceprint', []):
|
for tmpl in self.params['forceprint'].get('video', []):
|
||||||
mobj = re.match(r'\w+(=?)$', tmpl)
|
self._forceprint(tmpl, info_dict)
|
||||||
if mobj and mobj.group(1):
|
|
||||||
tmpl = f'{tmpl[:-1]} = %({tmpl[:-1]})s'
|
|
||||||
elif mobj:
|
|
||||||
tmpl = '%({})s'.format(tmpl)
|
|
||||||
self.to_stdout(self.evaluate_outtmpl(tmpl, info_dict))
|
|
||||||
|
|
||||||
print_mandatory('title')
|
print_mandatory('title')
|
||||||
print_mandatory('id')
|
print_mandatory('id')
|
||||||
@@ -2688,7 +2756,9 @@ class YoutubeDL(object):
|
|||||||
if not test:
|
if not test:
|
||||||
for ph in self._progress_hooks:
|
for ph in self._progress_hooks:
|
||||||
fd.add_progress_hook(ph)
|
fd.add_progress_hook(ph)
|
||||||
urls = '", "'.join([f['url'] for f in info.get('requested_formats', [])] or [info['url']])
|
urls = '", "'.join(
|
||||||
|
(f['url'].split(',')[0] + ',<data>' if f['url'].startswith('data:') else f['url'])
|
||||||
|
for f in info.get('requested_formats', []) or [info])
|
||||||
self.write_debug('Invoking downloader on "%s"' % urls)
|
self.write_debug('Invoking downloader on "%s"' % urls)
|
||||||
|
|
||||||
# Note: Ideally info should be a deep-copied so that hooks cannot modify it.
|
# Note: Ideally info should be a deep-copied so that hooks cannot modify it.
|
||||||
@@ -2698,26 +2768,27 @@ class YoutubeDL(object):
|
|||||||
new_info['http_headers'] = self._calc_headers(new_info)
|
new_info['http_headers'] = self._calc_headers(new_info)
|
||||||
return fd.download(name, new_info, subtitle)
|
return fd.download(name, new_info, subtitle)
|
||||||
|
|
||||||
|
def existing_file(self, filepaths, *, default_overwrite=True):
|
||||||
|
existing_files = list(filter(os.path.exists, orderedSet(filepaths)))
|
||||||
|
if existing_files and not self.params.get('overwrites', default_overwrite):
|
||||||
|
return existing_files[0]
|
||||||
|
|
||||||
|
for file in existing_files:
|
||||||
|
self.report_file_delete(file)
|
||||||
|
os.remove(file)
|
||||||
|
return None
|
||||||
|
|
||||||
def process_info(self, info_dict):
|
def process_info(self, info_dict):
|
||||||
"""Process a single resolved IE result."""
|
"""Process a single resolved IE result. (Modified it in-place)"""
|
||||||
|
|
||||||
assert info_dict.get('_type', 'video') == 'video'
|
assert info_dict.get('_type', 'video') == 'video'
|
||||||
|
original_infodict = info_dict
|
||||||
max_downloads = self.params.get('max_downloads')
|
|
||||||
if max_downloads is not None:
|
|
||||||
if self._num_downloads >= int(max_downloads):
|
|
||||||
raise MaxDownloadsReached()
|
|
||||||
|
|
||||||
if info_dict.get('is_live') and not self.params.get('live_from_start'):
|
|
||||||
info_dict['title'] += ' ' + datetime.datetime.now().strftime('%Y-%m-%d %H:%M')
|
|
||||||
|
|
||||||
# TODO: backward compatibility, to be removed
|
|
||||||
info_dict['fulltitle'] = info_dict['title']
|
|
||||||
|
|
||||||
if 'format' not in info_dict and 'ext' in info_dict:
|
if 'format' not in info_dict and 'ext' in info_dict:
|
||||||
info_dict['format'] = info_dict['ext']
|
info_dict['format'] = info_dict['ext']
|
||||||
|
|
||||||
if self._match_entry(info_dict) is not None:
|
if self._match_entry(info_dict) is not None:
|
||||||
|
info_dict['__write_download_archive'] = 'ignore'
|
||||||
return
|
return
|
||||||
|
|
||||||
self.post_extract(info_dict)
|
self.post_extract(info_dict)
|
||||||
@@ -2732,9 +2803,7 @@ class YoutubeDL(object):
|
|||||||
self.__forced_printings(info_dict, full_filename, incomplete=('format' not in info_dict))
|
self.__forced_printings(info_dict, full_filename, incomplete=('format' not in info_dict))
|
||||||
|
|
||||||
if self.params.get('simulate'):
|
if self.params.get('simulate'):
|
||||||
if self.params.get('force_write_download_archive', False):
|
info_dict['__write_download_archive'] = self.params.get('force_write_download_archive')
|
||||||
self.record_download_archive(info_dict)
|
|
||||||
# Do nothing else if in simulate mode
|
|
||||||
return
|
return
|
||||||
|
|
||||||
if full_filename is None:
|
if full_filename is None:
|
||||||
@@ -2829,43 +2898,39 @@ class YoutubeDL(object):
|
|||||||
for link_type, should_write in write_links.items()):
|
for link_type, should_write in write_links.items()):
|
||||||
return
|
return
|
||||||
|
|
||||||
|
def replace_info_dict(new_info):
|
||||||
|
nonlocal info_dict
|
||||||
|
if new_info == info_dict:
|
||||||
|
return
|
||||||
|
info_dict.clear()
|
||||||
|
info_dict.update(new_info)
|
||||||
|
|
||||||
try:
|
try:
|
||||||
info_dict, files_to_move = self.pre_process(info_dict, 'before_dl', files_to_move)
|
new_info, files_to_move = self.pre_process(info_dict, 'before_dl', files_to_move)
|
||||||
|
replace_info_dict(new_info)
|
||||||
except PostProcessingError as err:
|
except PostProcessingError as err:
|
||||||
self.report_error('Preprocessing: %s' % str(err))
|
self.report_error('Preprocessing: %s' % str(err))
|
||||||
return
|
return
|
||||||
|
|
||||||
must_record_download_archive = False
|
if self.params.get('skip_download'):
|
||||||
if self.params.get('skip_download', False):
|
|
||||||
info_dict['filepath'] = temp_filename
|
info_dict['filepath'] = temp_filename
|
||||||
info_dict['__finaldir'] = os.path.dirname(os.path.abspath(encodeFilename(full_filename)))
|
info_dict['__finaldir'] = os.path.dirname(os.path.abspath(encodeFilename(full_filename)))
|
||||||
info_dict['__files_to_move'] = files_to_move
|
info_dict['__files_to_move'] = files_to_move
|
||||||
info_dict = self.run_pp(MoveFilesAfterDownloadPP(self, False), info_dict)
|
replace_info_dict(self.run_pp(MoveFilesAfterDownloadPP(self, False), info_dict))
|
||||||
|
info_dict['__write_download_archive'] = self.params.get('force_write_download_archive')
|
||||||
else:
|
else:
|
||||||
# Download
|
# Download
|
||||||
info_dict.setdefault('__postprocessors', [])
|
info_dict.setdefault('__postprocessors', [])
|
||||||
try:
|
try:
|
||||||
|
|
||||||
def existing_file(*filepaths):
|
def existing_video_file(*filepaths):
|
||||||
ext = info_dict.get('ext')
|
ext = info_dict.get('ext')
|
||||||
final_ext = self.params.get('final_ext', ext)
|
converted = lambda file: replace_extension(file, self.params.get('final_ext') or ext, ext)
|
||||||
existing_files = []
|
file = self.existing_file(itertools.chain(*zip(map(converted, filepaths), filepaths)),
|
||||||
for file in orderedSet(filepaths):
|
default_overwrite=False)
|
||||||
if final_ext != ext:
|
if file:
|
||||||
converted = replace_extension(file, final_ext, ext)
|
info_dict['ext'] = os.path.splitext(file)[1][1:]
|
||||||
if os.path.exists(encodeFilename(converted)):
|
return file
|
||||||
existing_files.append(converted)
|
|
||||||
if os.path.exists(encodeFilename(file)):
|
|
||||||
existing_files.append(file)
|
|
||||||
|
|
||||||
if not existing_files or self.params.get('overwrites', False):
|
|
||||||
for file in orderedSet(existing_files):
|
|
||||||
self.report_file_delete(file)
|
|
||||||
os.remove(encodeFilename(file))
|
|
||||||
return None
|
|
||||||
|
|
||||||
info_dict['ext'] = os.path.splitext(existing_files[0])[1][1:]
|
|
||||||
return existing_files[0]
|
|
||||||
|
|
||||||
success = True
|
success = True
|
||||||
if info_dict.get('requested_formats') is not None:
|
if info_dict.get('requested_formats') is not None:
|
||||||
@@ -2919,7 +2984,7 @@ class YoutubeDL(object):
|
|||||||
# Ensure filename always has a correct extension for successful merge
|
# Ensure filename always has a correct extension for successful merge
|
||||||
full_filename = correct_ext(full_filename)
|
full_filename = correct_ext(full_filename)
|
||||||
temp_filename = correct_ext(temp_filename)
|
temp_filename = correct_ext(temp_filename)
|
||||||
dl_filename = existing_file(full_filename, temp_filename)
|
dl_filename = existing_video_file(full_filename, temp_filename)
|
||||||
info_dict['__real_download'] = False
|
info_dict['__real_download'] = False
|
||||||
|
|
||||||
downloaded = []
|
downloaded = []
|
||||||
@@ -2982,7 +3047,7 @@ class YoutubeDL(object):
|
|||||||
files_to_move[file] = None
|
files_to_move[file] = None
|
||||||
else:
|
else:
|
||||||
# Just a single file
|
# Just a single file
|
||||||
dl_filename = existing_file(full_filename, temp_filename)
|
dl_filename = existing_video_file(full_filename, temp_filename)
|
||||||
if dl_filename is None or dl_filename == temp_filename:
|
if dl_filename is None or dl_filename == temp_filename:
|
||||||
# dl_filename == temp_filename could mean that the file was partially downloaded with --no-part.
|
# dl_filename == temp_filename could mean that the file was partially downloaded with --no-part.
|
||||||
# So we should try to resume the download
|
# So we should try to resume the download
|
||||||
@@ -3059,7 +3124,7 @@ class YoutubeDL(object):
|
|||||||
|
|
||||||
fixup()
|
fixup()
|
||||||
try:
|
try:
|
||||||
info_dict = self.post_process(dl_filename, info_dict, files_to_move)
|
replace_info_dict(self.post_process(dl_filename, info_dict, files_to_move))
|
||||||
except PostProcessingError as err:
|
except PostProcessingError as err:
|
||||||
self.report_error('Postprocessing: %s' % str(err))
|
self.report_error('Postprocessing: %s' % str(err))
|
||||||
return
|
return
|
||||||
@@ -3069,10 +3134,14 @@ class YoutubeDL(object):
|
|||||||
except Exception as err:
|
except Exception as err:
|
||||||
self.report_error('post hooks: %s' % str(err))
|
self.report_error('post hooks: %s' % str(err))
|
||||||
return
|
return
|
||||||
must_record_download_archive = True
|
info_dict['__write_download_archive'] = True
|
||||||
|
|
||||||
|
if self.params.get('force_write_download_archive'):
|
||||||
|
info_dict['__write_download_archive'] = True
|
||||||
|
|
||||||
|
# Make sure the info_dict was modified in-place
|
||||||
|
assert info_dict is original_infodict
|
||||||
|
|
||||||
if must_record_download_archive or self.params.get('force_write_download_archive', False):
|
|
||||||
self.record_download_archive(info_dict)
|
|
||||||
max_downloads = self.params.get('max_downloads')
|
max_downloads = self.params.get('max_downloads')
|
||||||
if max_downloads is not None and self._num_downloads >= int(max_downloads):
|
if max_downloads is not None and self._num_downloads >= int(max_downloads):
|
||||||
raise MaxDownloadsReached()
|
raise MaxDownloadsReached()
|
||||||
@@ -3142,12 +3211,11 @@ class YoutubeDL(object):
|
|||||||
keep_keys = ['_type'] # Always keep this to facilitate load-info-json
|
keep_keys = ['_type'] # Always keep this to facilitate load-info-json
|
||||||
if remove_private_keys:
|
if remove_private_keys:
|
||||||
remove_keys |= {
|
remove_keys |= {
|
||||||
'requested_formats', 'requested_subtitles', 'requested_entries', 'entries',
|
'requested_downloads', 'requested_formats', 'requested_subtitles', 'requested_entries',
|
||||||
'filepath', 'infojson_filename', 'original_url', 'playlist_autonumber',
|
'entries', 'filepath', 'infojson_filename', 'original_url', 'playlist_autonumber',
|
||||||
}
|
}
|
||||||
empty_values = (None, {}, [], set(), tuple())
|
|
||||||
reject = lambda k, v: k not in keep_keys and (
|
reject = lambda k, v: k not in keep_keys and (
|
||||||
k.startswith('_') or k in remove_keys or v in empty_values)
|
k.startswith('_') or k in remove_keys or v is None)
|
||||||
else:
|
else:
|
||||||
reject = lambda k, v: k in remove_keys
|
reject = lambda k, v: k in remove_keys
|
||||||
|
|
||||||
@@ -3168,6 +3236,25 @@ class YoutubeDL(object):
|
|||||||
''' Alias of sanitize_info for backward compatibility '''
|
''' Alias of sanitize_info for backward compatibility '''
|
||||||
return YoutubeDL.sanitize_info(info_dict, actually_filter)
|
return YoutubeDL.sanitize_info(info_dict, actually_filter)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def post_extract(info_dict):
|
||||||
|
def actual_post_extract(info_dict):
|
||||||
|
if info_dict.get('_type') in ('playlist', 'multi_video'):
|
||||||
|
for video_dict in info_dict.get('entries', {}):
|
||||||
|
actual_post_extract(video_dict or {})
|
||||||
|
return
|
||||||
|
|
||||||
|
post_extractor = info_dict.get('__post_extractor') or (lambda: {})
|
||||||
|
extra = post_extractor().items()
|
||||||
|
info_dict.update(extra)
|
||||||
|
info_dict.pop('__post_extractor', None)
|
||||||
|
|
||||||
|
original_infodict = info_dict.get('__original_infodict') or {}
|
||||||
|
original_infodict.update(extra)
|
||||||
|
original_infodict.pop('__post_extractor', None)
|
||||||
|
|
||||||
|
actual_post_extract(info_dict or {})
|
||||||
|
|
||||||
def run_pp(self, pp, infodict):
|
def run_pp(self, pp, infodict):
|
||||||
files_to_delete = []
|
files_to_delete = []
|
||||||
if '__files_to_move' not in infodict:
|
if '__files_to_move' not in infodict:
|
||||||
@@ -3197,45 +3284,27 @@ class YoutubeDL(object):
|
|||||||
del infodict['__files_to_move'][old_filename]
|
del infodict['__files_to_move'][old_filename]
|
||||||
return infodict
|
return infodict
|
||||||
|
|
||||||
@staticmethod
|
def run_all_pps(self, key, info, *, additional_pps=None):
|
||||||
def post_extract(info_dict):
|
for tmpl in self.params['forceprint'].get(key, []):
|
||||||
def actual_post_extract(info_dict):
|
self._forceprint(tmpl, info)
|
||||||
if info_dict.get('_type') in ('playlist', 'multi_video'):
|
for pp in (additional_pps or []) + self._pps[key]:
|
||||||
for video_dict in info_dict.get('entries', {}):
|
info = self.run_pp(pp, info)
|
||||||
actual_post_extract(video_dict or {})
|
return info
|
||||||
return
|
|
||||||
|
|
||||||
post_extractor = info_dict.get('__post_extractor') or (lambda: {})
|
|
||||||
extra = post_extractor().items()
|
|
||||||
info_dict.update(extra)
|
|
||||||
info_dict.pop('__post_extractor', None)
|
|
||||||
|
|
||||||
original_infodict = info_dict.get('__original_infodict') or {}
|
|
||||||
original_infodict.update(extra)
|
|
||||||
original_infodict.pop('__post_extractor', None)
|
|
||||||
|
|
||||||
actual_post_extract(info_dict or {})
|
|
||||||
|
|
||||||
def pre_process(self, ie_info, key='pre_process', files_to_move=None):
|
def pre_process(self, ie_info, key='pre_process', files_to_move=None):
|
||||||
info = dict(ie_info)
|
info = dict(ie_info)
|
||||||
info['__files_to_move'] = files_to_move or {}
|
info['__files_to_move'] = files_to_move or {}
|
||||||
for pp in self._pps[key]:
|
info = self.run_all_pps(key, info)
|
||||||
info = self.run_pp(pp, info)
|
|
||||||
return info, info.pop('__files_to_move', None)
|
return info, info.pop('__files_to_move', None)
|
||||||
|
|
||||||
def post_process(self, filename, ie_info, files_to_move=None):
|
def post_process(self, filename, info, files_to_move=None):
|
||||||
"""Run all the postprocessors on the given file."""
|
"""Run all the postprocessors on the given file."""
|
||||||
info = dict(ie_info)
|
|
||||||
info['filepath'] = filename
|
info['filepath'] = filename
|
||||||
info['__files_to_move'] = files_to_move or {}
|
info['__files_to_move'] = files_to_move or {}
|
||||||
|
info = self.run_all_pps('post_process', info, additional_pps=info.get('__postprocessors'))
|
||||||
for pp in ie_info.get('__postprocessors', []) + self._pps['post_process']:
|
|
||||||
info = self.run_pp(pp, info)
|
|
||||||
info = self.run_pp(MoveFilesAfterDownloadPP(self), info)
|
info = self.run_pp(MoveFilesAfterDownloadPP(self), info)
|
||||||
del info['__files_to_move']
|
del info['__files_to_move']
|
||||||
for pp in self._pps['after_move']:
|
return self.run_all_pps('after_move', info)
|
||||||
info = self.run_pp(pp, info)
|
|
||||||
return info
|
|
||||||
|
|
||||||
def _make_archive_id(self, info_dict):
|
def _make_archive_id(self, info_dict):
|
||||||
video_id = info_dict.get('id')
|
video_id = info_dict.get('id')
|
||||||
@@ -3274,6 +3343,7 @@ class YoutubeDL(object):
|
|||||||
return
|
return
|
||||||
vid_id = self._make_archive_id(info_dict)
|
vid_id = self._make_archive_id(info_dict)
|
||||||
assert vid_id
|
assert vid_id
|
||||||
|
self.write_debug(f'Adding to archive: {vid_id}')
|
||||||
with locked_file(fn, 'a', encoding='utf-8') as archive_file:
|
with locked_file(fn, 'a', encoding='utf-8') as archive_file:
|
||||||
archive_file.write(vid_id + '\n')
|
archive_file.write(vid_id + '\n')
|
||||||
self.archive.add(vid_id)
|
self.archive.add(vid_id)
|
||||||
@@ -3292,6 +3362,11 @@ class YoutubeDL(object):
|
|||||||
return '%dx?' % format['width']
|
return '%dx?' % format['width']
|
||||||
return default
|
return default
|
||||||
|
|
||||||
|
def _list_format_headers(self, *headers):
|
||||||
|
if self.params.get('listformats_table', True) is not False:
|
||||||
|
return [self._format_screen(header, self.Styles.HEADERS) for header in headers]
|
||||||
|
return headers
|
||||||
|
|
||||||
def _format_note(self, fdict):
|
def _format_note(self, fdict):
|
||||||
res = ''
|
res = ''
|
||||||
if fdict.get('ext') in ['f4f', 'f4m']:
|
if fdict.get('ext') in ['f4f', 'f4m']:
|
||||||
@@ -3352,102 +3427,97 @@ class YoutubeDL(object):
|
|||||||
res += '~' + format_bytes(fdict['filesize_approx'])
|
res += '~' + format_bytes(fdict['filesize_approx'])
|
||||||
return res
|
return res
|
||||||
|
|
||||||
def _list_format_headers(self, *headers):
|
def render_formats_table(self, info_dict):
|
||||||
if self.params.get('listformats_table', True) is not False:
|
|
||||||
return [self._format_screen(header, self.Styles.HEADERS) for header in headers]
|
|
||||||
return headers
|
|
||||||
|
|
||||||
def list_formats(self, info_dict):
|
|
||||||
if not info_dict.get('formats') and not info_dict.get('url'):
|
if not info_dict.get('formats') and not info_dict.get('url'):
|
||||||
self.to_screen('%s has no formats' % info_dict['id'])
|
return None
|
||||||
return
|
|
||||||
self.to_screen('[info] Available formats for %s:' % info_dict['id'])
|
|
||||||
|
|
||||||
formats = info_dict.get('formats', [info_dict])
|
formats = info_dict.get('formats', [info_dict])
|
||||||
new_format = self.params.get('listformats_table', True) is not False
|
if not self.params.get('listformats_table', True) is not False:
|
||||||
if new_format:
|
|
||||||
delim = self._format_screen('\u2502', self.Styles.DELIM, '|', test_encoding=True)
|
|
||||||
table = [
|
|
||||||
[
|
|
||||||
self._format_screen(format_field(f, 'format_id'), self.Styles.ID),
|
|
||||||
format_field(f, 'ext'),
|
|
||||||
format_field(f, func=self.format_resolution, ignore=('audio only', 'images')),
|
|
||||||
format_field(f, 'fps', '\t%d'),
|
|
||||||
format_field(f, 'dynamic_range', '%s', ignore=(None, 'SDR')).replace('HDR', ''),
|
|
||||||
delim,
|
|
||||||
format_field(f, 'filesize', ' \t%s', func=format_bytes) + format_field(f, 'filesize_approx', '~\t%s', func=format_bytes),
|
|
||||||
format_field(f, 'tbr', '\t%dk'),
|
|
||||||
shorten_protocol_name(f.get('protocol', '')),
|
|
||||||
delim,
|
|
||||||
format_field(f, 'vcodec', default='unknown').replace(
|
|
||||||
'none',
|
|
||||||
'images' if f.get('acodec') == 'none'
|
|
||||||
else self._format_screen('audio only', self.Styles.SUPPRESS)),
|
|
||||||
format_field(f, 'vbr', '\t%dk'),
|
|
||||||
format_field(f, 'acodec', default='unknown').replace(
|
|
||||||
'none',
|
|
||||||
'' if f.get('vcodec') == 'none'
|
|
||||||
else self._format_screen('video only', self.Styles.SUPPRESS)),
|
|
||||||
format_field(f, 'abr', '\t%dk'),
|
|
||||||
format_field(f, 'asr', '\t%dHz'),
|
|
||||||
join_nonempty(
|
|
||||||
self._format_screen('UNSUPPORTED', 'light red') if f.get('ext') in ('f4f', 'f4m') else None,
|
|
||||||
format_field(f, 'language', '[%s]'),
|
|
||||||
join_nonempty(
|
|
||||||
format_field(f, 'format_note'),
|
|
||||||
format_field(f, 'container', ignore=(None, f.get('ext'))),
|
|
||||||
delim=', '),
|
|
||||||
delim=' '),
|
|
||||||
] for f in formats if f.get('preference') is None or f['preference'] >= -1000]
|
|
||||||
header_line = self._list_format_headers(
|
|
||||||
'ID', 'EXT', 'RESOLUTION', '\tFPS', 'HDR', delim, '\tFILESIZE', '\tTBR', 'PROTO',
|
|
||||||
delim, 'VCODEC', '\tVBR', 'ACODEC', '\tABR', '\tASR', 'MORE INFO')
|
|
||||||
else:
|
|
||||||
table = [
|
table = [
|
||||||
[
|
[
|
||||||
format_field(f, 'format_id'),
|
format_field(f, 'format_id'),
|
||||||
format_field(f, 'ext'),
|
format_field(f, 'ext'),
|
||||||
self.format_resolution(f),
|
self.format_resolution(f),
|
||||||
self._format_note(f)]
|
self._format_note(f)
|
||||||
for f in formats
|
] for f in formats if f.get('preference') is None or f['preference'] >= -1000]
|
||||||
if f.get('preference') is None or f['preference'] >= -1000]
|
return render_table(['format code', 'extension', 'resolution', 'note'], table, extra_gap=1)
|
||||||
header_line = ['format code', 'extension', 'resolution', 'note']
|
|
||||||
|
|
||||||
self.to_stdout(render_table(
|
delim = self._format_screen('\u2502', self.Styles.DELIM, '|', test_encoding=True)
|
||||||
header_line, table,
|
table = [
|
||||||
extra_gap=(0 if new_format else 1),
|
[
|
||||||
hide_empty=new_format,
|
self._format_screen(format_field(f, 'format_id'), self.Styles.ID),
|
||||||
delim=new_format and self._format_screen('\u2500', self.Styles.DELIM, '-', test_encoding=True)))
|
format_field(f, 'ext'),
|
||||||
|
format_field(f, func=self.format_resolution, ignore=('audio only', 'images')),
|
||||||
|
format_field(f, 'fps', '\t%d'),
|
||||||
|
format_field(f, 'dynamic_range', '%s', ignore=(None, 'SDR')).replace('HDR', ''),
|
||||||
|
delim,
|
||||||
|
format_field(f, 'filesize', ' \t%s', func=format_bytes) + format_field(f, 'filesize_approx', '~\t%s', func=format_bytes),
|
||||||
|
format_field(f, 'tbr', '\t%dk'),
|
||||||
|
shorten_protocol_name(f.get('protocol', '')),
|
||||||
|
delim,
|
||||||
|
format_field(f, 'vcodec', default='unknown').replace(
|
||||||
|
'none', 'images' if f.get('acodec') == 'none'
|
||||||
|
else self._format_screen('audio only', self.Styles.SUPPRESS)),
|
||||||
|
format_field(f, 'vbr', '\t%dk'),
|
||||||
|
format_field(f, 'acodec', default='unknown').replace(
|
||||||
|
'none', '' if f.get('vcodec') == 'none'
|
||||||
|
else self._format_screen('video only', self.Styles.SUPPRESS)),
|
||||||
|
format_field(f, 'abr', '\t%dk'),
|
||||||
|
format_field(f, 'asr', '\t%dHz'),
|
||||||
|
join_nonempty(
|
||||||
|
self._format_screen('UNSUPPORTED', 'light red') if f.get('ext') in ('f4f', 'f4m') else None,
|
||||||
|
format_field(f, 'language', '[%s]'),
|
||||||
|
join_nonempty(format_field(f, 'format_note'),
|
||||||
|
format_field(f, 'container', ignore=(None, f.get('ext'))),
|
||||||
|
delim=', '),
|
||||||
|
delim=' '),
|
||||||
|
] for f in formats if f.get('preference') is None or f['preference'] >= -1000]
|
||||||
|
header_line = self._list_format_headers(
|
||||||
|
'ID', 'EXT', 'RESOLUTION', '\tFPS', 'HDR', delim, '\tFILESIZE', '\tTBR', 'PROTO',
|
||||||
|
delim, 'VCODEC', '\tVBR', 'ACODEC', '\tABR', '\tASR', 'MORE INFO')
|
||||||
|
|
||||||
def list_thumbnails(self, info_dict):
|
return render_table(
|
||||||
|
header_line, table, hide_empty=True,
|
||||||
|
delim=self._format_screen('\u2500', self.Styles.DELIM, '-', test_encoding=True))
|
||||||
|
|
||||||
|
def render_thumbnails_table(self, info_dict):
|
||||||
thumbnails = list(info_dict.get('thumbnails'))
|
thumbnails = list(info_dict.get('thumbnails'))
|
||||||
if not thumbnails:
|
if not thumbnails:
|
||||||
self.to_screen('[info] No thumbnails present for %s' % info_dict['id'])
|
return None
|
||||||
return
|
return render_table(
|
||||||
|
|
||||||
self.to_screen(
|
|
||||||
'[info] Thumbnails for %s:' % info_dict['id'])
|
|
||||||
self.to_stdout(render_table(
|
|
||||||
self._list_format_headers('ID', 'Width', 'Height', 'URL'),
|
self._list_format_headers('ID', 'Width', 'Height', 'URL'),
|
||||||
[[t['id'], t.get('width', 'unknown'), t.get('height', 'unknown'), t['url']] for t in thumbnails]))
|
[[t.get('id'), t.get('width', 'unknown'), t.get('height', 'unknown'), t['url']] for t in thumbnails])
|
||||||
|
|
||||||
def list_subtitles(self, video_id, subtitles, name='subtitles'):
|
|
||||||
if not subtitles:
|
|
||||||
self.to_screen('%s has no %s' % (video_id, name))
|
|
||||||
return
|
|
||||||
self.to_screen(
|
|
||||||
'Available %s for %s:' % (name, video_id))
|
|
||||||
|
|
||||||
|
def render_subtitles_table(self, video_id, subtitles):
|
||||||
def _row(lang, formats):
|
def _row(lang, formats):
|
||||||
exts, names = zip(*((f['ext'], f.get('name') or 'unknown') for f in reversed(formats)))
|
exts, names = zip(*((f['ext'], f.get('name') or 'unknown') for f in reversed(formats)))
|
||||||
if len(set(names)) == 1:
|
if len(set(names)) == 1:
|
||||||
names = [] if names[0] == 'unknown' else names[:1]
|
names = [] if names[0] == 'unknown' else names[:1]
|
||||||
return [lang, ', '.join(names), ', '.join(exts)]
|
return [lang, ', '.join(names), ', '.join(exts)]
|
||||||
|
|
||||||
self.to_stdout(render_table(
|
if not subtitles:
|
||||||
|
return None
|
||||||
|
return render_table(
|
||||||
self._list_format_headers('Language', 'Name', 'Formats'),
|
self._list_format_headers('Language', 'Name', 'Formats'),
|
||||||
[_row(lang, formats) for lang, formats in subtitles.items()],
|
[_row(lang, formats) for lang, formats in subtitles.items()],
|
||||||
hide_empty=True))
|
hide_empty=True)
|
||||||
|
|
||||||
|
def __list_table(self, video_id, name, func, *args):
|
||||||
|
table = func(*args)
|
||||||
|
if not table:
|
||||||
|
self.to_screen(f'{video_id} has no {name}')
|
||||||
|
return
|
||||||
|
self.to_screen(f'[info] Available {name} for {video_id}:')
|
||||||
|
self.to_stdout(table)
|
||||||
|
|
||||||
|
def list_formats(self, info_dict):
|
||||||
|
self.__list_table(info_dict['id'], 'formats', self.render_formats_table, info_dict)
|
||||||
|
|
||||||
|
def list_thumbnails(self, info_dict):
|
||||||
|
self.__list_table(info_dict['id'], 'thumbnails', self.render_thumbnails_table, info_dict)
|
||||||
|
|
||||||
|
def list_subtitles(self, video_id, subtitles, name='subtitles'):
|
||||||
|
self.__list_table(video_id, name, self.render_subtitles_table, video_id, subtitles)
|
||||||
|
|
||||||
def urlopen(self, req):
|
def urlopen(self, req):
|
||||||
""" Start an HTTP download """
|
""" Start an HTTP download """
|
||||||
@@ -3540,11 +3610,11 @@ class YoutubeDL(object):
|
|||||||
|
|
||||||
from .downloader.websocket import has_websockets
|
from .downloader.websocket import has_websockets
|
||||||
from .postprocessor.embedthumbnail import has_mutagen
|
from .postprocessor.embedthumbnail import has_mutagen
|
||||||
from .cookies import SQLITE_AVAILABLE, KEYRING_AVAILABLE
|
from .cookies import SQLITE_AVAILABLE, SECRETSTORAGE_AVAILABLE
|
||||||
|
|
||||||
lib_str = join_nonempty(
|
lib_str = join_nonempty(
|
||||||
compat_pycrypto_AES and compat_pycrypto_AES.__name__.split('.')[0],
|
compat_pycrypto_AES and compat_pycrypto_AES.__name__.split('.')[0],
|
||||||
KEYRING_AVAILABLE and 'keyring',
|
SECRETSTORAGE_AVAILABLE and 'secretstorage',
|
||||||
has_mutagen and 'mutagen',
|
has_mutagen and 'mutagen',
|
||||||
SQLITE_AVAILABLE and 'sqlite',
|
SQLITE_AVAILABLE and 'sqlite',
|
||||||
has_websockets and 'websockets',
|
has_websockets and 'websockets',
|
||||||
@@ -3696,10 +3766,11 @@ class YoutubeDL(object):
|
|||||||
sub_format = sub_info['ext']
|
sub_format = sub_info['ext']
|
||||||
sub_filename = subtitles_filename(filename, sub_lang, sub_format, info_dict.get('ext'))
|
sub_filename = subtitles_filename(filename, sub_lang, sub_format, info_dict.get('ext'))
|
||||||
sub_filename_final = subtitles_filename(sub_filename_base, sub_lang, sub_format, info_dict.get('ext'))
|
sub_filename_final = subtitles_filename(sub_filename_base, sub_lang, sub_format, info_dict.get('ext'))
|
||||||
if not self.params.get('overwrites', True) and os.path.exists(sub_filename):
|
existing_sub = self.existing_file((sub_filename_final, sub_filename))
|
||||||
|
if existing_sub:
|
||||||
self.to_screen(f'[info] Video subtitle {sub_lang}.{sub_format} is already present')
|
self.to_screen(f'[info] Video subtitle {sub_lang}.{sub_format} is already present')
|
||||||
sub_info['filepath'] = sub_filename
|
sub_info['filepath'] = existing_sub
|
||||||
ret.append((sub_filename, sub_filename_final))
|
ret.append((existing_sub, sub_filename_final))
|
||||||
continue
|
continue
|
||||||
|
|
||||||
self.to_screen(f'[info] Writing video subtitles to: {sub_filename}')
|
self.to_screen(f'[info] Writing video subtitles to: {sub_filename}')
|
||||||
@@ -3722,9 +3793,10 @@ class YoutubeDL(object):
|
|||||||
self.dl(sub_filename, sub_copy, subtitle=True)
|
self.dl(sub_filename, sub_copy, subtitle=True)
|
||||||
sub_info['filepath'] = sub_filename
|
sub_info['filepath'] = sub_filename
|
||||||
ret.append((sub_filename, sub_filename_final))
|
ret.append((sub_filename, sub_filename_final))
|
||||||
except (ExtractorError, IOError, OSError, ValueError) + network_exceptions as err:
|
except (DownloadError, ExtractorError, IOError, OSError, ValueError) + network_exceptions as err:
|
||||||
|
if self.params.get('ignoreerrors') is not True: # False or 'only_download'
|
||||||
|
raise DownloadError(f'Unable to download video subtitles for {sub_lang!r}: {err}', err)
|
||||||
self.report_warning(f'Unable to download video subtitles for {sub_lang!r}: {err}')
|
self.report_warning(f'Unable to download video subtitles for {sub_lang!r}: {err}')
|
||||||
continue
|
|
||||||
return ret
|
return ret
|
||||||
|
|
||||||
def _write_thumbnails(self, label, info_dict, filename, thumb_filename_base=None):
|
def _write_thumbnails(self, label, info_dict, filename, thumb_filename_base=None):
|
||||||
@@ -3747,11 +3819,12 @@ class YoutubeDL(object):
|
|||||||
thumb_filename = replace_extension(filename, thumb_ext, info_dict.get('ext'))
|
thumb_filename = replace_extension(filename, thumb_ext, info_dict.get('ext'))
|
||||||
thumb_filename_final = replace_extension(thumb_filename_base, thumb_ext, info_dict.get('ext'))
|
thumb_filename_final = replace_extension(thumb_filename_base, thumb_ext, info_dict.get('ext'))
|
||||||
|
|
||||||
if not self.params.get('overwrites', True) and os.path.exists(thumb_filename):
|
existing_thumb = self.existing_file((thumb_filename_final, thumb_filename))
|
||||||
ret.append((thumb_filename, thumb_filename_final))
|
if existing_thumb:
|
||||||
t['filepath'] = thumb_filename
|
|
||||||
self.to_screen('[info] %s is already present' % (
|
self.to_screen('[info] %s is already present' % (
|
||||||
thumb_display_id if multiple else f'{label} thumbnail').capitalize())
|
thumb_display_id if multiple else f'{label} thumbnail').capitalize())
|
||||||
|
t['filepath'] = existing_thumb
|
||||||
|
ret.append((existing_thumb, thumb_filename_final))
|
||||||
else:
|
else:
|
||||||
self.to_screen(f'[info] Downloading {thumb_display_id} ...')
|
self.to_screen(f'[info] Downloading {thumb_display_id} ...')
|
||||||
try:
|
try:
|
||||||
|
|||||||
+37
-20
@@ -22,7 +22,7 @@ from .compat import (
|
|||||||
compat_shlex_quote,
|
compat_shlex_quote,
|
||||||
workaround_optparse_bug9161,
|
workaround_optparse_bug9161,
|
||||||
)
|
)
|
||||||
from .cookies import SUPPORTED_BROWSERS
|
from .cookies import SUPPORTED_BROWSERS, SUPPORTED_KEYRINGS
|
||||||
from .utils import (
|
from .utils import (
|
||||||
DateRange,
|
DateRange,
|
||||||
decodeOption,
|
decodeOption,
|
||||||
@@ -143,6 +143,8 @@ def _real_main(argv=None):
|
|||||||
'"-f best" selects the best pre-merged format which is often not the best option',
|
'"-f best" selects the best pre-merged format which is often not the best option',
|
||||||
'To let yt-dlp download and merge the best available formats, simply do not pass any format selection',
|
'To let yt-dlp download and merge the best available formats, simply do not pass any format selection',
|
||||||
'If you know what you are doing and want only the best pre-merged format, use "-f b" instead to suppress this warning')))
|
'If you know what you are doing and want only the best pre-merged format, use "-f b" instead to suppress this warning')))
|
||||||
|
if opts.exec_cmd.get('before_dl') and opts.exec_before_dl_cmd:
|
||||||
|
parser.error('using "--exec-before-download" conflicts with "--exec before_dl:"')
|
||||||
if opts.usenetrc and (opts.username is not None or opts.password is not None):
|
if opts.usenetrc and (opts.username is not None or opts.password is not None):
|
||||||
parser.error('using .netrc conflicts with giving username/password')
|
parser.error('using .netrc conflicts with giving username/password')
|
||||||
if opts.password is not None and opts.username is None:
|
if opts.password is not None and opts.username is None:
|
||||||
@@ -266,10 +268,20 @@ def _real_main(argv=None):
|
|||||||
if opts.convertthumbnails not in FFmpegThumbnailsConvertorPP.SUPPORTED_EXTS:
|
if opts.convertthumbnails not in FFmpegThumbnailsConvertorPP.SUPPORTED_EXTS:
|
||||||
parser.error('invalid thumbnail format specified')
|
parser.error('invalid thumbnail format specified')
|
||||||
if opts.cookiesfrombrowser is not None:
|
if opts.cookiesfrombrowser is not None:
|
||||||
opts.cookiesfrombrowser = [
|
mobj = re.match(r'(?P<name>[^+:]+)(\s*\+\s*(?P<keyring>[^:]+))?(\s*:(?P<profile>.+))?', opts.cookiesfrombrowser)
|
||||||
part.strip() or None for part in opts.cookiesfrombrowser.split(':', 1)]
|
if mobj is None:
|
||||||
if opts.cookiesfrombrowser[0].lower() not in SUPPORTED_BROWSERS:
|
parser.error(f'invalid cookies from browser arguments: {opts.cookiesfrombrowser}')
|
||||||
parser.error('unsupported browser specified for cookies')
|
browser_name, keyring, profile = mobj.group('name', 'keyring', 'profile')
|
||||||
|
browser_name = browser_name.lower()
|
||||||
|
if browser_name not in SUPPORTED_BROWSERS:
|
||||||
|
parser.error(f'unsupported browser specified for cookies: "{browser_name}". '
|
||||||
|
f'Supported browsers are: {", ".join(sorted(SUPPORTED_BROWSERS))}')
|
||||||
|
if keyring is not None:
|
||||||
|
keyring = keyring.upper()
|
||||||
|
if keyring not in SUPPORTED_KEYRINGS:
|
||||||
|
parser.error(f'unsupported keyring specified for cookies: "{keyring}". '
|
||||||
|
f'Supported keyrings are: {", ".join(sorted(SUPPORTED_KEYRINGS))}')
|
||||||
|
opts.cookiesfrombrowser = (browser_name, profile, keyring)
|
||||||
geo_bypass_code = opts.geo_bypass_ip_block or opts.geo_bypass_country
|
geo_bypass_code = opts.geo_bypass_ip_block or opts.geo_bypass_country
|
||||||
if geo_bypass_code is not None:
|
if geo_bypass_code is not None:
|
||||||
try:
|
try:
|
||||||
@@ -341,9 +353,9 @@ def _real_main(argv=None):
|
|||||||
|
|
||||||
for k, tmpl in opts.outtmpl.items():
|
for k, tmpl in opts.outtmpl.items():
|
||||||
validate_outtmpl(tmpl, f'{k} output template')
|
validate_outtmpl(tmpl, f'{k} output template')
|
||||||
opts.forceprint = opts.forceprint or []
|
for type_, tmpl_list in opts.forceprint.items():
|
||||||
for tmpl in opts.forceprint or []:
|
for tmpl in tmpl_list:
|
||||||
validate_outtmpl(tmpl, 'print template')
|
validate_outtmpl(tmpl, f'{type_} print template')
|
||||||
validate_outtmpl(opts.sponsorblock_chapter_title, 'SponsorBlock chapter title')
|
validate_outtmpl(opts.sponsorblock_chapter_title, 'SponsorBlock chapter title')
|
||||||
for k, tmpl in opts.progress_template.items():
|
for k, tmpl in opts.progress_template.items():
|
||||||
k = f'{k[:-6]} console title' if '-title' in k else f'{k} progress'
|
k = f'{k[:-6]} console title' if '-title' in k else f'{k} progress'
|
||||||
@@ -385,7 +397,10 @@ def _real_main(argv=None):
|
|||||||
opts.parse_metadata.append('title:%s' % opts.metafromtitle)
|
opts.parse_metadata.append('title:%s' % opts.metafromtitle)
|
||||||
opts.parse_metadata = list(itertools.chain(*map(metadataparser_actions, opts.parse_metadata)))
|
opts.parse_metadata = list(itertools.chain(*map(metadataparser_actions, opts.parse_metadata)))
|
||||||
|
|
||||||
any_getting = opts.forceprint or opts.geturl or opts.gettitle or opts.getid or opts.getthumbnail or opts.getdescription or opts.getfilename or opts.getformat or opts.getduration or opts.dumpjson or opts.dump_single_json
|
any_getting = (any(opts.forceprint.values()) or opts.dumpjson or opts.dump_single_json
|
||||||
|
or opts.geturl or opts.gettitle or opts.getid or opts.getthumbnail
|
||||||
|
or opts.getdescription or opts.getfilename or opts.getformat or opts.getduration)
|
||||||
|
|
||||||
any_printing = opts.print_json
|
any_printing = opts.print_json
|
||||||
download_archive_fn = expand_path(opts.download_archive) if opts.download_archive is not None else opts.download_archive
|
download_archive_fn = expand_path(opts.download_archive) if opts.download_archive is not None else opts.download_archive
|
||||||
|
|
||||||
@@ -476,13 +491,6 @@ def _real_main(argv=None):
|
|||||||
# Run this before the actual video download
|
# Run this before the actual video download
|
||||||
'when': 'before_dl'
|
'when': 'before_dl'
|
||||||
})
|
})
|
||||||
# Must be after all other before_dl
|
|
||||||
if opts.exec_before_dl_cmd:
|
|
||||||
postprocessors.append({
|
|
||||||
'key': 'Exec',
|
|
||||||
'exec_cmd': opts.exec_before_dl_cmd,
|
|
||||||
'when': 'before_dl'
|
|
||||||
})
|
|
||||||
if opts.extractaudio:
|
if opts.extractaudio:
|
||||||
postprocessors.append({
|
postprocessors.append({
|
||||||
'key': 'FFmpegExtractAudio',
|
'key': 'FFmpegExtractAudio',
|
||||||
@@ -583,13 +591,21 @@ def _real_main(argv=None):
|
|||||||
# XAttrMetadataPP should be run after post-processors that may change file contents
|
# XAttrMetadataPP should be run after post-processors that may change file contents
|
||||||
if opts.xattrs:
|
if opts.xattrs:
|
||||||
postprocessors.append({'key': 'XAttrMetadata'})
|
postprocessors.append({'key': 'XAttrMetadata'})
|
||||||
# Exec must be the last PP
|
if opts.concat_playlist != 'never':
|
||||||
if opts.exec_cmd:
|
postprocessors.append({
|
||||||
|
'key': 'FFmpegConcat',
|
||||||
|
'only_multi_video': opts.concat_playlist != 'always',
|
||||||
|
'when': 'playlist',
|
||||||
|
})
|
||||||
|
# Exec must be the last PP of each category
|
||||||
|
if opts.exec_before_dl_cmd:
|
||||||
|
opts.exec_cmd.setdefault('before_dl', opts.exec_before_dl_cmd)
|
||||||
|
for when, exec_cmd in opts.exec_cmd.items():
|
||||||
postprocessors.append({
|
postprocessors.append({
|
||||||
'key': 'Exec',
|
'key': 'Exec',
|
||||||
'exec_cmd': opts.exec_cmd,
|
'exec_cmd': exec_cmd,
|
||||||
# Run this only after the files have been moved to their final locations
|
# Run this only after the files have been moved to their final locations
|
||||||
'when': 'after_move'
|
'when': when,
|
||||||
})
|
})
|
||||||
|
|
||||||
def report_args_compat(arg, name):
|
def report_args_compat(arg, name):
|
||||||
@@ -740,6 +756,7 @@ def _real_main(argv=None):
|
|||||||
'skip_playlist_after_errors': opts.skip_playlist_after_errors,
|
'skip_playlist_after_errors': opts.skip_playlist_after_errors,
|
||||||
'cookiefile': opts.cookiefile,
|
'cookiefile': opts.cookiefile,
|
||||||
'cookiesfrombrowser': opts.cookiesfrombrowser,
|
'cookiesfrombrowser': opts.cookiesfrombrowser,
|
||||||
|
'legacyserverconnect': opts.legacy_server_connect,
|
||||||
'nocheckcertificate': opts.no_check_certificate,
|
'nocheckcertificate': opts.no_check_certificate,
|
||||||
'prefer_insecure': opts.prefer_insecure,
|
'prefer_insecure': opts.prefer_insecure,
|
||||||
'proxy': opts.proxy,
|
'proxy': opts.proxy,
|
||||||
|
|||||||
+264
-56
@@ -1,3 +1,4 @@
|
|||||||
|
import contextlib
|
||||||
import ctypes
|
import ctypes
|
||||||
import json
|
import json
|
||||||
import os
|
import os
|
||||||
@@ -7,6 +8,7 @@ import subprocess
|
|||||||
import sys
|
import sys
|
||||||
import tempfile
|
import tempfile
|
||||||
from datetime import datetime, timedelta, timezone
|
from datetime import datetime, timedelta, timezone
|
||||||
|
from enum import Enum, auto
|
||||||
from hashlib import pbkdf2_hmac
|
from hashlib import pbkdf2_hmac
|
||||||
|
|
||||||
from .aes import aes_cbc_decrypt_bytes, aes_gcm_decrypt_and_verify_bytes
|
from .aes import aes_cbc_decrypt_bytes, aes_gcm_decrypt_and_verify_bytes
|
||||||
@@ -15,7 +17,6 @@ from .compat import (
|
|||||||
compat_cookiejar_Cookie,
|
compat_cookiejar_Cookie,
|
||||||
)
|
)
|
||||||
from .utils import (
|
from .utils import (
|
||||||
bug_reports_message,
|
|
||||||
expand_path,
|
expand_path,
|
||||||
Popen,
|
Popen,
|
||||||
YoutubeDLCookieJar,
|
YoutubeDLCookieJar,
|
||||||
@@ -31,19 +32,16 @@ except ImportError:
|
|||||||
|
|
||||||
|
|
||||||
try:
|
try:
|
||||||
import keyring
|
import secretstorage
|
||||||
KEYRING_AVAILABLE = True
|
SECRETSTORAGE_AVAILABLE = True
|
||||||
KEYRING_UNAVAILABLE_REASON = f'due to unknown reasons{bug_reports_message()}'
|
|
||||||
except ImportError:
|
except ImportError:
|
||||||
KEYRING_AVAILABLE = False
|
SECRETSTORAGE_AVAILABLE = False
|
||||||
KEYRING_UNAVAILABLE_REASON = (
|
SECRETSTORAGE_UNAVAILABLE_REASON = (
|
||||||
'as the `keyring` module is not installed. '
|
'as the `secretstorage` module is not installed. '
|
||||||
'Please install by running `python3 -m pip install keyring`. '
|
'Please install by running `python3 -m pip install secretstorage`.')
|
||||||
'Depending on your platform, additional packages may be required '
|
|
||||||
'to access the keyring; see https://pypi.org/project/keyring')
|
|
||||||
except Exception as _err:
|
except Exception as _err:
|
||||||
KEYRING_AVAILABLE = False
|
SECRETSTORAGE_AVAILABLE = False
|
||||||
KEYRING_UNAVAILABLE_REASON = 'as the `keyring` module could not be initialized: %s' % _err
|
SECRETSTORAGE_UNAVAILABLE_REASON = f'as the `secretstorage` module could not be initialized. {_err}'
|
||||||
|
|
||||||
|
|
||||||
CHROMIUM_BASED_BROWSERS = {'brave', 'chrome', 'chromium', 'edge', 'opera', 'vivaldi'}
|
CHROMIUM_BASED_BROWSERS = {'brave', 'chrome', 'chromium', 'edge', 'opera', 'vivaldi'}
|
||||||
@@ -74,8 +72,8 @@ class YDLLogger:
|
|||||||
def load_cookies(cookie_file, browser_specification, ydl):
|
def load_cookies(cookie_file, browser_specification, ydl):
|
||||||
cookie_jars = []
|
cookie_jars = []
|
||||||
if browser_specification is not None:
|
if browser_specification is not None:
|
||||||
browser_name, profile = _parse_browser_specification(*browser_specification)
|
browser_name, profile, keyring = _parse_browser_specification(*browser_specification)
|
||||||
cookie_jars.append(extract_cookies_from_browser(browser_name, profile, YDLLogger(ydl)))
|
cookie_jars.append(extract_cookies_from_browser(browser_name, profile, YDLLogger(ydl), keyring=keyring))
|
||||||
|
|
||||||
if cookie_file is not None:
|
if cookie_file is not None:
|
||||||
cookie_file = expand_path(cookie_file)
|
cookie_file = expand_path(cookie_file)
|
||||||
@@ -87,13 +85,13 @@ def load_cookies(cookie_file, browser_specification, ydl):
|
|||||||
return _merge_cookie_jars(cookie_jars)
|
return _merge_cookie_jars(cookie_jars)
|
||||||
|
|
||||||
|
|
||||||
def extract_cookies_from_browser(browser_name, profile=None, logger=YDLLogger()):
|
def extract_cookies_from_browser(browser_name, profile=None, logger=YDLLogger(), *, keyring=None):
|
||||||
if browser_name == 'firefox':
|
if browser_name == 'firefox':
|
||||||
return _extract_firefox_cookies(profile, logger)
|
return _extract_firefox_cookies(profile, logger)
|
||||||
elif browser_name == 'safari':
|
elif browser_name == 'safari':
|
||||||
return _extract_safari_cookies(profile, logger)
|
return _extract_safari_cookies(profile, logger)
|
||||||
elif browser_name in CHROMIUM_BASED_BROWSERS:
|
elif browser_name in CHROMIUM_BASED_BROWSERS:
|
||||||
return _extract_chrome_cookies(browser_name, profile, logger)
|
return _extract_chrome_cookies(browser_name, profile, keyring, logger)
|
||||||
else:
|
else:
|
||||||
raise ValueError('unknown browser: {}'.format(browser_name))
|
raise ValueError('unknown browser: {}'.format(browser_name))
|
||||||
|
|
||||||
@@ -207,7 +205,7 @@ def _get_chromium_based_browser_settings(browser_name):
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
def _extract_chrome_cookies(browser_name, profile, logger):
|
def _extract_chrome_cookies(browser_name, profile, keyring, logger):
|
||||||
logger.info('Extracting cookies from {}'.format(browser_name))
|
logger.info('Extracting cookies from {}'.format(browser_name))
|
||||||
|
|
||||||
if not SQLITE_AVAILABLE:
|
if not SQLITE_AVAILABLE:
|
||||||
@@ -234,7 +232,7 @@ def _extract_chrome_cookies(browser_name, profile, logger):
|
|||||||
raise FileNotFoundError('could not find {} cookies database in "{}"'.format(browser_name, search_root))
|
raise FileNotFoundError('could not find {} cookies database in "{}"'.format(browser_name, search_root))
|
||||||
logger.debug('Extracting cookies from: "{}"'.format(cookie_database_path))
|
logger.debug('Extracting cookies from: "{}"'.format(cookie_database_path))
|
||||||
|
|
||||||
decryptor = get_cookie_decryptor(config['browser_dir'], config['keyring_name'], logger)
|
decryptor = get_cookie_decryptor(config['browser_dir'], config['keyring_name'], logger, keyring=keyring)
|
||||||
|
|
||||||
with tempfile.TemporaryDirectory(prefix='yt_dlp') as tmpdir:
|
with tempfile.TemporaryDirectory(prefix='yt_dlp') as tmpdir:
|
||||||
cursor = None
|
cursor = None
|
||||||
@@ -247,6 +245,7 @@ def _extract_chrome_cookies(browser_name, profile, logger):
|
|||||||
'expires_utc, {} FROM cookies'.format(secure_column))
|
'expires_utc, {} FROM cookies'.format(secure_column))
|
||||||
jar = YoutubeDLCookieJar()
|
jar = YoutubeDLCookieJar()
|
||||||
failed_cookies = 0
|
failed_cookies = 0
|
||||||
|
unencrypted_cookies = 0
|
||||||
for host_key, name, value, encrypted_value, path, expires_utc, is_secure in cursor.fetchall():
|
for host_key, name, value, encrypted_value, path, expires_utc, is_secure in cursor.fetchall():
|
||||||
host_key = host_key.decode('utf-8')
|
host_key = host_key.decode('utf-8')
|
||||||
name = name.decode('utf-8')
|
name = name.decode('utf-8')
|
||||||
@@ -258,6 +257,8 @@ def _extract_chrome_cookies(browser_name, profile, logger):
|
|||||||
if value is None:
|
if value is None:
|
||||||
failed_cookies += 1
|
failed_cookies += 1
|
||||||
continue
|
continue
|
||||||
|
else:
|
||||||
|
unencrypted_cookies += 1
|
||||||
|
|
||||||
cookie = compat_cookiejar_Cookie(
|
cookie = compat_cookiejar_Cookie(
|
||||||
version=0, name=name, value=value, port=None, port_specified=False,
|
version=0, name=name, value=value, port=None, port_specified=False,
|
||||||
@@ -270,6 +271,9 @@ def _extract_chrome_cookies(browser_name, profile, logger):
|
|||||||
else:
|
else:
|
||||||
failed_message = ''
|
failed_message = ''
|
||||||
logger.info('Extracted {} cookies from {}{}'.format(len(jar), browser_name, failed_message))
|
logger.info('Extracted {} cookies from {}{}'.format(len(jar), browser_name, failed_message))
|
||||||
|
counts = decryptor.cookie_counts.copy()
|
||||||
|
counts['unencrypted'] = unencrypted_cookies
|
||||||
|
logger.debug('cookie version breakdown: {}'.format(counts))
|
||||||
return jar
|
return jar
|
||||||
finally:
|
finally:
|
||||||
if cursor is not None:
|
if cursor is not None:
|
||||||
@@ -305,10 +309,14 @@ class ChromeCookieDecryptor:
|
|||||||
def decrypt(self, encrypted_value):
|
def decrypt(self, encrypted_value):
|
||||||
raise NotImplementedError
|
raise NotImplementedError
|
||||||
|
|
||||||
|
@property
|
||||||
|
def cookie_counts(self):
|
||||||
|
raise NotImplementedError
|
||||||
|
|
||||||
def get_cookie_decryptor(browser_root, browser_keyring_name, logger):
|
|
||||||
|
def get_cookie_decryptor(browser_root, browser_keyring_name, logger, *, keyring=None):
|
||||||
if sys.platform in ('linux', 'linux2'):
|
if sys.platform in ('linux', 'linux2'):
|
||||||
return LinuxChromeCookieDecryptor(browser_keyring_name, logger)
|
return LinuxChromeCookieDecryptor(browser_keyring_name, logger, keyring=keyring)
|
||||||
elif sys.platform == 'darwin':
|
elif sys.platform == 'darwin':
|
||||||
return MacChromeCookieDecryptor(browser_keyring_name, logger)
|
return MacChromeCookieDecryptor(browser_keyring_name, logger)
|
||||||
elif sys.platform == 'win32':
|
elif sys.platform == 'win32':
|
||||||
@@ -319,13 +327,12 @@ def get_cookie_decryptor(browser_root, browser_keyring_name, logger):
|
|||||||
|
|
||||||
|
|
||||||
class LinuxChromeCookieDecryptor(ChromeCookieDecryptor):
|
class LinuxChromeCookieDecryptor(ChromeCookieDecryptor):
|
||||||
def __init__(self, browser_keyring_name, logger):
|
def __init__(self, browser_keyring_name, logger, *, keyring=None):
|
||||||
self._logger = logger
|
self._logger = logger
|
||||||
self._v10_key = self.derive_key(b'peanuts')
|
self._v10_key = self.derive_key(b'peanuts')
|
||||||
if KEYRING_AVAILABLE:
|
password = _get_linux_keyring_password(browser_keyring_name, keyring, logger)
|
||||||
self._v11_key = self.derive_key(_get_linux_keyring_password(browser_keyring_name))
|
self._v11_key = None if password is None else self.derive_key(password)
|
||||||
else:
|
self._cookie_counts = {'v10': 0, 'v11': 0, 'other': 0}
|
||||||
self._v11_key = None
|
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def derive_key(password):
|
def derive_key(password):
|
||||||
@@ -333,20 +340,27 @@ class LinuxChromeCookieDecryptor(ChromeCookieDecryptor):
|
|||||||
# https://chromium.googlesource.com/chromium/src/+/refs/heads/main/components/os_crypt/os_crypt_linux.cc
|
# https://chromium.googlesource.com/chromium/src/+/refs/heads/main/components/os_crypt/os_crypt_linux.cc
|
||||||
return pbkdf2_sha1(password, salt=b'saltysalt', iterations=1, key_length=16)
|
return pbkdf2_sha1(password, salt=b'saltysalt', iterations=1, key_length=16)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def cookie_counts(self):
|
||||||
|
return self._cookie_counts
|
||||||
|
|
||||||
def decrypt(self, encrypted_value):
|
def decrypt(self, encrypted_value):
|
||||||
version = encrypted_value[:3]
|
version = encrypted_value[:3]
|
||||||
ciphertext = encrypted_value[3:]
|
ciphertext = encrypted_value[3:]
|
||||||
|
|
||||||
if version == b'v10':
|
if version == b'v10':
|
||||||
|
self._cookie_counts['v10'] += 1
|
||||||
return _decrypt_aes_cbc(ciphertext, self._v10_key, self._logger)
|
return _decrypt_aes_cbc(ciphertext, self._v10_key, self._logger)
|
||||||
|
|
||||||
elif version == b'v11':
|
elif version == b'v11':
|
||||||
|
self._cookie_counts['v11'] += 1
|
||||||
if self._v11_key is None:
|
if self._v11_key is None:
|
||||||
self._logger.warning(f'cannot decrypt cookie {KEYRING_UNAVAILABLE_REASON}', only_once=True)
|
self._logger.warning('cannot decrypt v11 cookies: no key found', only_once=True)
|
||||||
return None
|
return None
|
||||||
return _decrypt_aes_cbc(ciphertext, self._v11_key, self._logger)
|
return _decrypt_aes_cbc(ciphertext, self._v11_key, self._logger)
|
||||||
|
|
||||||
else:
|
else:
|
||||||
|
self._cookie_counts['other'] += 1
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
@@ -355,6 +369,7 @@ class MacChromeCookieDecryptor(ChromeCookieDecryptor):
|
|||||||
self._logger = logger
|
self._logger = logger
|
||||||
password = _get_mac_keyring_password(browser_keyring_name, logger)
|
password = _get_mac_keyring_password(browser_keyring_name, logger)
|
||||||
self._v10_key = None if password is None else self.derive_key(password)
|
self._v10_key = None if password is None else self.derive_key(password)
|
||||||
|
self._cookie_counts = {'v10': 0, 'other': 0}
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def derive_key(password):
|
def derive_key(password):
|
||||||
@@ -362,11 +377,16 @@ class MacChromeCookieDecryptor(ChromeCookieDecryptor):
|
|||||||
# https://chromium.googlesource.com/chromium/src/+/refs/heads/main/components/os_crypt/os_crypt_mac.mm
|
# https://chromium.googlesource.com/chromium/src/+/refs/heads/main/components/os_crypt/os_crypt_mac.mm
|
||||||
return pbkdf2_sha1(password, salt=b'saltysalt', iterations=1003, key_length=16)
|
return pbkdf2_sha1(password, salt=b'saltysalt', iterations=1003, key_length=16)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def cookie_counts(self):
|
||||||
|
return self._cookie_counts
|
||||||
|
|
||||||
def decrypt(self, encrypted_value):
|
def decrypt(self, encrypted_value):
|
||||||
version = encrypted_value[:3]
|
version = encrypted_value[:3]
|
||||||
ciphertext = encrypted_value[3:]
|
ciphertext = encrypted_value[3:]
|
||||||
|
|
||||||
if version == b'v10':
|
if version == b'v10':
|
||||||
|
self._cookie_counts['v10'] += 1
|
||||||
if self._v10_key is None:
|
if self._v10_key is None:
|
||||||
self._logger.warning('cannot decrypt v10 cookies: no key found', only_once=True)
|
self._logger.warning('cannot decrypt v10 cookies: no key found', only_once=True)
|
||||||
return None
|
return None
|
||||||
@@ -374,6 +394,7 @@ class MacChromeCookieDecryptor(ChromeCookieDecryptor):
|
|||||||
return _decrypt_aes_cbc(ciphertext, self._v10_key, self._logger)
|
return _decrypt_aes_cbc(ciphertext, self._v10_key, self._logger)
|
||||||
|
|
||||||
else:
|
else:
|
||||||
|
self._cookie_counts['other'] += 1
|
||||||
# other prefixes are considered 'old data' which were stored as plaintext
|
# other prefixes are considered 'old data' which were stored as plaintext
|
||||||
# https://chromium.googlesource.com/chromium/src/+/refs/heads/main/components/os_crypt/os_crypt_mac.mm
|
# https://chromium.googlesource.com/chromium/src/+/refs/heads/main/components/os_crypt/os_crypt_mac.mm
|
||||||
return encrypted_value
|
return encrypted_value
|
||||||
@@ -383,12 +404,18 @@ class WindowsChromeCookieDecryptor(ChromeCookieDecryptor):
|
|||||||
def __init__(self, browser_root, logger):
|
def __init__(self, browser_root, logger):
|
||||||
self._logger = logger
|
self._logger = logger
|
||||||
self._v10_key = _get_windows_v10_key(browser_root, logger)
|
self._v10_key = _get_windows_v10_key(browser_root, logger)
|
||||||
|
self._cookie_counts = {'v10': 0, 'other': 0}
|
||||||
|
|
||||||
|
@property
|
||||||
|
def cookie_counts(self):
|
||||||
|
return self._cookie_counts
|
||||||
|
|
||||||
def decrypt(self, encrypted_value):
|
def decrypt(self, encrypted_value):
|
||||||
version = encrypted_value[:3]
|
version = encrypted_value[:3]
|
||||||
ciphertext = encrypted_value[3:]
|
ciphertext = encrypted_value[3:]
|
||||||
|
|
||||||
if version == b'v10':
|
if version == b'v10':
|
||||||
|
self._cookie_counts['v10'] += 1
|
||||||
if self._v10_key is None:
|
if self._v10_key is None:
|
||||||
self._logger.warning('cannot decrypt v10 cookies: no key found', only_once=True)
|
self._logger.warning('cannot decrypt v10 cookies: no key found', only_once=True)
|
||||||
return None
|
return None
|
||||||
@@ -408,6 +435,7 @@ class WindowsChromeCookieDecryptor(ChromeCookieDecryptor):
|
|||||||
return _decrypt_aes_gcm(ciphertext, self._v10_key, nonce, authentication_tag, self._logger)
|
return _decrypt_aes_gcm(ciphertext, self._v10_key, nonce, authentication_tag, self._logger)
|
||||||
|
|
||||||
else:
|
else:
|
||||||
|
self._cookie_counts['other'] += 1
|
||||||
# any other prefix means the data is DPAPI encrypted
|
# any other prefix means the data is DPAPI encrypted
|
||||||
# https://chromium.googlesource.com/chromium/src/+/refs/heads/main/components/os_crypt/os_crypt_win.cc
|
# https://chromium.googlesource.com/chromium/src/+/refs/heads/main/components/os_crypt/os_crypt_win.cc
|
||||||
return _decrypt_windows_dpapi(encrypted_value, self._logger).decode('utf-8')
|
return _decrypt_windows_dpapi(encrypted_value, self._logger).decode('utf-8')
|
||||||
@@ -577,42 +605,221 @@ def parse_safari_cookies(data, jar=None, logger=YDLLogger()):
|
|||||||
return jar
|
return jar
|
||||||
|
|
||||||
|
|
||||||
def _get_linux_keyring_password(browser_keyring_name):
|
class _LinuxDesktopEnvironment(Enum):
|
||||||
password = keyring.get_password('{} Keys'.format(browser_keyring_name),
|
"""
|
||||||
'{} Safe Storage'.format(browser_keyring_name))
|
https://chromium.googlesource.com/chromium/src/+/refs/heads/main/base/nix/xdg_util.h
|
||||||
if password is None:
|
DesktopEnvironment
|
||||||
# this sometimes occurs in KDE because chrome does not check hasEntry and instead
|
"""
|
||||||
# just tries to read the value (which kwallet returns "") whereas keyring checks hasEntry
|
OTHER = auto()
|
||||||
# to verify this:
|
CINNAMON = auto()
|
||||||
# dbus-monitor "interface='org.kde.KWallet'" "type=method_return"
|
GNOME = auto()
|
||||||
# while starting chrome.
|
KDE = auto()
|
||||||
# this may be a bug as the intended behaviour is to generate a random password and store
|
PANTHEON = auto()
|
||||||
# it, but that doesn't matter here.
|
UNITY = auto()
|
||||||
password = ''
|
XFCE = auto()
|
||||||
return password.encode('utf-8')
|
|
||||||
|
|
||||||
|
class _LinuxKeyring(Enum):
|
||||||
|
"""
|
||||||
|
https://chromium.googlesource.com/chromium/src/+/refs/heads/main/components/os_crypt/key_storage_util_linux.h
|
||||||
|
SelectedLinuxBackend
|
||||||
|
"""
|
||||||
|
KWALLET = auto()
|
||||||
|
GNOMEKEYRING = auto()
|
||||||
|
BASICTEXT = auto()
|
||||||
|
|
||||||
|
|
||||||
|
SUPPORTED_KEYRINGS = _LinuxKeyring.__members__.keys()
|
||||||
|
|
||||||
|
|
||||||
|
def _get_linux_desktop_environment(env):
|
||||||
|
"""
|
||||||
|
https://chromium.googlesource.com/chromium/src/+/refs/heads/main/base/nix/xdg_util.cc
|
||||||
|
GetDesktopEnvironment
|
||||||
|
"""
|
||||||
|
xdg_current_desktop = env.get('XDG_CURRENT_DESKTOP', None)
|
||||||
|
desktop_session = env.get('DESKTOP_SESSION', None)
|
||||||
|
if xdg_current_desktop is not None:
|
||||||
|
xdg_current_desktop = xdg_current_desktop.split(':')[0].strip()
|
||||||
|
|
||||||
|
if xdg_current_desktop == 'Unity':
|
||||||
|
if desktop_session is not None and 'gnome-fallback' in desktop_session:
|
||||||
|
return _LinuxDesktopEnvironment.GNOME
|
||||||
|
else:
|
||||||
|
return _LinuxDesktopEnvironment.UNITY
|
||||||
|
elif xdg_current_desktop == 'GNOME':
|
||||||
|
return _LinuxDesktopEnvironment.GNOME
|
||||||
|
elif xdg_current_desktop == 'X-Cinnamon':
|
||||||
|
return _LinuxDesktopEnvironment.CINNAMON
|
||||||
|
elif xdg_current_desktop == 'KDE':
|
||||||
|
return _LinuxDesktopEnvironment.KDE
|
||||||
|
elif xdg_current_desktop == 'Pantheon':
|
||||||
|
return _LinuxDesktopEnvironment.PANTHEON
|
||||||
|
elif xdg_current_desktop == 'XFCE':
|
||||||
|
return _LinuxDesktopEnvironment.XFCE
|
||||||
|
elif desktop_session is not None:
|
||||||
|
if desktop_session in ('mate', 'gnome'):
|
||||||
|
return _LinuxDesktopEnvironment.GNOME
|
||||||
|
elif 'kde' in desktop_session:
|
||||||
|
return _LinuxDesktopEnvironment.KDE
|
||||||
|
elif 'xfce' in desktop_session:
|
||||||
|
return _LinuxDesktopEnvironment.XFCE
|
||||||
|
else:
|
||||||
|
if 'GNOME_DESKTOP_SESSION_ID' in env:
|
||||||
|
return _LinuxDesktopEnvironment.GNOME
|
||||||
|
elif 'KDE_FULL_SESSION' in env:
|
||||||
|
return _LinuxDesktopEnvironment.KDE
|
||||||
|
else:
|
||||||
|
return _LinuxDesktopEnvironment.OTHER
|
||||||
|
|
||||||
|
|
||||||
|
def _choose_linux_keyring(logger):
|
||||||
|
"""
|
||||||
|
https://chromium.googlesource.com/chromium/src/+/refs/heads/main/components/os_crypt/key_storage_util_linux.cc
|
||||||
|
SelectBackend
|
||||||
|
"""
|
||||||
|
desktop_environment = _get_linux_desktop_environment(os.environ)
|
||||||
|
logger.debug('detected desktop environment: {}'.format(desktop_environment.name))
|
||||||
|
if desktop_environment == _LinuxDesktopEnvironment.KDE:
|
||||||
|
linux_keyring = _LinuxKeyring.KWALLET
|
||||||
|
elif desktop_environment == _LinuxDesktopEnvironment.OTHER:
|
||||||
|
linux_keyring = _LinuxKeyring.BASICTEXT
|
||||||
|
else:
|
||||||
|
linux_keyring = _LinuxKeyring.GNOMEKEYRING
|
||||||
|
return linux_keyring
|
||||||
|
|
||||||
|
|
||||||
|
def _get_kwallet_network_wallet(logger):
|
||||||
|
""" The name of the wallet used to store network passwords.
|
||||||
|
|
||||||
|
https://chromium.googlesource.com/chromium/src/+/refs/heads/main/components/os_crypt/kwallet_dbus.cc
|
||||||
|
KWalletDBus::NetworkWallet
|
||||||
|
which does a dbus call to the following function:
|
||||||
|
https://api.kde.org/frameworks/kwallet/html/classKWallet_1_1Wallet.html
|
||||||
|
Wallet::NetworkWallet
|
||||||
|
"""
|
||||||
|
default_wallet = 'kdewallet'
|
||||||
|
try:
|
||||||
|
proc = Popen([
|
||||||
|
'dbus-send', '--session', '--print-reply=literal',
|
||||||
|
'--dest=org.kde.kwalletd5',
|
||||||
|
'/modules/kwalletd5',
|
||||||
|
'org.kde.KWallet.networkWallet'
|
||||||
|
], stdout=subprocess.PIPE, stderr=subprocess.DEVNULL)
|
||||||
|
|
||||||
|
stdout, stderr = proc.communicate_or_kill()
|
||||||
|
if proc.returncode != 0:
|
||||||
|
logger.warning('failed to read NetworkWallet')
|
||||||
|
return default_wallet
|
||||||
|
else:
|
||||||
|
network_wallet = stdout.decode('utf-8').strip()
|
||||||
|
logger.debug('NetworkWallet = "{}"'.format(network_wallet))
|
||||||
|
return network_wallet
|
||||||
|
except BaseException as e:
|
||||||
|
logger.warning('exception while obtaining NetworkWallet: {}'.format(e))
|
||||||
|
return default_wallet
|
||||||
|
|
||||||
|
|
||||||
|
def _get_kwallet_password(browser_keyring_name, logger):
|
||||||
|
logger.debug('using kwallet-query to obtain password from kwallet')
|
||||||
|
|
||||||
|
if shutil.which('kwallet-query') is None:
|
||||||
|
logger.error('kwallet-query command not found. KWallet and kwallet-query '
|
||||||
|
'must be installed to read from KWallet. kwallet-query should be'
|
||||||
|
'included in the kwallet package for your distribution')
|
||||||
|
return b''
|
||||||
|
|
||||||
|
network_wallet = _get_kwallet_network_wallet(logger)
|
||||||
|
|
||||||
|
try:
|
||||||
|
proc = Popen([
|
||||||
|
'kwallet-query',
|
||||||
|
'--read-password', '{} Safe Storage'.format(browser_keyring_name),
|
||||||
|
'--folder', '{} Keys'.format(browser_keyring_name),
|
||||||
|
network_wallet
|
||||||
|
], stdout=subprocess.PIPE, stderr=subprocess.DEVNULL)
|
||||||
|
|
||||||
|
stdout, stderr = proc.communicate_or_kill()
|
||||||
|
if proc.returncode != 0:
|
||||||
|
logger.error('kwallet-query failed with return code {}. Please consult '
|
||||||
|
'the kwallet-query man page for details'.format(proc.returncode))
|
||||||
|
return b''
|
||||||
|
else:
|
||||||
|
if stdout.lower().startswith(b'failed to read'):
|
||||||
|
logger.debug('failed to read password from kwallet. Using empty string instead')
|
||||||
|
# this sometimes occurs in KDE because chrome does not check hasEntry and instead
|
||||||
|
# just tries to read the value (which kwallet returns "") whereas kwallet-query
|
||||||
|
# checks hasEntry. To verify this:
|
||||||
|
# dbus-monitor "interface='org.kde.KWallet'" "type=method_return"
|
||||||
|
# while starting chrome.
|
||||||
|
# this may be a bug as the intended behaviour is to generate a random password and store
|
||||||
|
# it, but that doesn't matter here.
|
||||||
|
return b''
|
||||||
|
else:
|
||||||
|
logger.debug('password found')
|
||||||
|
if stdout[-1:] == b'\n':
|
||||||
|
stdout = stdout[:-1]
|
||||||
|
return stdout
|
||||||
|
except BaseException as e:
|
||||||
|
logger.warning(f'exception running kwallet-query: {type(e).__name__}({e})')
|
||||||
|
return b''
|
||||||
|
|
||||||
|
|
||||||
|
def _get_gnome_keyring_password(browser_keyring_name, logger):
|
||||||
|
if not SECRETSTORAGE_AVAILABLE:
|
||||||
|
logger.error('secretstorage not available {}'.format(SECRETSTORAGE_UNAVAILABLE_REASON))
|
||||||
|
return b''
|
||||||
|
# the Gnome keyring does not seem to organise keys in the same way as KWallet,
|
||||||
|
# using `dbus-monitor` during startup, it can be observed that chromium lists all keys
|
||||||
|
# and presumably searches for its key in the list. It appears that we must do the same.
|
||||||
|
# https://github.com/jaraco/keyring/issues/556
|
||||||
|
with contextlib.closing(secretstorage.dbus_init()) as con:
|
||||||
|
col = secretstorage.get_default_collection(con)
|
||||||
|
for item in col.get_all_items():
|
||||||
|
if item.get_label() == '{} Safe Storage'.format(browser_keyring_name):
|
||||||
|
return item.get_secret()
|
||||||
|
else:
|
||||||
|
logger.error('failed to read from keyring')
|
||||||
|
return b''
|
||||||
|
|
||||||
|
|
||||||
|
def _get_linux_keyring_password(browser_keyring_name, keyring, logger):
|
||||||
|
# note: chrome/chromium can be run with the following flags to determine which keyring backend
|
||||||
|
# it has chosen to use
|
||||||
|
# chromium --enable-logging=stderr --v=1 2>&1 | grep key_storage_
|
||||||
|
# Chromium supports a flag: --password-store=<basic|gnome|kwallet> so the automatic detection
|
||||||
|
# will not be sufficient in all cases.
|
||||||
|
|
||||||
|
keyring = _LinuxKeyring[keyring] if keyring else _choose_linux_keyring(logger)
|
||||||
|
logger.debug(f'Chosen keyring: {keyring.name}')
|
||||||
|
|
||||||
|
if keyring == _LinuxKeyring.KWALLET:
|
||||||
|
return _get_kwallet_password(browser_keyring_name, logger)
|
||||||
|
elif keyring == _LinuxKeyring.GNOMEKEYRING:
|
||||||
|
return _get_gnome_keyring_password(browser_keyring_name, logger)
|
||||||
|
elif keyring == _LinuxKeyring.BASICTEXT:
|
||||||
|
# when basic text is chosen, all cookies are stored as v10 (so no keyring password is required)
|
||||||
|
return None
|
||||||
|
assert False, f'Unknown keyring {keyring}'
|
||||||
|
|
||||||
|
|
||||||
def _get_mac_keyring_password(browser_keyring_name, logger):
|
def _get_mac_keyring_password(browser_keyring_name, logger):
|
||||||
if KEYRING_AVAILABLE:
|
logger.debug('using find-generic-password to obtain password from OSX keychain')
|
||||||
logger.debug('using keyring to obtain password')
|
try:
|
||||||
password = keyring.get_password('{} Safe Storage'.format(browser_keyring_name), browser_keyring_name)
|
|
||||||
return password.encode('utf-8')
|
|
||||||
else:
|
|
||||||
logger.debug('using find-generic-password to obtain password')
|
|
||||||
proc = Popen(
|
proc = Popen(
|
||||||
['security', 'find-generic-password',
|
['security', 'find-generic-password',
|
||||||
'-w', # write password to stdout
|
'-w', # write password to stdout
|
||||||
'-a', browser_keyring_name, # match 'account'
|
'-a', browser_keyring_name, # match 'account'
|
||||||
'-s', '{} Safe Storage'.format(browser_keyring_name)], # match 'service'
|
'-s', '{} Safe Storage'.format(browser_keyring_name)], # match 'service'
|
||||||
stdout=subprocess.PIPE, stderr=subprocess.DEVNULL)
|
stdout=subprocess.PIPE, stderr=subprocess.DEVNULL)
|
||||||
try:
|
|
||||||
stdout, stderr = proc.communicate_or_kill()
|
stdout, stderr = proc.communicate_or_kill()
|
||||||
if stdout[-1:] == b'\n':
|
if stdout[-1:] == b'\n':
|
||||||
stdout = stdout[:-1]
|
stdout = stdout[:-1]
|
||||||
return stdout
|
return stdout
|
||||||
except BaseException as e:
|
except BaseException as e:
|
||||||
logger.warning(f'exception running find-generic-password: {type(e).__name__}({e})')
|
logger.warning(f'exception running find-generic-password: {type(e).__name__}({e})')
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
def _get_windows_v10_key(browser_root, logger):
|
def _get_windows_v10_key(browser_root, logger):
|
||||||
@@ -736,10 +943,11 @@ def _is_path(value):
|
|||||||
return os.path.sep in value
|
return os.path.sep in value
|
||||||
|
|
||||||
|
|
||||||
def _parse_browser_specification(browser_name, profile=None):
|
def _parse_browser_specification(browser_name, profile=None, keyring=None):
|
||||||
browser_name = browser_name.lower()
|
|
||||||
if browser_name not in SUPPORTED_BROWSERS:
|
if browser_name not in SUPPORTED_BROWSERS:
|
||||||
raise ValueError(f'unsupported browser: "{browser_name}"')
|
raise ValueError(f'unsupported browser: "{browser_name}"')
|
||||||
|
if keyring not in (None, *SUPPORTED_KEYRINGS):
|
||||||
|
raise ValueError(f'unsupported keyring: "{keyring}"')
|
||||||
if profile is not None and _is_path(profile):
|
if profile is not None and _is_path(profile):
|
||||||
profile = os.path.expanduser(profile)
|
profile = os.path.expanduser(profile)
|
||||||
return browser_name, profile
|
return browser_name, profile, keyring
|
||||||
|
|||||||
@@ -265,6 +265,7 @@ class Aria2cFD(ExternalFD):
|
|||||||
cmd += self._option('--all-proxy', 'proxy')
|
cmd += self._option('--all-proxy', 'proxy')
|
||||||
cmd += self._bool_option('--check-certificate', 'nocheckcertificate', 'false', 'true', '=')
|
cmd += self._bool_option('--check-certificate', 'nocheckcertificate', 'false', 'true', '=')
|
||||||
cmd += self._bool_option('--remote-time', 'updatetime', 'true', 'false', '=')
|
cmd += self._bool_option('--remote-time', 'updatetime', 'true', 'false', '=')
|
||||||
|
cmd += self._bool_option('--show-console-readout', 'noprogress', 'false', 'true', '=')
|
||||||
cmd += self._configuration_args()
|
cmd += self._configuration_args()
|
||||||
|
|
||||||
# aria2c strips out spaces from the beginning/end of filenames and paths.
|
# aria2c strips out spaces from the beginning/end of filenames and paths.
|
||||||
@@ -303,7 +304,7 @@ class HttpieFD(ExternalFD):
|
|||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def available(cls, path=None):
|
def available(cls, path=None):
|
||||||
return ExternalFD.available(cls, path or 'http')
|
return super().available(path or 'http')
|
||||||
|
|
||||||
def _make_cmd(self, tmpfilename, info_dict):
|
def _make_cmd(self, tmpfilename, info_dict):
|
||||||
cmd = ['http', '--download', '--output', tmpfilename, info_dict['url']]
|
cmd = ['http', '--download', '--output', tmpfilename, info_dict['url']]
|
||||||
|
|||||||
@@ -433,6 +433,7 @@ class FragmentFD(FileDownloader):
|
|||||||
|
|
||||||
def download_fragment(fragment, ctx):
|
def download_fragment(fragment, ctx):
|
||||||
frag_index = ctx['fragment_index'] = fragment['frag_index']
|
frag_index = ctx['fragment_index'] = fragment['frag_index']
|
||||||
|
ctx['last_error'] = None
|
||||||
if not interrupt_trigger[0]:
|
if not interrupt_trigger[0]:
|
||||||
return False, frag_index
|
return False, frag_index
|
||||||
headers = info_dict.get('http_headers', {}).copy()
|
headers = info_dict.get('http_headers', {}).copy()
|
||||||
@@ -455,6 +456,7 @@ class FragmentFD(FileDownloader):
|
|||||||
# See https://github.com/ytdl-org/youtube-dl/issues/10165,
|
# See https://github.com/ytdl-org/youtube-dl/issues/10165,
|
||||||
# https://github.com/ytdl-org/youtube-dl/issues/10448).
|
# https://github.com/ytdl-org/youtube-dl/issues/10448).
|
||||||
count += 1
|
count += 1
|
||||||
|
ctx['last_error'] = err
|
||||||
if count <= fragment_retries:
|
if count <= fragment_retries:
|
||||||
self.report_retry_fragment(err, frag_index, count, fragment_retries)
|
self.report_retry_fragment(err, frag_index, count, fragment_retries)
|
||||||
except DownloadError:
|
except DownloadError:
|
||||||
|
|||||||
@@ -10,7 +10,11 @@ from ..utils import (
|
|||||||
determine_ext,
|
determine_ext,
|
||||||
ExtractorError,
|
ExtractorError,
|
||||||
int_or_none,
|
int_or_none,
|
||||||
|
qualities,
|
||||||
|
traverse_obj,
|
||||||
unified_strdate,
|
unified_strdate,
|
||||||
|
unified_timestamp,
|
||||||
|
update_url_query,
|
||||||
url_or_none,
|
url_or_none,
|
||||||
urlencode_postdata,
|
urlencode_postdata,
|
||||||
xpath_text,
|
xpath_text,
|
||||||
@@ -380,3 +384,96 @@ class AfreecaTVIE(InfoExtractor):
|
|||||||
})
|
})
|
||||||
|
|
||||||
return info
|
return info
|
||||||
|
|
||||||
|
|
||||||
|
class AfreecaTVLiveIE(AfreecaTVIE):
|
||||||
|
|
||||||
|
IE_NAME = 'afreecatv:live'
|
||||||
|
_VALID_URL = r'https?://play\.afreeca(?:tv)?\.com/(?P<id>[^/]+)(?:/(?P<bno>\d+))?'
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://play.afreecatv.com/pyh3646/237852185',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '237852185',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': '【 우루과이 오늘은 무슨일이? 】',
|
||||||
|
'uploader': '박진우[JINU]',
|
||||||
|
'uploader_id': 'pyh3646',
|
||||||
|
'timestamp': 1640661495,
|
||||||
|
'is_live': True,
|
||||||
|
},
|
||||||
|
'skip': 'Livestream has ended',
|
||||||
|
}, {
|
||||||
|
'url': 'http://play.afreeca.com/pyh3646/237852185',
|
||||||
|
'only_matching': True,
|
||||||
|
}, {
|
||||||
|
'url': 'http://play.afreeca.com/pyh3646',
|
||||||
|
'only_matching': True,
|
||||||
|
}]
|
||||||
|
|
||||||
|
_LIVE_API_URL = 'https://live.afreecatv.com/afreeca/player_live_api.php'
|
||||||
|
|
||||||
|
_QUALITIES = ('sd', 'hd', 'hd2k', 'original')
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
broadcaster_id, broadcast_no = self._match_valid_url(url).group('id', 'bno')
|
||||||
|
|
||||||
|
info = self._download_json(self._LIVE_API_URL, broadcaster_id, fatal=False,
|
||||||
|
data=urlencode_postdata({'bid': broadcaster_id})) or {}
|
||||||
|
channel_info = info.get('CHANNEL') or {}
|
||||||
|
broadcaster_id = channel_info.get('BJID') or broadcaster_id
|
||||||
|
broadcast_no = channel_info.get('BNO') or broadcast_no
|
||||||
|
if not broadcast_no:
|
||||||
|
raise ExtractorError(f'Unable to extract broadcast number ({broadcaster_id} may not be live)', expected=True)
|
||||||
|
|
||||||
|
formats = []
|
||||||
|
quality_key = qualities(self._QUALITIES)
|
||||||
|
for quality_str in self._QUALITIES:
|
||||||
|
aid_response = self._download_json(
|
||||||
|
self._LIVE_API_URL, broadcast_no, fatal=False,
|
||||||
|
data=urlencode_postdata({
|
||||||
|
'bno': broadcast_no,
|
||||||
|
'stream_type': 'common',
|
||||||
|
'type': 'aid',
|
||||||
|
'quality': quality_str,
|
||||||
|
}),
|
||||||
|
note=f'Downloading access token for {quality_str} stream',
|
||||||
|
errnote=f'Unable to download access token for {quality_str} stream')
|
||||||
|
aid = traverse_obj(aid_response, ('CHANNEL', 'AID'))
|
||||||
|
if not aid:
|
||||||
|
continue
|
||||||
|
|
||||||
|
stream_base_url = channel_info.get('RMD') or 'https://livestream-manager.afreecatv.com'
|
||||||
|
stream_info = self._download_json(
|
||||||
|
f'{stream_base_url}/broad_stream_assign.html', broadcast_no, fatal=False,
|
||||||
|
query={
|
||||||
|
'return_type': channel_info.get('CDN', 'gcp_cdn'),
|
||||||
|
'broad_key': f'{broadcast_no}-common-{quality_str}-hls',
|
||||||
|
},
|
||||||
|
note=f'Downloading metadata for {quality_str} stream',
|
||||||
|
errnote=f'Unable to download metadata for {quality_str} stream') or {}
|
||||||
|
|
||||||
|
if stream_info.get('view_url'):
|
||||||
|
formats.append({
|
||||||
|
'format_id': quality_str,
|
||||||
|
'url': update_url_query(stream_info['view_url'], {'aid': aid}),
|
||||||
|
'ext': 'mp4',
|
||||||
|
'protocol': 'm3u8',
|
||||||
|
'quality': quality_key(quality_str),
|
||||||
|
})
|
||||||
|
|
||||||
|
self._sort_formats(formats)
|
||||||
|
|
||||||
|
station_info = self._download_json(
|
||||||
|
'https://st.afreecatv.com/api/get_station_status.php', broadcast_no,
|
||||||
|
query={'szBjId': broadcaster_id}, fatal=False,
|
||||||
|
note='Downloading channel metadata', errnote='Unable to download channel metadata') or {}
|
||||||
|
|
||||||
|
return {
|
||||||
|
'id': broadcast_no,
|
||||||
|
'title': channel_info.get('TITLE') or station_info.get('station_title'),
|
||||||
|
'uploader': channel_info.get('BJNICK') or station_info.get('station_name'),
|
||||||
|
'uploader_id': broadcaster_id,
|
||||||
|
'timestamp': unified_timestamp(station_info.get('broad_start')),
|
||||||
|
'formats': formats,
|
||||||
|
'is_live': True,
|
||||||
|
}
|
||||||
|
|||||||
@@ -33,19 +33,22 @@ class AparatIE(InfoExtractor):
|
|||||||
'only_matching': True,
|
'only_matching': True,
|
||||||
}]
|
}]
|
||||||
|
|
||||||
|
def _parse_options(self, webpage, video_id, fatal=True):
|
||||||
|
return self._parse_json(self._search_regex(
|
||||||
|
r'options\s*=\s*({.+?})\s*;', webpage, 'options', default='{}'), video_id)
|
||||||
|
|
||||||
def _real_extract(self, url):
|
def _real_extract(self, url):
|
||||||
video_id = self._match_id(url)
|
video_id = self._match_id(url)
|
||||||
|
|
||||||
# Provides more metadata
|
# If available, provides more metadata
|
||||||
webpage = self._download_webpage(url, video_id, fatal=False)
|
webpage = self._download_webpage(url, video_id, fatal=False)
|
||||||
|
options = self._parse_options(webpage, video_id, fatal=False)
|
||||||
|
|
||||||
if not webpage:
|
if not options:
|
||||||
webpage = self._download_webpage(
|
webpage = self._download_webpage(
|
||||||
'http://www.aparat.com/video/video/embed/vt/frame/showvideo/yes/videohash/' + video_id,
|
'http://www.aparat.com/video/video/embed/vt/frame/showvideo/yes/videohash/' + video_id,
|
||||||
video_id)
|
video_id, 'Downloading embed webpage')
|
||||||
|
options = self._parse_options(webpage, video_id)
|
||||||
options = self._parse_json(self._search_regex(
|
|
||||||
r'options\s*=\s*({.+?})\s*;', webpage, 'options'), video_id)
|
|
||||||
|
|
||||||
formats = []
|
formats = []
|
||||||
for sources in (options.get('multiSRC') or []):
|
for sources in (options.get('multiSRC') or []):
|
||||||
|
|||||||
@@ -376,9 +376,24 @@ class ARDIE(InfoExtractor):
|
|||||||
formats.append(f)
|
formats.append(f)
|
||||||
self._sort_formats(formats)
|
self._sort_formats(formats)
|
||||||
|
|
||||||
|
_SUB_FORMATS = (
|
||||||
|
('./dataTimedText', 'ttml'),
|
||||||
|
('./dataTimedTextNoOffset', 'ttml'),
|
||||||
|
('./dataTimedTextVtt', 'vtt'),
|
||||||
|
)
|
||||||
|
|
||||||
|
subtitles = {}
|
||||||
|
for subsel, subext in _SUB_FORMATS:
|
||||||
|
for node in video_node.findall(subsel):
|
||||||
|
subtitles.setdefault('de', []).append({
|
||||||
|
'url': node.attrib['url'],
|
||||||
|
'ext': subext,
|
||||||
|
})
|
||||||
|
|
||||||
return {
|
return {
|
||||||
'id': xpath_text(video_node, './videoId', default=display_id),
|
'id': xpath_text(video_node, './videoId', default=display_id),
|
||||||
'formats': formats,
|
'formats': formats,
|
||||||
|
'subtitles': subtitles,
|
||||||
'display_id': display_id,
|
'display_id': display_id,
|
||||||
'title': video_node.find('./title').text,
|
'title': video_node.find('./title').text,
|
||||||
'duration': parse_duration(video_node.find('./duration').text),
|
'duration': parse_duration(video_node.find('./duration').text),
|
||||||
|
|||||||
@@ -7,6 +7,7 @@ from ..compat import (
|
|||||||
compat_urllib_parse_urlparse,
|
compat_urllib_parse_urlparse,
|
||||||
)
|
)
|
||||||
from ..utils import (
|
from ..utils import (
|
||||||
|
format_field,
|
||||||
float_or_none,
|
float_or_none,
|
||||||
int_or_none,
|
int_or_none,
|
||||||
parse_iso8601,
|
parse_iso8601,
|
||||||
@@ -92,7 +93,7 @@ class ArnesIE(InfoExtractor):
|
|||||||
'timestamp': parse_iso8601(video.get('creationTime')),
|
'timestamp': parse_iso8601(video.get('creationTime')),
|
||||||
'channel': channel.get('name'),
|
'channel': channel.get('name'),
|
||||||
'channel_id': channel_id,
|
'channel_id': channel_id,
|
||||||
'channel_url': self._BASE_URL + '/?channel=' + channel_id if channel_id else None,
|
'channel_url': format_field(channel_id, template=f'{self._BASE_URL}/?channel=%s'),
|
||||||
'duration': float_or_none(video.get('duration'), 1000),
|
'duration': float_or_none(video.get('duration'), 1000),
|
||||||
'view_count': int_or_none(video.get('views')),
|
'view_count': int_or_none(video.get('views')),
|
||||||
'tags': video.get('hashtags'),
|
'tags': video.get('hashtags'),
|
||||||
|
|||||||
@@ -9,6 +9,7 @@ from ..compat import (
|
|||||||
compat_str,
|
compat_str,
|
||||||
)
|
)
|
||||||
from ..utils import (
|
from ..utils import (
|
||||||
|
format_field,
|
||||||
int_or_none,
|
int_or_none,
|
||||||
parse_iso8601,
|
parse_iso8601,
|
||||||
smuggle_url,
|
smuggle_url,
|
||||||
@@ -43,7 +44,7 @@ class AWAANBaseIE(InfoExtractor):
|
|||||||
'id': video_id,
|
'id': video_id,
|
||||||
'title': title,
|
'title': title,
|
||||||
'description': video_data.get('description_en') or video_data.get('description_ar'),
|
'description': video_data.get('description_en') or video_data.get('description_ar'),
|
||||||
'thumbnail': 'http://admin.mangomolo.com/analytics/%s' % img if img else None,
|
'thumbnail': format_field(img, template='http://admin.mangomolo.com/analytics/%s'),
|
||||||
'duration': int_or_none(video_data.get('duration')),
|
'duration': int_or_none(video_data.get('duration')),
|
||||||
'timestamp': parse_iso8601(video_data.get('create_time'), ' '),
|
'timestamp': parse_iso8601(video_data.get('create_time'), ' '),
|
||||||
'is_live': is_live,
|
'is_live': is_live,
|
||||||
|
|||||||
+138
-60
@@ -1,5 +1,6 @@
|
|||||||
# coding: utf-8
|
# coding: utf-8
|
||||||
|
|
||||||
|
import base64
|
||||||
import hashlib
|
import hashlib
|
||||||
import itertools
|
import itertools
|
||||||
import functools
|
import functools
|
||||||
@@ -19,14 +20,15 @@ from ..utils import (
|
|||||||
parse_iso8601,
|
parse_iso8601,
|
||||||
traverse_obj,
|
traverse_obj,
|
||||||
try_get,
|
try_get,
|
||||||
|
parse_count,
|
||||||
smuggle_url,
|
smuggle_url,
|
||||||
srt_subtitles_timecode,
|
srt_subtitles_timecode,
|
||||||
str_or_none,
|
str_or_none,
|
||||||
str_to_int,
|
|
||||||
strip_jsonp,
|
strip_jsonp,
|
||||||
unified_timestamp,
|
unified_timestamp,
|
||||||
unsmuggle_url,
|
unsmuggle_url,
|
||||||
urlencode_postdata,
|
urlencode_postdata,
|
||||||
|
url_or_none,
|
||||||
OnDemandPagedList
|
OnDemandPagedList
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -722,40 +724,57 @@ class BiliBiliPlayerIE(InfoExtractor):
|
|||||||
|
|
||||||
|
|
||||||
class BiliIntlBaseIE(InfoExtractor):
|
class BiliIntlBaseIE(InfoExtractor):
|
||||||
_API_URL = 'https://api.bili{}/intl/gateway{}'
|
_API_URL = 'https://api.bilibili.tv/intl/gateway'
|
||||||
|
_NETRC_MACHINE = 'biliintl'
|
||||||
|
|
||||||
def _call_api(self, type, endpoint, id):
|
def _call_api(self, endpoint, *args, **kwargs):
|
||||||
return self._download_json(self._API_URL.format(type, endpoint), id)['data']
|
json = self._download_json(self._API_URL + endpoint, *args, **kwargs)
|
||||||
|
if json.get('code'):
|
||||||
|
if json['code'] in (10004004, 10004005, 10023006):
|
||||||
|
self.raise_login_required()
|
||||||
|
elif json['code'] == 10004001:
|
||||||
|
self.raise_geo_restricted()
|
||||||
|
else:
|
||||||
|
if json.get('message') and str(json['code']) != json['message']:
|
||||||
|
errmsg = f'{kwargs.get("errnote", "Unable to download JSON metadata")}: {self.IE_NAME} said: {json["message"]}'
|
||||||
|
else:
|
||||||
|
errmsg = kwargs.get('errnote', 'Unable to download JSON metadata')
|
||||||
|
if kwargs.get('fatal'):
|
||||||
|
raise ExtractorError(errmsg)
|
||||||
|
else:
|
||||||
|
self.report_warning(errmsg)
|
||||||
|
return json.get('data')
|
||||||
|
|
||||||
def json2srt(self, json):
|
def json2srt(self, json):
|
||||||
data = '\n\n'.join(
|
data = '\n\n'.join(
|
||||||
f'{i + 1}\n{srt_subtitles_timecode(line["from"])} --> {srt_subtitles_timecode(line["to"])}\n{line["content"]}'
|
f'{i + 1}\n{srt_subtitles_timecode(line["from"])} --> {srt_subtitles_timecode(line["to"])}\n{line["content"]}'
|
||||||
for i, line in enumerate(json['body']))
|
for i, line in enumerate(json['body']) if line.get('content'))
|
||||||
return data
|
return data
|
||||||
|
|
||||||
def _get_subtitles(self, type, ep_id):
|
def _get_subtitles(self, ep_id):
|
||||||
sub_json = self._call_api(type, f'/m/subtitle?ep_id={ep_id}&platform=web', ep_id)
|
sub_json = self._call_api(f'/web/v2/subtitle?episode_id={ep_id}&platform=web', ep_id)
|
||||||
subtitles = {}
|
subtitles = {}
|
||||||
for sub in sub_json.get('subtitles', []):
|
for sub in sub_json.get('subtitles') or []:
|
||||||
sub_url = sub.get('url')
|
sub_url = sub.get('url')
|
||||||
if not sub_url:
|
if not sub_url:
|
||||||
continue
|
continue
|
||||||
sub_data = self._download_json(sub_url, ep_id, fatal=False)
|
sub_data = self._download_json(
|
||||||
|
sub_url, ep_id, errnote='Unable to download subtitles', fatal=False,
|
||||||
|
note='Downloading subtitles%s' % f' for {sub["lang"]}' if sub.get('lang') else '')
|
||||||
if not sub_data:
|
if not sub_data:
|
||||||
continue
|
continue
|
||||||
subtitles.setdefault(sub.get('key', 'en'), []).append({
|
subtitles.setdefault(sub.get('lang_key', 'en'), []).append({
|
||||||
'ext': 'srt',
|
'ext': 'srt',
|
||||||
'data': self.json2srt(sub_data)
|
'data': self.json2srt(sub_data)
|
||||||
})
|
})
|
||||||
return subtitles
|
return subtitles
|
||||||
|
|
||||||
def _get_formats(self, type, ep_id):
|
def _get_formats(self, ep_id):
|
||||||
video_json = self._call_api(type, f'/web/playurl?ep_id={ep_id}&platform=web', ep_id)
|
video_json = self._call_api(f'/web/playurl?ep_id={ep_id}&platform=web', ep_id,
|
||||||
if not video_json:
|
note='Downloading video formats', errnote='Unable to download video formats')
|
||||||
self.raise_login_required(method='cookies')
|
|
||||||
video_json = video_json['playurl']
|
video_json = video_json['playurl']
|
||||||
formats = []
|
formats = []
|
||||||
for vid in video_json.get('video', []):
|
for vid in video_json.get('video') or []:
|
||||||
video_res = vid.get('video_resource') or {}
|
video_res = vid.get('video_resource') or {}
|
||||||
video_info = vid.get('stream_info') or {}
|
video_info = vid.get('stream_info') or {}
|
||||||
if not video_res.get('url'):
|
if not video_res.get('url'):
|
||||||
@@ -771,7 +790,7 @@ class BiliIntlBaseIE(InfoExtractor):
|
|||||||
'vcodec': video_res.get('codecs'),
|
'vcodec': video_res.get('codecs'),
|
||||||
'filesize': video_res.get('size'),
|
'filesize': video_res.get('size'),
|
||||||
})
|
})
|
||||||
for aud in video_json.get('audio_resource', []):
|
for aud in video_json.get('audio_resource') or []:
|
||||||
if not aud.get('url'):
|
if not aud.get('url'):
|
||||||
continue
|
continue
|
||||||
formats.append({
|
formats.append({
|
||||||
@@ -786,85 +805,144 @@ class BiliIntlBaseIE(InfoExtractor):
|
|||||||
self._sort_formats(formats)
|
self._sort_formats(formats)
|
||||||
return formats
|
return formats
|
||||||
|
|
||||||
def _extract_ep_info(self, type, episode_data, ep_id):
|
def _extract_ep_info(self, episode_data, ep_id):
|
||||||
return {
|
return {
|
||||||
'id': ep_id,
|
'id': ep_id,
|
||||||
'title': episode_data.get('long_title') or episode_data['title'],
|
'title': episode_data.get('title_display') or episode_data['title'],
|
||||||
'thumbnail': episode_data.get('cover'),
|
'thumbnail': episode_data.get('cover'),
|
||||||
'episode_number': str_to_int(episode_data.get('title')),
|
'episode_number': int_or_none(self._search_regex(
|
||||||
'formats': self._get_formats(type, ep_id),
|
r'^E(\d+)(?:$| - )', episode_data.get('title_display'), 'episode number', default=None)),
|
||||||
'subtitles': self._get_subtitles(type, ep_id),
|
'formats': self._get_formats(ep_id),
|
||||||
|
'subtitles': self._get_subtitles(ep_id),
|
||||||
'extractor_key': BiliIntlIE.ie_key(),
|
'extractor_key': BiliIntlIE.ie_key(),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
def _login(self):
|
||||||
|
username, password = self._get_login_info()
|
||||||
|
if username is None:
|
||||||
|
return
|
||||||
|
|
||||||
|
try:
|
||||||
|
from Cryptodome.PublicKey import RSA
|
||||||
|
from Cryptodome.Cipher import PKCS1_v1_5
|
||||||
|
except ImportError:
|
||||||
|
try:
|
||||||
|
from Crypto.PublicKey import RSA
|
||||||
|
from Crypto.Cipher import PKCS1_v1_5
|
||||||
|
except ImportError:
|
||||||
|
raise ExtractorError('pycryptodomex not found. Please install', expected=True)
|
||||||
|
|
||||||
|
key_data = self._download_json(
|
||||||
|
'https://passport.bilibili.tv/x/intl/passport-login/web/key?lang=en-US', None,
|
||||||
|
note='Downloading login key', errnote='Unable to download login key')['data']
|
||||||
|
|
||||||
|
public_key = RSA.importKey(key_data['key'])
|
||||||
|
password_hash = PKCS1_v1_5.new(public_key).encrypt((key_data['hash'] + password).encode('utf-8'))
|
||||||
|
login_post = self._download_json(
|
||||||
|
'https://passport.bilibili.tv/x/intl/passport-login/web/login/password?lang=en-US', None, data=urlencode_postdata({
|
||||||
|
'username': username,
|
||||||
|
'password': base64.b64encode(password_hash).decode('ascii'),
|
||||||
|
'keep_me': 'true',
|
||||||
|
's_locale': 'en_US',
|
||||||
|
'isTrusted': 'true'
|
||||||
|
}), note='Logging in', errnote='Unable to log in')
|
||||||
|
if login_post.get('code'):
|
||||||
|
if login_post.get('message'):
|
||||||
|
raise ExtractorError(f'Unable to log in: {self.IE_NAME} said: {login_post["message"]}', expected=True)
|
||||||
|
else:
|
||||||
|
raise ExtractorError('Unable to log in')
|
||||||
|
|
||||||
|
def _real_initialize(self):
|
||||||
|
self._login()
|
||||||
|
|
||||||
|
|
||||||
class BiliIntlIE(BiliIntlBaseIE):
|
class BiliIntlIE(BiliIntlBaseIE):
|
||||||
_VALID_URL = r'https?://(?:www\.)?bili(?P<type>bili\.tv|intl.com)/(?:[a-z]{2}/)?play/(?P<season_id>\d+)/(?P<id>\d+)'
|
_VALID_URL = r'https?://(?:www\.)?bili(?:bili\.tv|intl\.com)/(?:[a-z]{2}/)?play/(?P<season_id>\d+)/(?P<id>\d+)'
|
||||||
_TESTS = [{
|
_TESTS = [{
|
||||||
|
# Bstation page
|
||||||
'url': 'https://www.bilibili.tv/en/play/34613/341736',
|
'url': 'https://www.bilibili.tv/en/play/34613/341736',
|
||||||
'info_dict': {
|
'info_dict': {
|
||||||
'id': '341736',
|
'id': '341736',
|
||||||
'ext': 'mp4',
|
'ext': 'mp4',
|
||||||
'title': 'The First Night',
|
'title': 'E2 - The First Night',
|
||||||
'thumbnail': 'https://i0.hdslb.com/bfs/intl/management/91e30e5521235d9b163339a26a0b030ebda54310.png',
|
'thumbnail': r're:^https://pic\.bstarstatic\.com/ogv/.+\.png$',
|
||||||
'episode_number': 2,
|
'episode_number': 2,
|
||||||
|
}
|
||||||
|
}, {
|
||||||
|
# Non-Bstation page
|
||||||
|
'url': 'https://www.bilibili.tv/en/play/1033760/11005006',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '11005006',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'E3 - Who?',
|
||||||
|
'thumbnail': r're:^https://pic\.bstarstatic\.com/ogv/.+\.png$',
|
||||||
|
'episode_number': 3,
|
||||||
|
}
|
||||||
|
}, {
|
||||||
|
# Subtitle with empty content
|
||||||
|
'url': 'https://www.bilibili.tv/en/play/1005144/10131790',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '10131790',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'E140 - Two Heartbeats: Kabuto\'s Trap',
|
||||||
|
'thumbnail': r're:^https://pic\.bstarstatic\.com/ogv/.+\.png$',
|
||||||
|
'episode_number': 140,
|
||||||
},
|
},
|
||||||
'params': {
|
'skip': 'According to the copyright owner\'s request, you may only watch the video after you log in.'
|
||||||
'format': 'bv',
|
|
||||||
},
|
|
||||||
}, {
|
}, {
|
||||||
'url': 'https://www.biliintl.com/en/play/34613/341736',
|
'url': 'https://www.biliintl.com/en/play/34613/341736',
|
||||||
'info_dict': {
|
'only_matching': True,
|
||||||
'id': '341736',
|
|
||||||
'ext': 'mp4',
|
|
||||||
'title': 'The First Night',
|
|
||||||
'thumbnail': 'https://i0.hdslb.com/bfs/intl/management/91e30e5521235d9b163339a26a0b030ebda54310.png',
|
|
||||||
'episode_number': 2,
|
|
||||||
},
|
|
||||||
'params': {
|
|
||||||
'format': 'bv',
|
|
||||||
},
|
|
||||||
}]
|
}]
|
||||||
|
|
||||||
def _real_extract(self, url):
|
def _real_extract(self, url):
|
||||||
type, season_id, id = self._match_valid_url(url).groups()
|
season_id, video_id = self._match_valid_url(url).groups()
|
||||||
data_json = self._call_api(type, f'/web/view/ogv_collection?season_id={season_id}', id)
|
webpage = self._download_webpage(url, video_id)
|
||||||
episode_data = next(
|
# Bstation layout
|
||||||
episode for episode in data_json.get('episodes', [])
|
initial_data = self._parse_json(self._search_regex(
|
||||||
if str(episode.get('ep_id')) == id)
|
r'window\.__INITIAL_DATA__\s*=\s*({.+?});', webpage,
|
||||||
return self._extract_ep_info(type, episode_data, id)
|
'preload state', default='{}'), video_id, fatal=False) or {}
|
||||||
|
episode_data = traverse_obj(initial_data, ('OgvVideo', 'epDetail'), expected_type=dict)
|
||||||
|
|
||||||
|
if not episode_data:
|
||||||
|
# Non-Bstation layout, read through episode list
|
||||||
|
season_json = self._call_api(f'/web/v2/ogv/play/episodes?season_id={season_id}&platform=web', video_id)
|
||||||
|
episode_data = next(
|
||||||
|
episode for episode in traverse_obj(season_json, ('sections', ..., 'episodes', ...), expected_type=dict)
|
||||||
|
if str(episode.get('episode_id')) == video_id)
|
||||||
|
return self._extract_ep_info(episode_data, video_id)
|
||||||
|
|
||||||
|
|
||||||
class BiliIntlSeriesIE(BiliIntlBaseIE):
|
class BiliIntlSeriesIE(BiliIntlBaseIE):
|
||||||
_VALID_URL = r'https?://(?:www\.)?bili(?P<type>bili\.tv|intl.com)/(?:[a-z]{2}/)?play/(?P<id>\d+)$'
|
_VALID_URL = r'https?://(?:www\.)?bili(?:bili\.tv|intl\.com)/(?:[a-z]{2}/)?play/(?P<id>\d+)$'
|
||||||
_TESTS = [{
|
_TESTS = [{
|
||||||
'url': 'https://www.bilibili.tv/en/play/34613',
|
'url': 'https://www.bilibili.tv/en/play/34613',
|
||||||
'playlist_mincount': 15,
|
'playlist_mincount': 15,
|
||||||
'info_dict': {
|
'info_dict': {
|
||||||
'id': '34613',
|
'id': '34613',
|
||||||
|
'title': 'Fly Me to the Moon',
|
||||||
|
'description': 'md5:a861ee1c4dc0acfad85f557cc42ac627',
|
||||||
|
'categories': ['Romance', 'Comedy', 'Slice of life'],
|
||||||
|
'thumbnail': r're:^https://pic\.bstarstatic\.com/ogv/.+\.png$',
|
||||||
|
'view_count': int,
|
||||||
},
|
},
|
||||||
'params': {
|
'params': {
|
||||||
'skip_download': True,
|
'skip_download': True,
|
||||||
'format': 'bv',
|
|
||||||
},
|
},
|
||||||
}, {
|
}, {
|
||||||
'url': 'https://www.biliintl.com/en/play/34613',
|
'url': 'https://www.biliintl.com/en/play/34613',
|
||||||
'playlist_mincount': 15,
|
'only_matching': True,
|
||||||
'info_dict': {
|
|
||||||
'id': '34613',
|
|
||||||
},
|
|
||||||
'params': {
|
|
||||||
'skip_download': True,
|
|
||||||
'format': 'bv',
|
|
||||||
},
|
|
||||||
}]
|
}]
|
||||||
|
|
||||||
def _entries(self, id, type):
|
def _entries(self, series_id):
|
||||||
data_json = self._call_api(type, f'/web/view/ogv_collection?season_id={id}', id)
|
series_json = self._call_api(f'/web/v2/ogv/play/episodes?season_id={series_id}&platform=web', series_id)
|
||||||
for episode in data_json.get('episodes', []):
|
for episode in traverse_obj(series_json, ('sections', ..., 'episodes', ...), expected_type=dict, default=[]):
|
||||||
episode_id = str(episode.get('ep_id'))
|
episode_id = str(episode.get('episode_id'))
|
||||||
yield self._extract_ep_info(type, episode, episode_id)
|
yield self._extract_ep_info(episode, episode_id)
|
||||||
|
|
||||||
def _real_extract(self, url):
|
def _real_extract(self, url):
|
||||||
type, id = self._match_valid_url(url).groups()
|
series_id = self._match_id(url)
|
||||||
return self.playlist_result(self._entries(id, type), playlist_id=id)
|
series_info = self._call_api(f'/web/v2/ogv/play/season_info?season_id={series_id}&platform=web', series_id).get('season') or {}
|
||||||
|
return self.playlist_result(
|
||||||
|
self._entries(series_id), series_id, series_info.get('title'), series_info.get('description'),
|
||||||
|
categories=traverse_obj(series_info, ('styles', ..., 'title'), expected_type=str_or_none),
|
||||||
|
thumbnail=url_or_none(series_info.get('horizontal_cover')), view_count=parse_count(series_info.get('view')))
|
||||||
|
|||||||
@@ -0,0 +1,114 @@
|
|||||||
|
# coding: utf-8
|
||||||
|
from .common import InfoExtractor
|
||||||
|
from ..utils import (
|
||||||
|
traverse_obj,
|
||||||
|
float_or_none,
|
||||||
|
int_or_none
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class CallinIE(InfoExtractor):
|
||||||
|
_VALID_URL = r'https?://(?:www\.)?callin\.com/(episode)/(?P<id>[-a-zA-Z]+)'
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://www.callin.com/episode/the-title-ix-regime-and-the-long-march-through-EBfXYSrsjc',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '218b979630a35ead12c6fd096f2996c56c37e4d0dc1f6dc0feada32dcf7b31cd',
|
||||||
|
'title': 'The Title IX Regime and the Long March Through and Beyond the Institutions',
|
||||||
|
'ext': 'ts',
|
||||||
|
'display_id': 'the-title-ix-regime-and-the-long-march-through-EBfXYSrsjc',
|
||||||
|
'thumbnail': 're:https://.+\\.png',
|
||||||
|
'description': 'First episode',
|
||||||
|
'uploader': 'Wesley Yang',
|
||||||
|
'timestamp': 1639404128.65,
|
||||||
|
'upload_date': '20211213',
|
||||||
|
'uploader_id': 'wesyang',
|
||||||
|
'uploader_url': 'http://wesleyyang.substack.com',
|
||||||
|
'channel': 'Conversations in Year Zero',
|
||||||
|
'channel_id': '436d1f82ddeb30cd2306ea9156044d8d2cfdc3f1f1552d245117a42173e78553',
|
||||||
|
'channel_url': 'https://callin.com/show/conversations-in-year-zero-oJNllRFSfx',
|
||||||
|
'duration': 9951.936,
|
||||||
|
'view_count': int,
|
||||||
|
'categories': ['News & Politics', 'History', 'Technology'],
|
||||||
|
'cast': ['Wesley Yang', 'KC Johnson', 'Gabi Abramovich'],
|
||||||
|
'series': 'Conversations in Year Zero',
|
||||||
|
'series_id': '436d1f82ddeb30cd2306ea9156044d8d2cfdc3f1f1552d245117a42173e78553',
|
||||||
|
'episode': 'The Title IX Regime and the Long March Through and Beyond the Institutions',
|
||||||
|
'episode_number': 1,
|
||||||
|
'episode_id': '218b979630a35ead12c6fd096f2996c56c37e4d0dc1f6dc0feada32dcf7b31cd'
|
||||||
|
}
|
||||||
|
}]
|
||||||
|
|
||||||
|
def try_get_user_name(self, d):
|
||||||
|
names = [d.get(n) for n in ('first', 'last')]
|
||||||
|
if None in names:
|
||||||
|
return next((n for n in names if n), default=None)
|
||||||
|
return ' '.join(names)
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
display_id = self._match_id(url)
|
||||||
|
webpage = self._download_webpage(url, display_id)
|
||||||
|
|
||||||
|
next_data = self._search_nextjs_data(webpage, display_id)
|
||||||
|
episode = next_data['props']['pageProps']['episode']
|
||||||
|
|
||||||
|
id = episode['id']
|
||||||
|
title = (episode.get('title')
|
||||||
|
or self._og_search_title(webpage, fatal=False)
|
||||||
|
or self._html_search_regex('<title>(.*?)</title>', webpage, 'title'))
|
||||||
|
url = episode['m3u8']
|
||||||
|
formats = self._extract_m3u8_formats(url, display_id, ext='ts')
|
||||||
|
self._sort_formats(formats)
|
||||||
|
|
||||||
|
show = traverse_obj(episode, ('show', 'title'))
|
||||||
|
show_id = traverse_obj(episode, ('show', 'id'))
|
||||||
|
|
||||||
|
show_json = None
|
||||||
|
app_slug = (self._html_search_regex(
|
||||||
|
'<script\\s+src=["\']/_next/static/([-_a-zA-Z0-9]+)/_',
|
||||||
|
webpage, 'app slug', fatal=False) or next_data.get('buildId'))
|
||||||
|
show_slug = traverse_obj(episode, ('show', 'linkObj', 'resourceUrl'))
|
||||||
|
if app_slug and show_slug and '/' in show_slug:
|
||||||
|
show_slug = show_slug.rsplit('/', 1)[1]
|
||||||
|
show_json_url = f'https://www.callin.com/_next/data/{app_slug}/show/{show_slug}.json'
|
||||||
|
show_json = self._download_json(show_json_url, display_id, fatal=False)
|
||||||
|
|
||||||
|
host = (traverse_obj(show_json, ('pageProps', 'show', 'hosts', 0))
|
||||||
|
or traverse_obj(episode, ('speakers', 0)))
|
||||||
|
|
||||||
|
host_nick = traverse_obj(host, ('linkObj', 'resourceUrl'))
|
||||||
|
host_nick = host_nick.rsplit('/', 1)[1] if (host_nick and '/' in host_nick) else None
|
||||||
|
|
||||||
|
cast = list(filter(None, [
|
||||||
|
self.try_get_user_name(u) for u in
|
||||||
|
traverse_obj(episode, (('speakers', 'callerTags'), ...)) or []
|
||||||
|
]))
|
||||||
|
|
||||||
|
episode_list = traverse_obj(show_json, ('pageProps', 'show', 'episodes')) or []
|
||||||
|
episode_number = next(
|
||||||
|
(len(episode_list) - i for (i, e) in enumerate(episode_list) if e.get('id') == id),
|
||||||
|
None)
|
||||||
|
|
||||||
|
return {
|
||||||
|
'id': id,
|
||||||
|
'display_id': display_id,
|
||||||
|
'title': title,
|
||||||
|
'formats': formats,
|
||||||
|
'thumbnail': traverse_obj(episode, ('show', 'photo')),
|
||||||
|
'description': episode.get('description'),
|
||||||
|
'uploader': self.try_get_user_name(host) if host else None,
|
||||||
|
'timestamp': episode.get('publishedAt'),
|
||||||
|
'uploader_id': host_nick,
|
||||||
|
'uploader_url': traverse_obj(show_json, ('pageProps', 'show', 'url')),
|
||||||
|
'channel': show,
|
||||||
|
'channel_id': show_id,
|
||||||
|
'channel_url': traverse_obj(episode, ('show', 'linkObj', 'resourceUrl')),
|
||||||
|
'duration': float_or_none(episode.get('runtime')),
|
||||||
|
'view_count': int_or_none(episode.get('plays')),
|
||||||
|
'categories': traverse_obj(episode, ('show', 'categorizations', ..., 'name')),
|
||||||
|
'cast': cast if cast else None,
|
||||||
|
'series': show,
|
||||||
|
'series_id': show_id,
|
||||||
|
'episode': title,
|
||||||
|
'episode_number': episode_number,
|
||||||
|
'episode_id': id
|
||||||
|
}
|
||||||
@@ -78,11 +78,11 @@ class CanalAlphaIE(InfoExtractor):
|
|||||||
'height': try_get(video, lambda x: x['res']['height'], expected_type=int),
|
'height': try_get(video, lambda x: x['res']['height'], expected_type=int),
|
||||||
} for video in try_get(data_json, lambda x: x['video']['mp4'], expected_type=list) or [] if video.get('$url')]
|
} for video in try_get(data_json, lambda x: x['video']['mp4'], expected_type=list) or [] if video.get('$url')]
|
||||||
if manifests.get('hls'):
|
if manifests.get('hls'):
|
||||||
m3u8_frmts, m3u8_subs = self._parse_m3u8_formats_and_subtitles(manifests['hls'], id)
|
m3u8_frmts, m3u8_subs = self._parse_m3u8_formats_and_subtitles(manifests['hls'], video_id=id)
|
||||||
formats.extend(m3u8_frmts)
|
formats.extend(m3u8_frmts)
|
||||||
subtitles = self._merge_subtitles(subtitles, m3u8_subs)
|
subtitles = self._merge_subtitles(subtitles, m3u8_subs)
|
||||||
if manifests.get('dash'):
|
if manifests.get('dash'):
|
||||||
dash_frmts, dash_subs = self._parse_mpd_formats_and_subtitles(manifests['dash'], id)
|
dash_frmts, dash_subs = self._parse_mpd_formats_and_subtitles(manifests['dash'])
|
||||||
formats.extend(dash_frmts)
|
formats.extend(dash_frmts)
|
||||||
subtitles = self._merge_subtitles(subtitles, dash_subs)
|
subtitles = self._merge_subtitles(subtitles, dash_subs)
|
||||||
self._sort_formats(formats)
|
self._sort_formats(formats)
|
||||||
|
|||||||
@@ -76,7 +76,7 @@ class CanvasIE(InfoExtractor):
|
|||||||
'vrtPlayerToken': vrtPlayerToken,
|
'vrtPlayerToken': vrtPlayerToken,
|
||||||
'client': 'null',
|
'client': 'null',
|
||||||
}, expected_status=400)
|
}, expected_status=400)
|
||||||
if not data.get('title'):
|
if 'title' not in data:
|
||||||
code = data.get('code')
|
code = data.get('code')
|
||||||
if code == 'AUTHENTICATION_REQUIRED':
|
if code == 'AUTHENTICATION_REQUIRED':
|
||||||
self.raise_login_required()
|
self.raise_login_required()
|
||||||
@@ -84,7 +84,8 @@ class CanvasIE(InfoExtractor):
|
|||||||
self.raise_geo_restricted(countries=['BE'])
|
self.raise_geo_restricted(countries=['BE'])
|
||||||
raise ExtractorError(data.get('message') or code, expected=True)
|
raise ExtractorError(data.get('message') or code, expected=True)
|
||||||
|
|
||||||
title = data['title']
|
# Note: The title may be an empty string
|
||||||
|
title = data['title'] or f'{site_id} {video_id}'
|
||||||
description = data.get('description')
|
description = data.get('description')
|
||||||
|
|
||||||
formats = []
|
formats = []
|
||||||
|
|||||||
@@ -4,6 +4,7 @@ from __future__ import unicode_literals
|
|||||||
from .common import InfoExtractor
|
from .common import InfoExtractor
|
||||||
from ..compat import compat_str
|
from ..compat import compat_str
|
||||||
from ..utils import (
|
from ..utils import (
|
||||||
|
format_field,
|
||||||
float_or_none,
|
float_or_none,
|
||||||
int_or_none,
|
int_or_none,
|
||||||
try_get,
|
try_get,
|
||||||
@@ -43,7 +44,7 @@ class CarambaTVIE(InfoExtractor):
|
|||||||
formats = [{
|
formats = [{
|
||||||
'url': base_url + f['fn'],
|
'url': base_url + f['fn'],
|
||||||
'height': int_or_none(f.get('height')),
|
'height': int_or_none(f.get('height')),
|
||||||
'format_id': '%sp' % f['height'] if f.get('height') else None,
|
'format_id': format_field(f, 'height', '%sp'),
|
||||||
} for f in video['qualities'] if f.get('fn')]
|
} for f in video['qualities'] if f.get('fn')]
|
||||||
self._sort_formats(formats)
|
self._sort_formats(formats)
|
||||||
|
|
||||||
|
|||||||
@@ -340,7 +340,8 @@ class CBCGemIE(InfoExtractor):
|
|||||||
yield {
|
yield {
|
||||||
**base_format,
|
**base_format,
|
||||||
'format_id': join_nonempty('sec', height),
|
'format_id': join_nonempty('sec', height),
|
||||||
'url': re.sub(r'(QualityLevels\()\d+(\))', fr'\1{bitrate}\2', base_url),
|
# Note: \g<1> is necessary instead of \1 since bitrate is a number
|
||||||
|
'url': re.sub(r'(QualityLevels\()\d+(\))', fr'\g<1>{bitrate}\2', base_url),
|
||||||
'width': int_or_none(video_quality.attrib.get('MaxWidth')),
|
'width': int_or_none(video_quality.attrib.get('MaxWidth')),
|
||||||
'tbr': bitrate / 1000.0,
|
'tbr': bitrate / 1000.0,
|
||||||
'height': height,
|
'height': height,
|
||||||
|
|||||||
@@ -177,6 +177,7 @@ class CeskaTelevizeIE(InfoExtractor):
|
|||||||
is_live = item.get('type') == 'LIVE'
|
is_live = item.get('type') == 'LIVE'
|
||||||
formats = []
|
formats = []
|
||||||
for format_id, stream_url in item.get('streamUrls', {}).items():
|
for format_id, stream_url in item.get('streamUrls', {}).items():
|
||||||
|
stream_url = stream_url.replace('https://', 'http://')
|
||||||
if 'playerType=flash' in stream_url:
|
if 'playerType=flash' in stream_url:
|
||||||
stream_formats = self._extract_m3u8_formats(
|
stream_formats = self._extract_m3u8_formats(
|
||||||
stream_url, playlist_id, 'mp4', 'm3u8_native',
|
stream_url, playlist_id, 'mp4', 'm3u8_native',
|
||||||
|
|||||||
+71
-37
@@ -45,6 +45,7 @@ from ..utils import (
|
|||||||
determine_ext,
|
determine_ext,
|
||||||
determine_protocol,
|
determine_protocol,
|
||||||
dict_get,
|
dict_get,
|
||||||
|
encode_data_uri,
|
||||||
error_to_compat_str,
|
error_to_compat_str,
|
||||||
extract_attributes,
|
extract_attributes,
|
||||||
ExtractorError,
|
ExtractorError,
|
||||||
@@ -243,11 +244,16 @@ class InfoExtractor(object):
|
|||||||
uploader: Full name of the video uploader.
|
uploader: Full name of the video uploader.
|
||||||
license: License name the video is licensed under.
|
license: License name the video is licensed under.
|
||||||
creator: The creator of the video.
|
creator: The creator of the video.
|
||||||
release_timestamp: UNIX timestamp of the moment the video was released.
|
|
||||||
release_date: The date (YYYYMMDD) when the video was released.
|
|
||||||
timestamp: UNIX timestamp of the moment the video was uploaded
|
timestamp: UNIX timestamp of the moment the video was uploaded
|
||||||
upload_date: Video upload date (YYYYMMDD).
|
upload_date: Video upload date (YYYYMMDD).
|
||||||
If not explicitly set, calculated from timestamp.
|
If not explicitly set, calculated from timestamp
|
||||||
|
release_timestamp: UNIX timestamp of the moment the video was released.
|
||||||
|
If it is not clear whether to use timestamp or this, use the former
|
||||||
|
release_date: The date (YYYYMMDD) when the video was released.
|
||||||
|
If not explicitly set, calculated from release_timestamp
|
||||||
|
modified_timestamp: UNIX timestamp of the moment the video was last modified.
|
||||||
|
modified_date: The date (YYYYMMDD) when the video was last modified.
|
||||||
|
If not explicitly set, calculated from modified_timestamp
|
||||||
uploader_id: Nickname or id of the video uploader.
|
uploader_id: Nickname or id of the video uploader.
|
||||||
uploader_url: Full URL to a personal webpage of the video uploader.
|
uploader_url: Full URL to a personal webpage of the video uploader.
|
||||||
channel: Full name of the channel the video is uploaded on.
|
channel: Full name of the channel the video is uploaded on.
|
||||||
@@ -255,6 +261,7 @@ class InfoExtractor(object):
|
|||||||
fields. This depends on a particular extractor.
|
fields. This depends on a particular extractor.
|
||||||
channel_id: Id of the channel.
|
channel_id: Id of the channel.
|
||||||
channel_url: Full URL to a channel webpage.
|
channel_url: Full URL to a channel webpage.
|
||||||
|
channel_follower_count: Number of followers of the channel.
|
||||||
location: Physical location where the video was filmed.
|
location: Physical location where the video was filmed.
|
||||||
subtitles: The available subtitles as a dictionary in the format
|
subtitles: The available subtitles as a dictionary in the format
|
||||||
{tag: subformats}. "tag" is usually a language code, and
|
{tag: subformats}. "tag" is usually a language code, and
|
||||||
@@ -370,6 +377,7 @@ class InfoExtractor(object):
|
|||||||
disc_number: Number of the disc or other physical medium the track belongs to,
|
disc_number: Number of the disc or other physical medium the track belongs to,
|
||||||
as an integer.
|
as an integer.
|
||||||
release_year: Year (YYYY) when the album was released.
|
release_year: Year (YYYY) when the album was released.
|
||||||
|
composer: Composer of the piece
|
||||||
|
|
||||||
Unless mentioned otherwise, the fields should be Unicode strings.
|
Unless mentioned otherwise, the fields should be Unicode strings.
|
||||||
|
|
||||||
@@ -383,6 +391,11 @@ class InfoExtractor(object):
|
|||||||
Additionally, playlists can have "id", "title", and any other relevent
|
Additionally, playlists can have "id", "title", and any other relevent
|
||||||
attributes with the same semantics as videos (see above).
|
attributes with the same semantics as videos (see above).
|
||||||
|
|
||||||
|
It can also have the following optional fields:
|
||||||
|
|
||||||
|
playlist_count: The total number of videos in a playlist. If not given,
|
||||||
|
YoutubeDL tries to calculate it from "entries"
|
||||||
|
|
||||||
|
|
||||||
_type "multi_video" indicates that there are multiple videos that
|
_type "multi_video" indicates that there are multiple videos that
|
||||||
form a single show, for examples multiple acts of an opera or TV episode.
|
form a single show, for examples multiple acts of an opera or TV episode.
|
||||||
@@ -1108,39 +1121,39 @@ class InfoExtractor(object):
|
|||||||
|
|
||||||
# Methods for following #608
|
# Methods for following #608
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def url_result(url, ie=None, video_id=None, video_title=None, **kwargs):
|
def url_result(url, ie=None, video_id=None, video_title=None, *, url_transparent=False, **kwargs):
|
||||||
"""Returns a URL that points to a page that should be processed"""
|
"""Returns a URL that points to a page that should be processed"""
|
||||||
# TODO: ie should be the class used for getting the info
|
if ie is not None:
|
||||||
video_info = {'_type': 'url',
|
kwargs['ie_key'] = ie if isinstance(ie, str) else ie.ie_key()
|
||||||
'url': url,
|
|
||||||
'ie_key': ie}
|
|
||||||
video_info.update(kwargs)
|
|
||||||
if video_id is not None:
|
if video_id is not None:
|
||||||
video_info['id'] = video_id
|
kwargs['id'] = video_id
|
||||||
if video_title is not None:
|
if video_title is not None:
|
||||||
video_info['title'] = video_title
|
kwargs['title'] = video_title
|
||||||
return video_info
|
return {
|
||||||
|
**kwargs,
|
||||||
|
'_type': 'url_transparent' if url_transparent else 'url',
|
||||||
|
'url': url,
|
||||||
|
}
|
||||||
|
|
||||||
def playlist_from_matches(self, matches, playlist_id=None, playlist_title=None, getter=None, ie=None):
|
def playlist_from_matches(self, matches, playlist_id=None, playlist_title=None, getter=None, ie=None, **kwargs):
|
||||||
urls = orderedSet(
|
urls = (self.url_result(self._proto_relative_url(m), ie)
|
||||||
self.url_result(self._proto_relative_url(getter(m) if getter else m), ie)
|
for m in orderedSet(map(getter, matches) if getter else matches))
|
||||||
for m in matches)
|
return self.playlist_result(urls, playlist_id, playlist_title, **kwargs)
|
||||||
return self.playlist_result(
|
|
||||||
urls, playlist_id=playlist_id, playlist_title=playlist_title)
|
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def playlist_result(entries, playlist_id=None, playlist_title=None, playlist_description=None, **kwargs):
|
def playlist_result(entries, playlist_id=None, playlist_title=None, playlist_description=None, *, multi_video=False, **kwargs):
|
||||||
"""Returns a playlist"""
|
"""Returns a playlist"""
|
||||||
video_info = {'_type': 'playlist',
|
|
||||||
'entries': entries}
|
|
||||||
video_info.update(kwargs)
|
|
||||||
if playlist_id:
|
if playlist_id:
|
||||||
video_info['id'] = playlist_id
|
kwargs['id'] = playlist_id
|
||||||
if playlist_title:
|
if playlist_title:
|
||||||
video_info['title'] = playlist_title
|
kwargs['title'] = playlist_title
|
||||||
if playlist_description is not None:
|
if playlist_description is not None:
|
||||||
video_info['description'] = playlist_description
|
kwargs['description'] = playlist_description
|
||||||
return video_info
|
return {
|
||||||
|
**kwargs,
|
||||||
|
'_type': 'multi_video' if multi_video else 'playlist',
|
||||||
|
'entries': entries,
|
||||||
|
}
|
||||||
|
|
||||||
def _search_regex(self, pattern, string, name, default=NO_DEFAULT, fatal=True, flags=0, group=None):
|
def _search_regex(self, pattern, string, name, default=NO_DEFAULT, fatal=True, flags=0, group=None):
|
||||||
"""
|
"""
|
||||||
@@ -1429,6 +1442,23 @@ class InfoExtractor(object):
|
|||||||
continue
|
continue
|
||||||
info[count_key] = interaction_count
|
info[count_key] = interaction_count
|
||||||
|
|
||||||
|
def extract_chapter_information(e):
|
||||||
|
chapters = [{
|
||||||
|
'title': part.get('name'),
|
||||||
|
'start_time': part.get('startOffset'),
|
||||||
|
'end_time': part.get('endOffset'),
|
||||||
|
} for part in e.get('hasPart', []) if part.get('@type') == 'Clip']
|
||||||
|
for idx, (last_c, current_c, next_c) in enumerate(zip(
|
||||||
|
[{'end_time': 0}] + chapters, chapters, chapters[1:])):
|
||||||
|
current_c['end_time'] = current_c['end_time'] or next_c['start_time']
|
||||||
|
current_c['start_time'] = current_c['start_time'] or last_c['end_time']
|
||||||
|
if None in current_c.values():
|
||||||
|
self.report_warning(f'Chapter {idx} contains broken data. Not extracting chapters')
|
||||||
|
return
|
||||||
|
if chapters:
|
||||||
|
chapters[-1]['end_time'] = chapters[-1]['end_time'] or info['duration']
|
||||||
|
info['chapters'] = chapters
|
||||||
|
|
||||||
def extract_video_object(e):
|
def extract_video_object(e):
|
||||||
assert e['@type'] == 'VideoObject'
|
assert e['@type'] == 'VideoObject'
|
||||||
author = e.get('author')
|
author = e.get('author')
|
||||||
@@ -1436,7 +1466,8 @@ class InfoExtractor(object):
|
|||||||
'url': url_or_none(e.get('contentUrl')),
|
'url': url_or_none(e.get('contentUrl')),
|
||||||
'title': unescapeHTML(e.get('name')),
|
'title': unescapeHTML(e.get('name')),
|
||||||
'description': unescapeHTML(e.get('description')),
|
'description': unescapeHTML(e.get('description')),
|
||||||
'thumbnail': url_or_none(e.get('thumbnailUrl') or e.get('thumbnailURL')),
|
'thumbnails': [{'url': url_or_none(url)}
|
||||||
|
for url in variadic(traverse_obj(e, 'thumbnailUrl', 'thumbnailURL'))],
|
||||||
'duration': parse_duration(e.get('duration')),
|
'duration': parse_duration(e.get('duration')),
|
||||||
'timestamp': unified_timestamp(e.get('uploadDate')),
|
'timestamp': unified_timestamp(e.get('uploadDate')),
|
||||||
# author can be an instance of 'Organization' or 'Person' types.
|
# author can be an instance of 'Organization' or 'Person' types.
|
||||||
@@ -1451,6 +1482,7 @@ class InfoExtractor(object):
|
|||||||
'view_count': int_or_none(e.get('interactionCount')),
|
'view_count': int_or_none(e.get('interactionCount')),
|
||||||
})
|
})
|
||||||
extract_interaction_statistic(e)
|
extract_interaction_statistic(e)
|
||||||
|
extract_chapter_information(e)
|
||||||
|
|
||||||
def traverse_json_ld(json_ld, at_top_level=True):
|
def traverse_json_ld(json_ld, at_top_level=True):
|
||||||
for e in json_ld:
|
for e in json_ld:
|
||||||
@@ -1513,12 +1545,12 @@ class InfoExtractor(object):
|
|||||||
|
|
||||||
return dict((k, v) for k, v in info.items() if v is not None)
|
return dict((k, v) for k, v in info.items() if v is not None)
|
||||||
|
|
||||||
def _search_nextjs_data(self, webpage, video_id, **kw):
|
def _search_nextjs_data(self, webpage, video_id, *, transform_source=None, fatal=True, **kw):
|
||||||
return self._parse_json(
|
return self._parse_json(
|
||||||
self._search_regex(
|
self._search_regex(
|
||||||
r'(?s)<script[^>]+id=[\'"]__NEXT_DATA__[\'"][^>]*>([^<]+)</script>',
|
r'(?s)<script[^>]+id=[\'"]__NEXT_DATA__[\'"][^>]*>([^<]+)</script>',
|
||||||
webpage, 'next.js data', **kw),
|
webpage, 'next.js data', fatal=fatal, **kw),
|
||||||
video_id, **kw)
|
video_id, transform_source=transform_source, fatal=fatal)
|
||||||
|
|
||||||
def _search_nuxt_data(self, webpage, video_id, context_name='__NUXT__'):
|
def _search_nuxt_data(self, webpage, video_id, context_name='__NUXT__'):
|
||||||
''' Parses Nuxt.js metadata. This works as long as the function __NUXT__ invokes is a pure function. '''
|
''' Parses Nuxt.js metadata. This works as long as the function __NUXT__ invokes is a pure function. '''
|
||||||
@@ -2076,7 +2108,7 @@ class InfoExtractor(object):
|
|||||||
headers=headers, query=query, video_id=video_id)
|
headers=headers, query=query, video_id=video_id)
|
||||||
|
|
||||||
def _parse_m3u8_formats_and_subtitles(
|
def _parse_m3u8_formats_and_subtitles(
|
||||||
self, m3u8_doc, m3u8_url, ext=None, entry_protocol='m3u8_native',
|
self, m3u8_doc, m3u8_url=None, ext=None, entry_protocol='m3u8_native',
|
||||||
preference=None, quality=None, m3u8_id=None, live=False, note=None,
|
preference=None, quality=None, m3u8_id=None, live=False, note=None,
|
||||||
errnote=None, fatal=True, data=None, headers={}, query={},
|
errnote=None, fatal=True, data=None, headers={}, query={},
|
||||||
video_id=None):
|
video_id=None):
|
||||||
@@ -2126,7 +2158,7 @@ class InfoExtractor(object):
|
|||||||
formats = [{
|
formats = [{
|
||||||
'format_id': join_nonempty(m3u8_id, idx),
|
'format_id': join_nonempty(m3u8_id, idx),
|
||||||
'format_index': idx,
|
'format_index': idx,
|
||||||
'url': m3u8_url,
|
'url': m3u8_url or encode_data_uri(m3u8_doc.encode('utf-8'), 'application/x-mpegurl'),
|
||||||
'ext': ext,
|
'ext': ext,
|
||||||
'protocol': entry_protocol,
|
'protocol': entry_protocol,
|
||||||
'preference': preference,
|
'preference': preference,
|
||||||
@@ -2712,11 +2744,15 @@ class InfoExtractor(object):
|
|||||||
mime_type = representation_attrib['mimeType']
|
mime_type = representation_attrib['mimeType']
|
||||||
content_type = representation_attrib.get('contentType', mime_type.split('/')[0])
|
content_type = representation_attrib.get('contentType', mime_type.split('/')[0])
|
||||||
|
|
||||||
codecs = representation_attrib.get('codecs', '')
|
codecs = parse_codecs(representation_attrib.get('codecs', ''))
|
||||||
if content_type not in ('video', 'audio', 'text'):
|
if content_type not in ('video', 'audio', 'text'):
|
||||||
if mime_type == 'image/jpeg':
|
if mime_type == 'image/jpeg':
|
||||||
content_type = mime_type
|
content_type = mime_type
|
||||||
elif codecs.split('.')[0] == 'stpp':
|
elif codecs['vcodec'] != 'none':
|
||||||
|
content_type = 'video'
|
||||||
|
elif codecs['acodec'] != 'none':
|
||||||
|
content_type = 'audio'
|
||||||
|
elif codecs.get('tcodec', 'none') != 'none':
|
||||||
content_type = 'text'
|
content_type = 'text'
|
||||||
elif mimetype2ext(mime_type) in ('tt', 'dfxp', 'ttml', 'xml', 'json'):
|
elif mimetype2ext(mime_type) in ('tt', 'dfxp', 'ttml', 'xml', 'json'):
|
||||||
content_type = 'text'
|
content_type = 'text'
|
||||||
@@ -2762,8 +2798,8 @@ class InfoExtractor(object):
|
|||||||
'format_note': 'DASH %s' % content_type,
|
'format_note': 'DASH %s' % content_type,
|
||||||
'filesize': filesize,
|
'filesize': filesize,
|
||||||
'container': mimetype2ext(mime_type) + '_dash',
|
'container': mimetype2ext(mime_type) + '_dash',
|
||||||
|
**codecs
|
||||||
}
|
}
|
||||||
f.update(parse_codecs(codecs))
|
|
||||||
elif content_type == 'text':
|
elif content_type == 'text':
|
||||||
f = {
|
f = {
|
||||||
'ext': mimetype2ext(mime_type),
|
'ext': mimetype2ext(mime_type),
|
||||||
@@ -3468,8 +3504,6 @@ class InfoExtractor(object):
|
|||||||
|
|
||||||
def _int(self, v, name, fatal=False, **kwargs):
|
def _int(self, v, name, fatal=False, **kwargs):
|
||||||
res = int_or_none(v, **kwargs)
|
res = int_or_none(v, **kwargs)
|
||||||
if 'get_attr' in kwargs:
|
|
||||||
print(getattr(v, kwargs['get_attr']))
|
|
||||||
if res is None:
|
if res is None:
|
||||||
msg = 'Failed to extract %s: Could not parse value %r' % (name, v)
|
msg = 'Failed to extract %s: Could not parse value %r' % (name, v)
|
||||||
if fatal:
|
if fatal:
|
||||||
|
|||||||
@@ -0,0 +1,113 @@
|
|||||||
|
# coding: utf-8
|
||||||
|
from __future__ import unicode_literals
|
||||||
|
|
||||||
|
import itertools
|
||||||
|
|
||||||
|
from .common import InfoExtractor
|
||||||
|
from ..utils import (
|
||||||
|
int_or_none,
|
||||||
|
try_get,
|
||||||
|
unified_strdate,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class CrowdBunkerIE(InfoExtractor):
|
||||||
|
_VALID_URL = r'https?://(?:www\.)?crowdbunker\.com/v/(?P<id>[^/?#$&]+)'
|
||||||
|
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://crowdbunker.com/v/0z4Kms8pi8I',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '0z4Kms8pi8I',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': '117) Pass vax et solutions',
|
||||||
|
'description': 'md5:86bcb422c29475dbd2b5dcfa6ec3749c',
|
||||||
|
'view_count': int,
|
||||||
|
'duration': 5386,
|
||||||
|
'uploader': 'Jérémie Mercier',
|
||||||
|
'uploader_id': 'UCeN_qQV829NYf0pvPJhW5dQ',
|
||||||
|
'like_count': int,
|
||||||
|
'upload_date': '20211218',
|
||||||
|
'thumbnail': 'https://scw.divulg.org/cb-medias4/images/0z4Kms8pi8I/maxres.jpg'
|
||||||
|
},
|
||||||
|
'params': {'skip_download': True}
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
id = self._match_id(url)
|
||||||
|
data_json = self._download_json(f'https://api.divulg.org/post/{id}/details',
|
||||||
|
id, headers={'accept': 'application/json, text/plain, */*'})
|
||||||
|
video_json = data_json['video']
|
||||||
|
formats, subtitles = [], {}
|
||||||
|
for sub in video_json.get('captions') or []:
|
||||||
|
sub_url = try_get(sub, lambda x: x['file']['url'])
|
||||||
|
if not sub_url:
|
||||||
|
continue
|
||||||
|
subtitles.setdefault(sub.get('languageCode', 'fr'), []).append({
|
||||||
|
'url': sub_url,
|
||||||
|
})
|
||||||
|
|
||||||
|
mpd_url = try_get(video_json, lambda x: x['dashManifest']['url'])
|
||||||
|
if mpd_url:
|
||||||
|
fmts, subs = self._extract_mpd_formats_and_subtitles(mpd_url, id)
|
||||||
|
formats.extend(fmts)
|
||||||
|
subtitles = self._merge_subtitles(subtitles, subs)
|
||||||
|
m3u8_url = try_get(video_json, lambda x: x['hlsManifest']['url'])
|
||||||
|
if m3u8_url:
|
||||||
|
fmts, subs = self._extract_m3u8_formats_and_subtitles(mpd_url, id)
|
||||||
|
formats.extend(fmts)
|
||||||
|
subtitles = self._merge_subtitles(subtitles, subs)
|
||||||
|
|
||||||
|
thumbnails = [{
|
||||||
|
'url': image['url'],
|
||||||
|
'height': int_or_none(image.get('height')),
|
||||||
|
'width': int_or_none(image.get('width')),
|
||||||
|
} for image in video_json.get('thumbnails') or [] if image.get('url')]
|
||||||
|
|
||||||
|
self._sort_formats(formats)
|
||||||
|
return {
|
||||||
|
'id': id,
|
||||||
|
'title': video_json.get('title'),
|
||||||
|
'description': video_json.get('description'),
|
||||||
|
'view_count': video_json.get('viewCount'),
|
||||||
|
'duration': video_json.get('duration'),
|
||||||
|
'uploader': try_get(data_json, lambda x: x['channel']['name']),
|
||||||
|
'uploader_id': try_get(data_json, lambda x: x['channel']['id']),
|
||||||
|
'like_count': data_json.get('likesCount'),
|
||||||
|
'upload_date': unified_strdate(video_json.get('publishedAt') or video_json.get('createdAt')),
|
||||||
|
'thumbnails': thumbnails,
|
||||||
|
'formats': formats,
|
||||||
|
'subtitles': subtitles,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class CrowdBunkerChannelIE(InfoExtractor):
|
||||||
|
_VALID_URL = r'https?://(?:www\.)?crowdbunker\.com/@(?P<id>[^/?#$&]+)'
|
||||||
|
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://crowdbunker.com/@Milan_UHRIN',
|
||||||
|
'playlist_mincount': 14,
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'Milan_UHRIN',
|
||||||
|
},
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _entries(self, id):
|
||||||
|
last = None
|
||||||
|
|
||||||
|
for page in itertools.count():
|
||||||
|
channel_json = self._download_json(
|
||||||
|
f'https://api.divulg.org/organization/{id}/posts', id, headers={'accept': 'application/json, text/plain, */*'},
|
||||||
|
query={'after': last} if last else {}, note=f'Downloading Page {page}')
|
||||||
|
for item in channel_json.get('items') or []:
|
||||||
|
v_id = item.get('uid')
|
||||||
|
if not v_id:
|
||||||
|
continue
|
||||||
|
yield self.url_result(
|
||||||
|
'https://crowdbunker.com/v/%s' % v_id, ie=CrowdBunkerIE.ie_key(), video_id=v_id)
|
||||||
|
last = channel_json.get('last')
|
||||||
|
if not last:
|
||||||
|
break
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
id = self._match_id(url)
|
||||||
|
return self.playlist_result(self._entries(id), playlist_id=id)
|
||||||
@@ -65,4 +65,9 @@ class CTVNewsIE(InfoExtractor):
|
|||||||
})
|
})
|
||||||
entries = [ninecninemedia_url_result(clip_id) for clip_id in orderedSet(
|
entries = [ninecninemedia_url_result(clip_id) for clip_id in orderedSet(
|
||||||
re.findall(r'clip\.id\s*=\s*(\d+);', webpage))]
|
re.findall(r'clip\.id\s*=\s*(\d+);', webpage))]
|
||||||
|
if not entries:
|
||||||
|
webpage = self._download_webpage(url, page_id)
|
||||||
|
if 'getAuthStates("' in webpage:
|
||||||
|
entries = [ninecninemedia_url_result(clip_id) for clip_id in
|
||||||
|
self._search_regex(r'getAuthStates\("([\d+,]+)"', webpage, 'clip ids').split(',')]
|
||||||
return self.playlist_result(entries, page_id)
|
return self.playlist_result(entries, page_id)
|
||||||
|
|||||||
@@ -0,0 +1,79 @@
|
|||||||
|
# coding: utf-8
|
||||||
|
from __future__ import unicode_literals
|
||||||
|
|
||||||
|
from .common import InfoExtractor
|
||||||
|
from ..compat import compat_b64decode
|
||||||
|
from ..utils import (
|
||||||
|
get_elements_by_class,
|
||||||
|
int_or_none,
|
||||||
|
js_to_json,
|
||||||
|
parse_count,
|
||||||
|
parse_duration,
|
||||||
|
try_get,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class DaftsexIE(InfoExtractor):
|
||||||
|
_VALID_URL = r'https?://(?:www\.)?daftsex\.com/watch/(?P<id>-?\d+_\d+)'
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://daftsex.com/watch/-156601359_456242791',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '-156601359_456242791',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'Skye Blue - Dinner And A Show',
|
||||||
|
},
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
video_id = self._match_id(url)
|
||||||
|
webpage = self._download_webpage(url, video_id)
|
||||||
|
title = get_elements_by_class('heading', webpage)[-1]
|
||||||
|
duration = parse_duration(self._search_regex(
|
||||||
|
r'Duration: ((?:[0-9]{2}:){0,2}[0-9]{2})',
|
||||||
|
webpage, 'duration', fatal=False))
|
||||||
|
views = parse_count(self._search_regex(
|
||||||
|
r'Views: ([0-9 ]+)',
|
||||||
|
webpage, 'views', fatal=False))
|
||||||
|
|
||||||
|
player_hash = self._search_regex(
|
||||||
|
r'DaxabPlayer\.Init\({[\s\S]*hash:\s*"([0-9a-zA-Z_\-]+)"[\s\S]*}',
|
||||||
|
webpage, 'player hash')
|
||||||
|
player_color = self._search_regex(
|
||||||
|
r'DaxabPlayer\.Init\({[\s\S]*color:\s*"([0-9a-z]+)"[\s\S]*}',
|
||||||
|
webpage, 'player color', fatal=False) or ''
|
||||||
|
|
||||||
|
embed_page = self._download_webpage(
|
||||||
|
'https://daxab.com/player/%s?color=%s' % (player_hash, player_color),
|
||||||
|
video_id, headers={'Referer': url})
|
||||||
|
video_params = self._parse_json(
|
||||||
|
self._search_regex(
|
||||||
|
r'window\.globParams\s*=\s*({[\S\s]+})\s*;\s*<\/script>',
|
||||||
|
embed_page, 'video parameters'),
|
||||||
|
video_id, transform_source=js_to_json)
|
||||||
|
|
||||||
|
server_domain = 'https://%s' % compat_b64decode(video_params['server'][::-1]).decode('utf-8')
|
||||||
|
formats = []
|
||||||
|
for format_id, format_data in video_params['video']['cdn_files'].items():
|
||||||
|
ext, height = format_id.split('_')
|
||||||
|
extra_quality_data = format_data.split('.')[-1]
|
||||||
|
url = f'{server_domain}/videos/{video_id.replace("_", "/")}/{height}.mp4?extra={extra_quality_data}'
|
||||||
|
formats.append({
|
||||||
|
'format_id': format_id,
|
||||||
|
'url': url,
|
||||||
|
'height': int_or_none(height),
|
||||||
|
'ext': ext,
|
||||||
|
})
|
||||||
|
self._sort_formats(formats)
|
||||||
|
|
||||||
|
thumbnail = try_get(video_params,
|
||||||
|
lambda vi: 'https:' + compat_b64decode(vi['video']['thumb']).decode('utf-8'))
|
||||||
|
|
||||||
|
return {
|
||||||
|
'id': video_id,
|
||||||
|
'title': title,
|
||||||
|
'formats': formats,
|
||||||
|
'duration': duration,
|
||||||
|
'thumbnail': thumbnail,
|
||||||
|
'view_count': views,
|
||||||
|
'age_limit': 18,
|
||||||
|
}
|
||||||
@@ -0,0 +1,143 @@
|
|||||||
|
# coding: utf-8
|
||||||
|
from __future__ import unicode_literals
|
||||||
|
|
||||||
|
from .common import InfoExtractor
|
||||||
|
|
||||||
|
from ..utils import (
|
||||||
|
ExtractorError,
|
||||||
|
parse_resolution,
|
||||||
|
traverse_obj,
|
||||||
|
try_get,
|
||||||
|
urlencode_postdata,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class DigitalConcertHallIE(InfoExtractor):
|
||||||
|
IE_DESC = 'DigitalConcertHall extractor'
|
||||||
|
_VALID_URL = r'https?://(?:www\.)?digitalconcerthall\.com/(?P<language>[a-z]+)/concert/(?P<id>[0-9]+)'
|
||||||
|
_OAUTH_URL = 'https://api.digitalconcerthall.com/v2/oauth2/token'
|
||||||
|
_ACCESS_TOKEN = None
|
||||||
|
_NETRC_MACHINE = 'digitalconcerthall'
|
||||||
|
_TESTS = [{
|
||||||
|
'note': 'Playlist with only one video',
|
||||||
|
'url': 'https://www.digitalconcerthall.com/en/concert/53201',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '53201-1',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'composer': 'Kurt Weill',
|
||||||
|
'title': '[Magic Night]',
|
||||||
|
'thumbnail': r're:^https?://images.digitalconcerthall.com/cms/thumbnails.*\.jpg$',
|
||||||
|
'upload_date': '20210624',
|
||||||
|
'timestamp': 1624548600,
|
||||||
|
'duration': 2798,
|
||||||
|
'album_artist': 'Members of the Berliner Philharmoniker / Simon Rössler',
|
||||||
|
},
|
||||||
|
'params': {'skip_download': 'm3u8'},
|
||||||
|
}, {
|
||||||
|
'note': 'Concert with several works and an interview',
|
||||||
|
'url': 'https://www.digitalconcerthall.com/en/concert/53785',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '53785',
|
||||||
|
'album_artist': 'Berliner Philharmoniker / Kirill Petrenko',
|
||||||
|
'title': 'Kirill Petrenko conducts Mendelssohn and Shostakovich',
|
||||||
|
},
|
||||||
|
'params': {'skip_download': 'm3u8'},
|
||||||
|
'playlist_count': 3,
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _login(self):
|
||||||
|
username, password = self._get_login_info()
|
||||||
|
if not username:
|
||||||
|
self.raise_login_required()
|
||||||
|
token_response = self._download_json(
|
||||||
|
self._OAUTH_URL,
|
||||||
|
None, 'Obtaining token', errnote='Unable to obtain token', data=urlencode_postdata({
|
||||||
|
'affiliate': 'none',
|
||||||
|
'grant_type': 'device',
|
||||||
|
'device_vendor': 'unknown',
|
||||||
|
'app_id': 'dch.webapp',
|
||||||
|
'app_version': '1.0.0',
|
||||||
|
'client_secret': '2ySLN+2Fwb',
|
||||||
|
}), headers={
|
||||||
|
'Content-Type': 'application/x-www-form-urlencoded',
|
||||||
|
})
|
||||||
|
self._ACCESS_TOKEN = token_response['access_token']
|
||||||
|
try:
|
||||||
|
self._download_json(
|
||||||
|
self._OAUTH_URL,
|
||||||
|
None, note='Logging in', errnote='Unable to login', data=urlencode_postdata({
|
||||||
|
'grant_type': 'password',
|
||||||
|
'username': username,
|
||||||
|
'password': password,
|
||||||
|
}), headers={
|
||||||
|
'Content-Type': 'application/x-www-form-urlencoded',
|
||||||
|
'Referer': 'https://www.digitalconcerthall.com',
|
||||||
|
'Authorization': f'Bearer {self._ACCESS_TOKEN}'
|
||||||
|
})
|
||||||
|
except ExtractorError:
|
||||||
|
self.raise_login_required(msg='Login info incorrect')
|
||||||
|
|
||||||
|
def _real_initialize(self):
|
||||||
|
self._login()
|
||||||
|
|
||||||
|
def _entries(self, items, language, **kwargs):
|
||||||
|
for item in items:
|
||||||
|
video_id = item['id']
|
||||||
|
stream_info = self._download_json(
|
||||||
|
self._proto_relative_url(item['_links']['streams']['href']), video_id, headers={
|
||||||
|
'Accept': 'application/json',
|
||||||
|
'Authorization': f'Bearer {self._ACCESS_TOKEN}',
|
||||||
|
'Accept-Language': language
|
||||||
|
})
|
||||||
|
|
||||||
|
m3u8_url = traverse_obj(
|
||||||
|
stream_info, ('channel', lambda x: x.startswith('vod_mixed'), 'stream', 0, 'url'), get_all=False)
|
||||||
|
formats = self._extract_m3u8_formats(m3u8_url, video_id, 'mp4', 'm3u8_native', fatal=False)
|
||||||
|
self._sort_formats(formats)
|
||||||
|
|
||||||
|
yield {
|
||||||
|
'id': video_id,
|
||||||
|
'title': item.get('title'),
|
||||||
|
'composer': item.get('name_composer'),
|
||||||
|
'url': m3u8_url,
|
||||||
|
'formats': formats,
|
||||||
|
'duration': item.get('duration_total'),
|
||||||
|
'timestamp': traverse_obj(item, ('date', 'published')),
|
||||||
|
'description': item.get('short_description') or stream_info.get('short_description'),
|
||||||
|
**kwargs,
|
||||||
|
'chapters': [{
|
||||||
|
'start_time': chapter.get('time'),
|
||||||
|
'end_time': try_get(chapter, lambda x: x['time'] + x['duration']),
|
||||||
|
'title': chapter.get('text'),
|
||||||
|
} for chapter in item['cuepoints']] if item.get('cuepoints') else None,
|
||||||
|
}
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
language, video_id = self._match_valid_url(url).group('language', 'id')
|
||||||
|
if not language:
|
||||||
|
language = 'en'
|
||||||
|
|
||||||
|
thumbnail_url = self._html_search_regex(
|
||||||
|
r'(https?://images\.digitalconcerthall\.com/cms/thumbnails/.*\.jpg)',
|
||||||
|
self._download_webpage(url, video_id), 'thumbnail')
|
||||||
|
thumbnails = [{
|
||||||
|
'url': thumbnail_url,
|
||||||
|
**parse_resolution(thumbnail_url)
|
||||||
|
}]
|
||||||
|
|
||||||
|
vid_info = self._download_json(
|
||||||
|
f'https://api.digitalconcerthall.com/v2/concert/{video_id}', video_id, headers={
|
||||||
|
'Accept': 'application/json',
|
||||||
|
'Accept-Language': language
|
||||||
|
})
|
||||||
|
album_artist = ' / '.join(traverse_obj(vid_info, ('_links', 'artist', ..., 'name')) or '')
|
||||||
|
|
||||||
|
return {
|
||||||
|
'_type': 'playlist',
|
||||||
|
'id': video_id,
|
||||||
|
'title': vid_info.get('title'),
|
||||||
|
'entries': self._entries(traverse_obj(vid_info, ('_embedded', ..., ...)), language,
|
||||||
|
thumbnails=thumbnails, album_artist=album_artist),
|
||||||
|
'thumbnails': thumbnails,
|
||||||
|
'album_artist': album_artist,
|
||||||
|
}
|
||||||
@@ -74,13 +74,11 @@ class DigitallySpeakingIE(InfoExtractor):
|
|||||||
tbr = int_or_none(bitrate)
|
tbr = int_or_none(bitrate)
|
||||||
vbr = int_or_none(self._search_regex(
|
vbr = int_or_none(self._search_regex(
|
||||||
r'-(\d+)\.mp4', video_path, 'vbr', default=None))
|
r'-(\d+)\.mp4', video_path, 'vbr', default=None))
|
||||||
abr = tbr - vbr if tbr and vbr else None
|
|
||||||
video_formats.append({
|
video_formats.append({
|
||||||
'format_id': bitrate,
|
'format_id': bitrate,
|
||||||
'url': url,
|
'url': url,
|
||||||
'tbr': tbr,
|
'tbr': tbr,
|
||||||
'vbr': vbr,
|
'vbr': vbr,
|
||||||
'abr': abr,
|
|
||||||
})
|
})
|
||||||
return video_formats
|
return video_formats
|
||||||
|
|
||||||
@@ -121,6 +119,7 @@ class DigitallySpeakingIE(InfoExtractor):
|
|||||||
video_formats = self._parse_mp4(metadata)
|
video_formats = self._parse_mp4(metadata)
|
||||||
if video_formats is None:
|
if video_formats is None:
|
||||||
video_formats = self._parse_flv(metadata)
|
video_formats = self._parse_flv(metadata)
|
||||||
|
self._sort_formats(video_formats)
|
||||||
|
|
||||||
return {
|
return {
|
||||||
'id': video_id,
|
'id': video_id,
|
||||||
|
|||||||
+132
-106
@@ -347,7 +347,101 @@ class HGTVDeIE(DPlayBaseIE):
|
|||||||
url, display_id, 'eu1-prod.disco-api.com', 'hgtv', 'de')
|
url, display_id, 'eu1-prod.disco-api.com', 'hgtv', 'de')
|
||||||
|
|
||||||
|
|
||||||
class DiscoveryPlusIE(DPlayBaseIE):
|
class DiscoveryPlusBaseIE(DPlayBaseIE):
|
||||||
|
def _update_disco_api_headers(self, headers, disco_base, display_id, realm):
|
||||||
|
headers['x-disco-client'] = f'WEB:UNKNOWN:{self._PRODUCT}:25.2.6'
|
||||||
|
|
||||||
|
def _download_video_playback_info(self, disco_base, video_id, headers):
|
||||||
|
return self._download_json(
|
||||||
|
disco_base + 'playback/v3/videoPlaybackInfo',
|
||||||
|
video_id, headers=headers, data=json.dumps({
|
||||||
|
'deviceInfo': {
|
||||||
|
'adBlocker': False,
|
||||||
|
},
|
||||||
|
'videoId': video_id,
|
||||||
|
'wisteriaProperties': {
|
||||||
|
'platform': 'desktop',
|
||||||
|
'product': self._PRODUCT,
|
||||||
|
},
|
||||||
|
}).encode('utf-8'))['data']['attributes']['streaming']
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
return self._get_disco_api_info(url, self._match_id(url), **self._DISCO_API_PARAMS)
|
||||||
|
|
||||||
|
|
||||||
|
class ScienceChannelIE(DiscoveryPlusBaseIE):
|
||||||
|
_VALID_URL = r'https?://(?:www\.)?sciencechannel\.com/video' + DPlayBaseIE._PATH_REGEX
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://www.sciencechannel.com/video/strangest-things-science-atve-us/nazi-mystery-machine',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '2842849',
|
||||||
|
'display_id': 'strangest-things-science-atve-us/nazi-mystery-machine',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'Nazi Mystery Machine',
|
||||||
|
'description': 'Experts investigate the secrets of a revolutionary encryption machine.',
|
||||||
|
'season_number': 1,
|
||||||
|
'episode_number': 1,
|
||||||
|
},
|
||||||
|
'skip': 'Available for Premium users',
|
||||||
|
}]
|
||||||
|
|
||||||
|
_PRODUCT = 'sci'
|
||||||
|
_DISCO_API_PARAMS = {
|
||||||
|
'disco_host': 'us1-prod-direct.sciencechannel.com',
|
||||||
|
'realm': 'go',
|
||||||
|
'country': 'us',
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class DIYNetworkIE(DiscoveryPlusBaseIE):
|
||||||
|
_VALID_URL = r'https?://(?:watch\.)?diynetwork\.com/video' + DPlayBaseIE._PATH_REGEX
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://watch.diynetwork.com/video/pool-kings-diy-network/bringing-beach-life-to-texas',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '2309730',
|
||||||
|
'display_id': 'pool-kings-diy-network/bringing-beach-life-to-texas',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'Bringing Beach Life to Texas',
|
||||||
|
'description': 'The Pool Kings give a family a day at the beach in their own backyard.',
|
||||||
|
'season_number': 10,
|
||||||
|
'episode_number': 2,
|
||||||
|
},
|
||||||
|
'skip': 'Available for Premium users',
|
||||||
|
}]
|
||||||
|
|
||||||
|
_PRODUCT = 'diy'
|
||||||
|
_DISCO_API_PARAMS = {
|
||||||
|
'disco_host': 'us1-prod-direct.watch.diynetwork.com',
|
||||||
|
'realm': 'go',
|
||||||
|
'country': 'us',
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class AnimalPlanetIE(DiscoveryPlusBaseIE):
|
||||||
|
_VALID_URL = r'https?://(?:www\.)?animalplanet\.com/video' + DPlayBaseIE._PATH_REGEX
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://www.animalplanet.com/video/north-woods-law-animal-planet/squirrel-showdown',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '3338923',
|
||||||
|
'display_id': 'north-woods-law-animal-planet/squirrel-showdown',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'Squirrel Showdown',
|
||||||
|
'description': 'A woman is suspected of being in possession of flying squirrel kits.',
|
||||||
|
'season_number': 16,
|
||||||
|
'episode_number': 11,
|
||||||
|
},
|
||||||
|
'skip': 'Available for Premium users',
|
||||||
|
}]
|
||||||
|
|
||||||
|
_PRODUCT = 'apl'
|
||||||
|
_DISCO_API_PARAMS = {
|
||||||
|
'disco_host': 'us1-prod-direct.animalplanet.com',
|
||||||
|
'realm': 'go',
|
||||||
|
'country': 'us',
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class DiscoveryPlusIE(DiscoveryPlusBaseIE):
|
||||||
_VALID_URL = r'https?://(?:www\.)?discoveryplus\.com/(?!it/)(?:\w{2}/)?video' + DPlayBaseIE._PATH_REGEX
|
_VALID_URL = r'https?://(?:www\.)?discoveryplus\.com/(?!it/)(?:\w{2}/)?video' + DPlayBaseIE._PATH_REGEX
|
||||||
_TESTS = [{
|
_TESTS = [{
|
||||||
'url': 'https://www.discoveryplus.com/video/property-brothers-forever-home/food-and-family',
|
'url': 'https://www.discoveryplus.com/video/property-brothers-forever-home/food-and-family',
|
||||||
@@ -372,92 +466,14 @@ class DiscoveryPlusIE(DPlayBaseIE):
|
|||||||
}]
|
}]
|
||||||
|
|
||||||
_PRODUCT = 'dplus_us'
|
_PRODUCT = 'dplus_us'
|
||||||
_API_URL = 'us1-prod-direct.discoveryplus.com'
|
_DISCO_API_PARAMS = {
|
||||||
|
'disco_host': 'us1-prod-direct.discoveryplus.com',
|
||||||
def _update_disco_api_headers(self, headers, disco_base, display_id, realm):
|
'realm': 'go',
|
||||||
headers['x-disco-client'] = f'WEB:UNKNOWN:{self._PRODUCT}:25.2.6'
|
'country': 'us',
|
||||||
|
}
|
||||||
def _download_video_playback_info(self, disco_base, video_id, headers):
|
|
||||||
return self._download_json(
|
|
||||||
disco_base + 'playback/v3/videoPlaybackInfo',
|
|
||||||
video_id, headers=headers, data=json.dumps({
|
|
||||||
'deviceInfo': {
|
|
||||||
'adBlocker': False,
|
|
||||||
},
|
|
||||||
'videoId': video_id,
|
|
||||||
'wisteriaProperties': {
|
|
||||||
'platform': 'desktop',
|
|
||||||
'product': self._PRODUCT,
|
|
||||||
},
|
|
||||||
}).encode('utf-8'))['data']['attributes']['streaming']
|
|
||||||
|
|
||||||
def _real_extract(self, url):
|
|
||||||
display_id = self._match_id(url)
|
|
||||||
return self._get_disco_api_info(
|
|
||||||
url, display_id, self._API_URL, 'go', 'us')
|
|
||||||
|
|
||||||
|
|
||||||
class ScienceChannelIE(DiscoveryPlusIE):
|
class DiscoveryPlusIndiaIE(DiscoveryPlusBaseIE):
|
||||||
_VALID_URL = r'https?://(?:www\.)?sciencechannel\.com/video' + DPlayBaseIE._PATH_REGEX
|
|
||||||
_TESTS = [{
|
|
||||||
'url': 'https://www.sciencechannel.com/video/strangest-things-science-atve-us/nazi-mystery-machine',
|
|
||||||
'info_dict': {
|
|
||||||
'id': '2842849',
|
|
||||||
'display_id': 'strangest-things-science-atve-us/nazi-mystery-machine',
|
|
||||||
'ext': 'mp4',
|
|
||||||
'title': 'Nazi Mystery Machine',
|
|
||||||
'description': 'Experts investigate the secrets of a revolutionary encryption machine.',
|
|
||||||
'season_number': 1,
|
|
||||||
'episode_number': 1,
|
|
||||||
},
|
|
||||||
'skip': 'Available for Premium users',
|
|
||||||
}]
|
|
||||||
|
|
||||||
_PRODUCT = 'sci'
|
|
||||||
_API_URL = 'us1-prod-direct.sciencechannel.com'
|
|
||||||
|
|
||||||
|
|
||||||
class DIYNetworkIE(DiscoveryPlusIE):
|
|
||||||
_VALID_URL = r'https?://(?:watch\.)?diynetwork\.com/video' + DPlayBaseIE._PATH_REGEX
|
|
||||||
_TESTS = [{
|
|
||||||
'url': 'https://watch.diynetwork.com/video/pool-kings-diy-network/bringing-beach-life-to-texas',
|
|
||||||
'info_dict': {
|
|
||||||
'id': '2309730',
|
|
||||||
'display_id': 'pool-kings-diy-network/bringing-beach-life-to-texas',
|
|
||||||
'ext': 'mp4',
|
|
||||||
'title': 'Bringing Beach Life to Texas',
|
|
||||||
'description': 'The Pool Kings give a family a day at the beach in their own backyard.',
|
|
||||||
'season_number': 10,
|
|
||||||
'episode_number': 2,
|
|
||||||
},
|
|
||||||
'skip': 'Available for Premium users',
|
|
||||||
}]
|
|
||||||
|
|
||||||
_PRODUCT = 'diy'
|
|
||||||
_API_URL = 'us1-prod-direct.watch.diynetwork.com'
|
|
||||||
|
|
||||||
|
|
||||||
class AnimalPlanetIE(DiscoveryPlusIE):
|
|
||||||
_VALID_URL = r'https?://(?:www\.)?animalplanet\.com/video' + DPlayBaseIE._PATH_REGEX
|
|
||||||
_TESTS = [{
|
|
||||||
'url': 'https://www.animalplanet.com/video/north-woods-law-animal-planet/squirrel-showdown',
|
|
||||||
'info_dict': {
|
|
||||||
'id': '3338923',
|
|
||||||
'display_id': 'north-woods-law-animal-planet/squirrel-showdown',
|
|
||||||
'ext': 'mp4',
|
|
||||||
'title': 'Squirrel Showdown',
|
|
||||||
'description': 'A woman is suspected of being in possession of flying squirrel kits.',
|
|
||||||
'season_number': 16,
|
|
||||||
'episode_number': 11,
|
|
||||||
},
|
|
||||||
'skip': 'Available for Premium users',
|
|
||||||
}]
|
|
||||||
|
|
||||||
_PRODUCT = 'apl'
|
|
||||||
_API_URL = 'us1-prod-direct.animalplanet.com'
|
|
||||||
|
|
||||||
|
|
||||||
class DiscoveryPlusIndiaIE(DPlayBaseIE):
|
|
||||||
_VALID_URL = r'https?://(?:www\.)?discoveryplus\.in/videos?' + DPlayBaseIE._PATH_REGEX
|
_VALID_URL = r'https?://(?:www\.)?discoveryplus\.in/videos?' + DPlayBaseIE._PATH_REGEX
|
||||||
_TESTS = [{
|
_TESTS = [{
|
||||||
'url': 'https://www.discoveryplus.in/videos/how-do-they-do-it/fugu-and-more?seasonId=8&type=EPISODE',
|
'url': 'https://www.discoveryplus.in/videos/how-do-they-do-it/fugu-and-more?seasonId=8&type=EPISODE',
|
||||||
@@ -467,41 +483,38 @@ class DiscoveryPlusIndiaIE(DPlayBaseIE):
|
|||||||
'display_id': 'how-do-they-do-it/fugu-and-more',
|
'display_id': 'how-do-they-do-it/fugu-and-more',
|
||||||
'title': 'Fugu and More',
|
'title': 'Fugu and More',
|
||||||
'description': 'The Japanese catch, prepare and eat the deadliest fish on the planet.',
|
'description': 'The Japanese catch, prepare and eat the deadliest fish on the planet.',
|
||||||
'duration': 1319,
|
'duration': 1319.32,
|
||||||
'timestamp': 1582309800,
|
'timestamp': 1582309800,
|
||||||
'upload_date': '20200221',
|
'upload_date': '20200221',
|
||||||
'series': 'How Do They Do It?',
|
'series': 'How Do They Do It?',
|
||||||
'season_number': 8,
|
'season_number': 8,
|
||||||
'episode_number': 2,
|
'episode_number': 2,
|
||||||
'creator': 'Discovery Channel',
|
'creator': 'Discovery Channel',
|
||||||
|
'thumbnail': r're:https://.+\.jpeg',
|
||||||
|
'episode': 'Episode 2',
|
||||||
|
'season': 'Season 8',
|
||||||
|
'tags': [],
|
||||||
},
|
},
|
||||||
'params': {
|
'params': {
|
||||||
'skip_download': True,
|
'skip_download': True,
|
||||||
}
|
}
|
||||||
}]
|
}]
|
||||||
|
|
||||||
|
_PRODUCT = 'dplus-india'
|
||||||
|
_DISCO_API_PARAMS = {
|
||||||
|
'disco_host': 'ap2-prod-direct.discoveryplus.in',
|
||||||
|
'realm': 'dplusindia',
|
||||||
|
'country': 'in',
|
||||||
|
'domain': 'https://www.discoveryplus.in/',
|
||||||
|
}
|
||||||
|
|
||||||
def _update_disco_api_headers(self, headers, disco_base, display_id, realm):
|
def _update_disco_api_headers(self, headers, disco_base, display_id, realm):
|
||||||
headers.update({
|
headers.update({
|
||||||
'x-disco-params': 'realm=%s' % realm,
|
'x-disco-params': 'realm=%s' % realm,
|
||||||
'x-disco-client': 'WEB:UNKNOWN:dplus-india:17.0.0',
|
'x-disco-client': f'WEB:UNKNOWN:{self._PRODUCT}:17.0.0',
|
||||||
'Authorization': self._get_auth(disco_base, display_id, realm),
|
'Authorization': self._get_auth(disco_base, display_id, realm),
|
||||||
})
|
})
|
||||||
|
|
||||||
def _download_video_playback_info(self, disco_base, video_id, headers):
|
|
||||||
return self._download_json(
|
|
||||||
disco_base + 'playback/v3/videoPlaybackInfo',
|
|
||||||
video_id, headers=headers, data=json.dumps({
|
|
||||||
'deviceInfo': {
|
|
||||||
'adBlocker': False,
|
|
||||||
},
|
|
||||||
'videoId': video_id,
|
|
||||||
}).encode('utf-8'))['data']['attributes']['streaming']
|
|
||||||
|
|
||||||
def _real_extract(self, url):
|
|
||||||
display_id = self._match_id(url)
|
|
||||||
return self._get_disco_api_info(
|
|
||||||
url, display_id, 'ap2-prod-direct.discoveryplus.in', 'dplusindia', 'in', 'https://www.discoveryplus.in/')
|
|
||||||
|
|
||||||
|
|
||||||
class DiscoveryNetworksDeIE(DPlayBaseIE):
|
class DiscoveryNetworksDeIE(DPlayBaseIE):
|
||||||
_VALID_URL = r'https?://(?:www\.)?(?P<domain>(?:tlc|dmax)\.de|dplay\.co\.uk)/(?:programme|show|sendungen)/(?P<programme>[^/]+)/(?:video/)?(?P<alternate_id>[^/]+)'
|
_VALID_URL = r'https?://(?:www\.)?(?P<domain>(?:tlc|dmax)\.de|dplay\.co\.uk)/(?:programme|show|sendungen)/(?P<programme>[^/]+)/(?:video/)?(?P<alternate_id>[^/]+)'
|
||||||
@@ -515,6 +528,16 @@ class DiscoveryNetworksDeIE(DPlayBaseIE):
|
|||||||
'description': 'md5:61033c12b73286e409d99a41742ef608',
|
'description': 'md5:61033c12b73286e409d99a41742ef608',
|
||||||
'timestamp': 1554069600,
|
'timestamp': 1554069600,
|
||||||
'upload_date': '20190331',
|
'upload_date': '20190331',
|
||||||
|
'creator': 'TLC',
|
||||||
|
'season': 'Season 1',
|
||||||
|
'series': 'Breaking Amish',
|
||||||
|
'episode_number': 1,
|
||||||
|
'tags': ['new york', 'großstadt', 'amische', 'landleben', 'modern', 'infos', 'tradition', 'herausforderung'],
|
||||||
|
'display_id': 'breaking-amish/die-welt-da-drauen',
|
||||||
|
'episode': 'Episode 1',
|
||||||
|
'duration': 2625.024,
|
||||||
|
'season_number': 1,
|
||||||
|
'thumbnail': r're:https://.+\.jpg',
|
||||||
},
|
},
|
||||||
'params': {
|
'params': {
|
||||||
'skip_download': True,
|
'skip_download': True,
|
||||||
@@ -575,16 +598,19 @@ class DiscoveryPlusShowBaseIE(DPlayBaseIE):
|
|||||||
return self.playlist_result(self._entries(show_name), playlist_id=show_name)
|
return self.playlist_result(self._entries(show_name), playlist_id=show_name)
|
||||||
|
|
||||||
|
|
||||||
class DiscoveryPlusItalyIE(InfoExtractor):
|
class DiscoveryPlusItalyIE(DiscoveryPlusBaseIE):
|
||||||
_VALID_URL = r'https?://(?:www\.)?discoveryplus\.com/it/video' + DPlayBaseIE._PATH_REGEX
|
_VALID_URL = r'https?://(?:www\.)?discoveryplus\.com/it/video' + DPlayBaseIE._PATH_REGEX
|
||||||
_TESTS = [{
|
_TESTS = [{
|
||||||
'url': 'https://www.discoveryplus.com/it/video/i-signori-della-neve/stagione-2-episodio-1-i-preparativi',
|
'url': 'https://www.discoveryplus.com/it/video/i-signori-della-neve/stagione-2-episodio-1-i-preparativi',
|
||||||
'only_matching': True,
|
'only_matching': True,
|
||||||
}]
|
}]
|
||||||
|
|
||||||
def _real_extract(self, url):
|
_PRODUCT = 'dplus_us'
|
||||||
video_id = self._match_id(url)
|
_DISCO_API_PARAMS = {
|
||||||
return self.url_result(f'https://discoveryplus.it/video/{video_id}', DPlayIE.ie_key(), video_id)
|
'disco_host': 'eu1-prod-direct.discoveryplus.com',
|
||||||
|
'realm': 'dplay',
|
||||||
|
'country': 'it',
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
class DiscoveryPlusItalyShowIE(DiscoveryPlusShowBaseIE):
|
class DiscoveryPlusItalyShowIE(DiscoveryPlusShowBaseIE):
|
||||||
|
|||||||
@@ -0,0 +1,116 @@
|
|||||||
|
# coding: utf-8
|
||||||
|
from __future__ import unicode_literals
|
||||||
|
|
||||||
|
import json
|
||||||
|
|
||||||
|
from .common import InfoExtractor
|
||||||
|
from ..utils import (
|
||||||
|
ExtractorError,
|
||||||
|
int_or_none,
|
||||||
|
try_get,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class DroobleIE(InfoExtractor):
|
||||||
|
_VALID_URL = r'''(?x)https?://drooble\.com/(?:
|
||||||
|
(?:(?P<user>[^/]+)/)?(?P<kind>song|videos|music/albums)/(?P<id>\d+)|
|
||||||
|
(?P<user_2>[^/]+)/(?P<kind_2>videos|music))
|
||||||
|
'''
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://drooble.com/song/2858030',
|
||||||
|
'md5': '5ffda90f61c7c318dc0c3df4179eb064',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '2858030',
|
||||||
|
'ext': 'mp3',
|
||||||
|
'title': 'Skankocillin',
|
||||||
|
'upload_date': '20200801',
|
||||||
|
'timestamp': 1596241390,
|
||||||
|
'uploader_id': '95894',
|
||||||
|
'uploader': 'Bluebeat Shelter',
|
||||||
|
}
|
||||||
|
}, {
|
||||||
|
'url': 'https://drooble.com/karl340758/videos/2859183',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'J6QCQY_I5Tk',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'Skankocillin',
|
||||||
|
'uploader_id': 'UCrSRoI5vVyeYihtWEYua7rg',
|
||||||
|
'description': 'md5:ffc0bd8ba383db5341a86a6cd7d9bcca',
|
||||||
|
'upload_date': '20200731',
|
||||||
|
'uploader': 'Bluebeat Shelter',
|
||||||
|
}
|
||||||
|
}, {
|
||||||
|
'url': 'https://drooble.com/karl340758/music/albums/2858031',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '2858031',
|
||||||
|
},
|
||||||
|
'playlist_mincount': 8,
|
||||||
|
}, {
|
||||||
|
'url': 'https://drooble.com/karl340758/music',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'karl340758',
|
||||||
|
},
|
||||||
|
'playlist_mincount': 8,
|
||||||
|
}, {
|
||||||
|
'url': 'https://drooble.com/karl340758/videos',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'karl340758',
|
||||||
|
},
|
||||||
|
'playlist_mincount': 8,
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _call_api(self, method, video_id, data=None):
|
||||||
|
response = self._download_json(
|
||||||
|
f'https://drooble.com/api/dt/{method}', video_id, data=json.dumps(data).encode())
|
||||||
|
if not response[0]:
|
||||||
|
raise ExtractorError('Unable to download JSON metadata')
|
||||||
|
return response[1]
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
mobj = self._match_valid_url(url)
|
||||||
|
user = mobj.group('user') or mobj.group('user_2')
|
||||||
|
kind = mobj.group('kind') or mobj.group('kind_2')
|
||||||
|
display_id = mobj.group('id') or user
|
||||||
|
|
||||||
|
if mobj.group('kind_2') == 'videos':
|
||||||
|
data = {'from_user': display_id, 'album': -1, 'limit': 18, 'offset': 0, 'order': 'new2old', 'type': 'video'}
|
||||||
|
elif kind in ('music/albums', 'music'):
|
||||||
|
data = {'user': user, 'public_only': True, 'individual_limit': {'singles': 1, 'albums': 1, 'playlists': 1}}
|
||||||
|
else:
|
||||||
|
data = {'url_slug': display_id, 'children': 10, 'order': 'old2new'}
|
||||||
|
|
||||||
|
method = 'getMusicOverview' if kind in ('music/albums', 'music') else 'getElements'
|
||||||
|
json_data = self._call_api(method, display_id, data=data)
|
||||||
|
if kind in ('music/albums', 'music'):
|
||||||
|
json_data = json_data['singles']['list']
|
||||||
|
|
||||||
|
entites = []
|
||||||
|
for media in json_data:
|
||||||
|
url = media.get('external_media_url') or media.get('link')
|
||||||
|
if url.startswith('https://www.youtube.com'):
|
||||||
|
entites.append({
|
||||||
|
'_type': 'url',
|
||||||
|
'url': url,
|
||||||
|
'ie_key': 'Youtube'
|
||||||
|
})
|
||||||
|
continue
|
||||||
|
is_audio = (media.get('type') or '').lower() == 'audio'
|
||||||
|
entites.append({
|
||||||
|
'url': url,
|
||||||
|
'id': media['id'],
|
||||||
|
'title': media['title'],
|
||||||
|
'duration': int_or_none(media.get('duration')),
|
||||||
|
'timestamp': int_or_none(media.get('timestamp')),
|
||||||
|
'album': try_get(media, lambda x: x['album']['title']),
|
||||||
|
'uploader': try_get(media, lambda x: x['creator']['display_name']),
|
||||||
|
'uploader_id': try_get(media, lambda x: x['creator']['id']),
|
||||||
|
'thumbnail': media.get('image_comment'),
|
||||||
|
'like_count': int_or_none(media.get('likes')),
|
||||||
|
'vcodec': 'none' if is_audio else None,
|
||||||
|
'ext': 'mp3' if is_audio else None,
|
||||||
|
})
|
||||||
|
|
||||||
|
if len(entites) > 1:
|
||||||
|
return self.playlist_result(entites, display_id)
|
||||||
|
|
||||||
|
return entites[0]
|
||||||
@@ -6,7 +6,12 @@ import re
|
|||||||
|
|
||||||
from .common import InfoExtractor
|
from .common import InfoExtractor
|
||||||
from ..compat import compat_urllib_parse_unquote
|
from ..compat import compat_urllib_parse_unquote
|
||||||
from ..utils import url_basename
|
from ..utils import (
|
||||||
|
ExtractorError,
|
||||||
|
traverse_obj,
|
||||||
|
try_get,
|
||||||
|
url_basename,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
class DropboxIE(InfoExtractor):
|
class DropboxIE(InfoExtractor):
|
||||||
@@ -28,13 +33,44 @@ class DropboxIE(InfoExtractor):
|
|||||||
def _real_extract(self, url):
|
def _real_extract(self, url):
|
||||||
mobj = self._match_valid_url(url)
|
mobj = self._match_valid_url(url)
|
||||||
video_id = mobj.group('id')
|
video_id = mobj.group('id')
|
||||||
|
webpage = self._download_webpage(url, video_id)
|
||||||
fn = compat_urllib_parse_unquote(url_basename(url))
|
fn = compat_urllib_parse_unquote(url_basename(url))
|
||||||
title = os.path.splitext(fn)[0]
|
title = os.path.splitext(fn)[0]
|
||||||
video_url = re.sub(r'[?&]dl=0', '', url)
|
|
||||||
video_url += ('?' if '?' not in video_url else '&') + 'dl=1'
|
password = self.get_param('videopassword')
|
||||||
|
if (self._og_search_title(webpage) == 'Dropbox - Password Required'
|
||||||
|
or 'Enter the password for this link' in webpage):
|
||||||
|
|
||||||
|
if password:
|
||||||
|
content_id = self._search_regex(r'content_id=(.*?)["\']', webpage, 'content_id')
|
||||||
|
payload = f'is_xhr=true&t={self._get_cookies("https://www.dropbox.com").get("t").value}&content_id={content_id}&password={password}&url={url}'
|
||||||
|
response = self._download_json(
|
||||||
|
'https://www.dropbox.com/sm/auth', video_id, 'POSTing video password', data=payload.encode('UTF-8'),
|
||||||
|
headers={'content-type': 'application/x-www-form-urlencoded; charset=UTF-8'})
|
||||||
|
|
||||||
|
if response.get('status') != 'authed':
|
||||||
|
raise ExtractorError('Authentication failed!', expected=True)
|
||||||
|
webpage = self._download_webpage(url, video_id)
|
||||||
|
elif self._get_cookies('https://dropbox.com').get('sm_auth'):
|
||||||
|
webpage = self._download_webpage(url, video_id)
|
||||||
|
else:
|
||||||
|
raise ExtractorError('Password protected video, use --video-password <password>', expected=True)
|
||||||
|
|
||||||
|
json_string = self._html_search_regex(r'InitReact\.mountComponent.+ "props":(.+), "elem_id"', webpage, 'Info JSON')
|
||||||
|
info_json = self._parse_json(json_string, video_id)
|
||||||
|
transcode_url = traverse_obj(info_json, ((None, 'preview'), 'file', 'preview', 'content', 'transcode_url'), get_all=False)
|
||||||
|
formats, subtitles = self._extract_m3u8_formats_and_subtitles(transcode_url, video_id)
|
||||||
|
|
||||||
|
# downloads enabled we can get the original file
|
||||||
|
if 'anonymous' in (try_get(info_json, lambda x: x['sharePermission']['canDownloadRoles']) or []):
|
||||||
|
video_url = re.sub(r'[?&]dl=0', '', url)
|
||||||
|
video_url += ('?' if '?' not in video_url else '&') + 'dl=1'
|
||||||
|
formats.append({'url': video_url, 'format_id': 'original', 'format_note': 'Original', 'quality': 1})
|
||||||
|
self._sort_formats(formats)
|
||||||
|
|
||||||
return {
|
return {
|
||||||
'id': video_id,
|
'id': video_id,
|
||||||
'title': title,
|
'title': title,
|
||||||
'url': video_url,
|
'formats': formats,
|
||||||
|
'subtitles': subtitles
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,37 @@
|
|||||||
|
# coding: utf-8
|
||||||
|
from __future__ import unicode_literals
|
||||||
|
|
||||||
|
import re
|
||||||
|
|
||||||
|
from .common import InfoExtractor
|
||||||
|
|
||||||
|
|
||||||
|
class EuropeanTourIE(InfoExtractor):
|
||||||
|
_VALID_URL = r'https?://(?:www\.)?europeantour\.com/dpworld-tour/news/video/(?P<id>[^/&?#$]+)'
|
||||||
|
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://www.europeantour.com/dpworld-tour/news/video/the-best-shots-of-the-2021-seasons/',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '6287788195001',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'The best shots of the 2021 seasons',
|
||||||
|
'duration': 2416.512,
|
||||||
|
'timestamp': 1640010141,
|
||||||
|
'uploader_id': '5136026580001',
|
||||||
|
'tags': ['prod-imported'],
|
||||||
|
'thumbnail': 'md5:fdac52bc826548860edf8145ee74e71a',
|
||||||
|
'upload_date': '20211220'
|
||||||
|
},
|
||||||
|
'params': {'skip_download': True}
|
||||||
|
}]
|
||||||
|
|
||||||
|
BRIGHTCOVE_URL_TEMPLATE = 'http://players.brightcove.net/%s/default_default/index.html?videoId=%s'
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
id = self._match_id(url)
|
||||||
|
webpage = self._download_webpage(url, id)
|
||||||
|
vid, aid = re.search(r'(?s)brightcove-player\s?video-id="([^"]+)".*"ACCOUNT_ID":"([^"]+)"', webpage).groups()
|
||||||
|
if not aid:
|
||||||
|
aid = '5136026580001'
|
||||||
|
return self.url_result(
|
||||||
|
self.BRIGHTCOVE_URL_TEMPLATE % (aid, vid), 'BrightcoveNew')
|
||||||
@@ -37,7 +37,10 @@ from .aenetworks import (
|
|||||||
HistoryPlayerIE,
|
HistoryPlayerIE,
|
||||||
BiographyIE,
|
BiographyIE,
|
||||||
)
|
)
|
||||||
from .afreecatv import AfreecaTVIE
|
from .afreecatv import (
|
||||||
|
AfreecaTVIE,
|
||||||
|
AfreecaTVLiveIE,
|
||||||
|
)
|
||||||
from .airmozilla import AirMozillaIE
|
from .airmozilla import AirMozillaIE
|
||||||
from .aljazeera import AlJazeeraIE
|
from .aljazeera import AlJazeeraIE
|
||||||
from .alphaporno import AlphaPornoIE
|
from .alphaporno import AlphaPornoIE
|
||||||
@@ -190,6 +193,7 @@ from .buzzfeed import BuzzFeedIE
|
|||||||
from .byutv import BYUtvIE
|
from .byutv import BYUtvIE
|
||||||
from .c56 import C56IE
|
from .c56 import C56IE
|
||||||
from .cableav import CableAVIE
|
from .cableav import CableAVIE
|
||||||
|
from .callin import CallinIE
|
||||||
from .cam4 import CAM4IE
|
from .cam4 import CAM4IE
|
||||||
from .camdemy import (
|
from .camdemy import (
|
||||||
CamdemyIE,
|
CamdemyIE,
|
||||||
@@ -300,6 +304,10 @@ from .cozytv import CozyTVIE
|
|||||||
from .cracked import CrackedIE
|
from .cracked import CrackedIE
|
||||||
from .crackle import CrackleIE
|
from .crackle import CrackleIE
|
||||||
from .crooksandliars import CrooksAndLiarsIE
|
from .crooksandliars import CrooksAndLiarsIE
|
||||||
|
from .crowdbunker import (
|
||||||
|
CrowdBunkerIE,
|
||||||
|
CrowdBunkerChannelIE,
|
||||||
|
)
|
||||||
from .crunchyroll import (
|
from .crunchyroll import (
|
||||||
CrunchyrollIE,
|
CrunchyrollIE,
|
||||||
CrunchyrollShowPlaylistIE,
|
CrunchyrollShowPlaylistIE,
|
||||||
@@ -317,6 +325,7 @@ from .curiositystream import (
|
|||||||
CuriosityStreamSeriesIE,
|
CuriosityStreamSeriesIE,
|
||||||
)
|
)
|
||||||
from .cwtv import CWTVIE
|
from .cwtv import CWTVIE
|
||||||
|
from .daftsex import DaftsexIE
|
||||||
from .dailymail import DailyMailIE
|
from .dailymail import DailyMailIE
|
||||||
from .dailymotion import (
|
from .dailymotion import (
|
||||||
DailymotionIE,
|
DailymotionIE,
|
||||||
@@ -376,6 +385,7 @@ from .duboku import (
|
|||||||
)
|
)
|
||||||
from .dumpert import DumpertIE
|
from .dumpert import DumpertIE
|
||||||
from .defense import DefenseGouvFrIE
|
from .defense import DefenseGouvFrIE
|
||||||
|
from .digitalconcerthall import DigitalConcertHallIE
|
||||||
from .discovery import DiscoveryIE
|
from .discovery import DiscoveryIE
|
||||||
from .discoverygo import (
|
from .discoverygo import (
|
||||||
DiscoveryGoIE,
|
DiscoveryGoIE,
|
||||||
@@ -432,6 +442,7 @@ from .espn import (
|
|||||||
)
|
)
|
||||||
from .esri import EsriVideoIE
|
from .esri import EsriVideoIE
|
||||||
from .europa import EuropaIE
|
from .europa import EuropaIE
|
||||||
|
from .europeantour import EuropeanTourIE
|
||||||
from .euscreen import EUScreenIE
|
from .euscreen import EUScreenIE
|
||||||
from .expotv import ExpoTVIE
|
from .expotv import ExpoTVIE
|
||||||
from .expressen import ExpressenIE
|
from .expressen import ExpressenIE
|
||||||
@@ -621,6 +632,7 @@ from .instagram import (
|
|||||||
InstagramIOSIE,
|
InstagramIOSIE,
|
||||||
InstagramUserIE,
|
InstagramUserIE,
|
||||||
InstagramTagIE,
|
InstagramTagIE,
|
||||||
|
InstagramStoryIE,
|
||||||
)
|
)
|
||||||
from .internazionale import InternazionaleIE
|
from .internazionale import InternazionaleIE
|
||||||
from .internetvideoarchive import InternetVideoArchiveIE
|
from .internetvideoarchive import InternetVideoArchiveIE
|
||||||
@@ -628,7 +640,11 @@ from .iprima import (
|
|||||||
IPrimaIE,
|
IPrimaIE,
|
||||||
IPrimaCNNIE
|
IPrimaCNNIE
|
||||||
)
|
)
|
||||||
from .iqiyi import IqiyiIE
|
from .iqiyi import (
|
||||||
|
IqiyiIE,
|
||||||
|
IqIE,
|
||||||
|
IqAlbumIE
|
||||||
|
)
|
||||||
from .ir90tv import Ir90TvIE
|
from .ir90tv import Ir90TvIE
|
||||||
from .itv import (
|
from .itv import (
|
||||||
ITVIE,
|
ITVIE,
|
||||||
@@ -655,6 +671,7 @@ from .kankan import KankanIE
|
|||||||
from .karaoketv import KaraoketvIE
|
from .karaoketv import KaraoketvIE
|
||||||
from .karrierevideos import KarriereVideosIE
|
from .karrierevideos import KarriereVideosIE
|
||||||
from .keezmovies import KeezMoviesIE
|
from .keezmovies import KeezMoviesIE
|
||||||
|
from .kelbyone import KelbyOneIE
|
||||||
from .ketnet import KetnetIE
|
from .ketnet import KetnetIE
|
||||||
from .khanacademy import (
|
from .khanacademy import (
|
||||||
KhanAcademyIE,
|
KhanAcademyIE,
|
||||||
@@ -722,7 +739,6 @@ from .limelight import (
|
|||||||
LimelightChannelListIE,
|
LimelightChannelListIE,
|
||||||
)
|
)
|
||||||
from .line import (
|
from .line import (
|
||||||
LineTVIE,
|
|
||||||
LineLiveIE,
|
LineLiveIE,
|
||||||
LineLiveChannelIE,
|
LineLiveChannelIE,
|
||||||
)
|
)
|
||||||
@@ -739,7 +755,10 @@ from .livestream import (
|
|||||||
LivestreamOriginalIE,
|
LivestreamOriginalIE,
|
||||||
LivestreamShortenerIE,
|
LivestreamShortenerIE,
|
||||||
)
|
)
|
||||||
from .lnkgo import LnkGoIE
|
from .lnkgo import (
|
||||||
|
LnkGoIE,
|
||||||
|
LnkIE,
|
||||||
|
)
|
||||||
from .localnews8 import LocalNews8IE
|
from .localnews8 import LocalNews8IE
|
||||||
from .lovehomeporn import LoveHomePornIE
|
from .lovehomeporn import LoveHomePornIE
|
||||||
from .lrt import LRTIE
|
from .lrt import LRTIE
|
||||||
@@ -754,6 +773,7 @@ from .mailru import (
|
|||||||
MailRuMusicIE,
|
MailRuMusicIE,
|
||||||
MailRuMusicSearchIE,
|
MailRuMusicSearchIE,
|
||||||
)
|
)
|
||||||
|
from .mainstreaming import MainStreamingIE
|
||||||
from .malltv import MallTVIE
|
from .malltv import MallTVIE
|
||||||
from .mangomolo import (
|
from .mangomolo import (
|
||||||
MangomoloVideoIE,
|
MangomoloVideoIE,
|
||||||
@@ -819,7 +839,10 @@ from .mirrativ import (
|
|||||||
)
|
)
|
||||||
from .mit import TechTVMITIE, OCWMITIE
|
from .mit import TechTVMITIE, OCWMITIE
|
||||||
from .mitele import MiTeleIE
|
from .mitele import MiTeleIE
|
||||||
from .mixch import MixchIE
|
from .mixch import (
|
||||||
|
MixchIE,
|
||||||
|
MixchArchiveIE,
|
||||||
|
)
|
||||||
from .mixcloud import (
|
from .mixcloud import (
|
||||||
MixcloudIE,
|
MixcloudIE,
|
||||||
MixcloudUserIE,
|
MixcloudUserIE,
|
||||||
@@ -934,6 +957,7 @@ from .newgrounds import (
|
|||||||
NewgroundsUserIE,
|
NewgroundsUserIE,
|
||||||
)
|
)
|
||||||
from .newstube import NewstubeIE
|
from .newstube import NewstubeIE
|
||||||
|
from .newsy import NewsyIE
|
||||||
from .nextmedia import (
|
from .nextmedia import (
|
||||||
NextMediaIE,
|
NextMediaIE,
|
||||||
NextMediaActionNewsIE,
|
NextMediaActionNewsIE,
|
||||||
@@ -981,6 +1005,7 @@ from .nitter import NitterIE
|
|||||||
from .njpwworld import NJPWWorldIE
|
from .njpwworld import NJPWWorldIE
|
||||||
from .nobelprize import NobelPrizeIE
|
from .nobelprize import NobelPrizeIE
|
||||||
from .nonktube import NonkTubeIE
|
from .nonktube import NonkTubeIE
|
||||||
|
from .noodlemagazine import NoodleMagazineIE
|
||||||
from .noovo import NoovoIE
|
from .noovo import NoovoIE
|
||||||
from .normalboots import NormalbootsIE
|
from .normalboots import NormalbootsIE
|
||||||
from .nosvideo import NosVideoIE
|
from .nosvideo import NosVideoIE
|
||||||
@@ -1056,6 +1081,7 @@ from .opencast import (
|
|||||||
from .openrec import (
|
from .openrec import (
|
||||||
OpenRecIE,
|
OpenRecIE,
|
||||||
OpenRecCaptureIE,
|
OpenRecCaptureIE,
|
||||||
|
OpenRecMovieIE,
|
||||||
)
|
)
|
||||||
from .ora import OraTVIE
|
from .ora import OraTVIE
|
||||||
from .orf import (
|
from .orf import (
|
||||||
@@ -1126,6 +1152,10 @@ from .pinterest import (
|
|||||||
PinterestIE,
|
PinterestIE,
|
||||||
PinterestCollectionIE,
|
PinterestCollectionIE,
|
||||||
)
|
)
|
||||||
|
from .pixivsketch import (
|
||||||
|
PixivSketchIE,
|
||||||
|
PixivSketchUserIE,
|
||||||
|
)
|
||||||
from .pladform import PladformIE
|
from .pladform import PladformIE
|
||||||
from .planetmarathi import PlanetMarathiIE
|
from .planetmarathi import PlanetMarathiIE
|
||||||
from .platzi import (
|
from .platzi import (
|
||||||
@@ -1149,6 +1179,10 @@ from .pokemon import (
|
|||||||
PokemonIE,
|
PokemonIE,
|
||||||
PokemonWatchIE,
|
PokemonWatchIE,
|
||||||
)
|
)
|
||||||
|
from .pokergo import (
|
||||||
|
PokerGoIE,
|
||||||
|
PokerGoCollectionIE,
|
||||||
|
)
|
||||||
from .polsatgo import PolsatGoIE
|
from .polsatgo import PolsatGoIE
|
||||||
from .polskieradio import (
|
from .polskieradio import (
|
||||||
PolskieRadioIE,
|
PolskieRadioIE,
|
||||||
@@ -1174,6 +1208,7 @@ from .pornhub import (
|
|||||||
from .pornotube import PornotubeIE
|
from .pornotube import PornotubeIE
|
||||||
from .pornovoisines import PornoVoisinesIE
|
from .pornovoisines import PornoVoisinesIE
|
||||||
from .pornoxo import PornoXOIE
|
from .pornoxo import PornoXOIE
|
||||||
|
from .pornez import PornezIE
|
||||||
from .puhutv import (
|
from .puhutv import (
|
||||||
PuhuTVIE,
|
PuhuTVIE,
|
||||||
PuhuTVSerieIE,
|
PuhuTVSerieIE,
|
||||||
@@ -1181,6 +1216,13 @@ from .puhutv import (
|
|||||||
from .presstv import PressTVIE
|
from .presstv import PressTVIE
|
||||||
from .projectveritas import ProjectVeritasIE
|
from .projectveritas import ProjectVeritasIE
|
||||||
from .prosiebensat1 import ProSiebenSat1IE
|
from .prosiebensat1 import ProSiebenSat1IE
|
||||||
|
from .prx import (
|
||||||
|
PRXStoryIE,
|
||||||
|
PRXSeriesIE,
|
||||||
|
PRXAccountIE,
|
||||||
|
PRXStoriesSearchIE,
|
||||||
|
PRXSeriesSearchIE
|
||||||
|
)
|
||||||
from .puls4 import Puls4IE
|
from .puls4 import Puls4IE
|
||||||
from .pyvideo import PyvideoIE
|
from .pyvideo import PyvideoIE
|
||||||
from .qqmusic import (
|
from .qqmusic import (
|
||||||
@@ -1217,9 +1259,10 @@ from .rai import (
|
|||||||
RaiPlayIE,
|
RaiPlayIE,
|
||||||
RaiPlayLiveIE,
|
RaiPlayLiveIE,
|
||||||
RaiPlayPlaylistIE,
|
RaiPlayPlaylistIE,
|
||||||
|
RaiPlaySoundIE,
|
||||||
|
RaiPlaySoundLiveIE,
|
||||||
|
RaiPlaySoundPlaylistIE,
|
||||||
RaiIE,
|
RaiIE,
|
||||||
RaiPlayRadioIE,
|
|
||||||
RaiPlayRadioPlaylistIE,
|
|
||||||
)
|
)
|
||||||
from .raywenderlich import (
|
from .raywenderlich import (
|
||||||
RayWenderlichIE,
|
RayWenderlichIE,
|
||||||
@@ -1274,6 +1317,12 @@ from .rtl2 import (
|
|||||||
RTL2YouIE,
|
RTL2YouIE,
|
||||||
RTL2YouSeriesIE,
|
RTL2YouSeriesIE,
|
||||||
)
|
)
|
||||||
|
from .rtnews import (
|
||||||
|
RTNewsIE,
|
||||||
|
RTDocumentryIE,
|
||||||
|
RTDocumentryPlaylistIE,
|
||||||
|
RuptlyIE,
|
||||||
|
)
|
||||||
from .rtp import RTPIE
|
from .rtp import RTPIE
|
||||||
from .rtrfm import RTRFMIE
|
from .rtrfm import RTRFMIE
|
||||||
from .rts import RTSIE
|
from .rts import RTSIE
|
||||||
@@ -1287,6 +1336,7 @@ from .rtve import (
|
|||||||
from .rtvnh import RTVNHIE
|
from .rtvnh import RTVNHIE
|
||||||
from .rtvs import RTVSIE
|
from .rtvs import RTVSIE
|
||||||
from .ruhd import RUHDIE
|
from .ruhd import RUHDIE
|
||||||
|
from .rule34video import Rule34VideoIE
|
||||||
from .rumble import (
|
from .rumble import (
|
||||||
RumbleEmbedIE,
|
RumbleEmbedIE,
|
||||||
RumbleChannelIE,
|
RumbleChannelIE,
|
||||||
@@ -1300,6 +1350,14 @@ from .rutube import (
|
|||||||
RutubePlaylistIE,
|
RutubePlaylistIE,
|
||||||
RutubeTagsIE,
|
RutubeTagsIE,
|
||||||
)
|
)
|
||||||
|
from .glomex import (
|
||||||
|
GlomexIE,
|
||||||
|
GlomexEmbedIE,
|
||||||
|
)
|
||||||
|
from .megatvcom import (
|
||||||
|
MegaTVComIE,
|
||||||
|
MegaTVComEmbedIE,
|
||||||
|
)
|
||||||
from .rutv import RUTVIE
|
from .rutv import RUTVIE
|
||||||
from .ruutu import RuutuIE
|
from .ruutu import RuutuIE
|
||||||
from .ruv import RuvIE
|
from .ruv import RuvIE
|
||||||
@@ -1488,7 +1546,12 @@ from .teachingchannel import TeachingChannelIE
|
|||||||
from .teamcoco import TeamcocoIE
|
from .teamcoco import TeamcocoIE
|
||||||
from .teamtreehouse import TeamTreeHouseIE
|
from .teamtreehouse import TeamTreeHouseIE
|
||||||
from .techtalks import TechTalksIE
|
from .techtalks import TechTalksIE
|
||||||
from .ted import TEDIE
|
from .ted import (
|
||||||
|
TedEmbedIE,
|
||||||
|
TedPlaylistIE,
|
||||||
|
TedSeriesIE,
|
||||||
|
TedTalkIE,
|
||||||
|
)
|
||||||
from .tele5 import Tele5IE
|
from .tele5 import Tele5IE
|
||||||
from .tele13 import Tele13IE
|
from .tele13 import Tele13IE
|
||||||
from .telebruxelles import TeleBruxellesIE
|
from .telebruxelles import TeleBruxellesIE
|
||||||
@@ -1534,6 +1597,9 @@ from .threeqsdn import ThreeQSDNIE
|
|||||||
from .tiktok import (
|
from .tiktok import (
|
||||||
TikTokIE,
|
TikTokIE,
|
||||||
TikTokUserIE,
|
TikTokUserIE,
|
||||||
|
TikTokSoundIE,
|
||||||
|
TikTokEffectIE,
|
||||||
|
TikTokTagIE,
|
||||||
DouyinIE,
|
DouyinIE,
|
||||||
)
|
)
|
||||||
from .tinypic import TinyPicIE
|
from .tinypic import TinyPicIE
|
||||||
@@ -1631,6 +1697,10 @@ from .tvnow import (
|
|||||||
TVNowAnnualIE,
|
TVNowAnnualIE,
|
||||||
TVNowShowIE,
|
TVNowShowIE,
|
||||||
)
|
)
|
||||||
|
from .tvopengr import (
|
||||||
|
TVOpenGrWatchIE,
|
||||||
|
TVOpenGrEmbedIE,
|
||||||
|
)
|
||||||
from .tvp import (
|
from .tvp import (
|
||||||
TVPEmbedIE,
|
TVPEmbedIE,
|
||||||
TVPIE,
|
TVPIE,
|
||||||
@@ -1684,6 +1754,7 @@ from .dlive import (
|
|||||||
DLiveVODIE,
|
DLiveVODIE,
|
||||||
DLiveStreamIE,
|
DLiveStreamIE,
|
||||||
)
|
)
|
||||||
|
from .drooble import DroobleIE
|
||||||
from .umg import UMGDeIE
|
from .umg import UMGDeIE
|
||||||
from .unistra import UnistraIE
|
from .unistra import UnistraIE
|
||||||
from .unity import UnityIE
|
from .unity import UnityIE
|
||||||
@@ -1758,6 +1829,7 @@ from .vimeo import (
|
|||||||
VimeoWatchLaterIE,
|
VimeoWatchLaterIE,
|
||||||
VHXEmbedIE,
|
VHXEmbedIE,
|
||||||
)
|
)
|
||||||
|
from .vimm import VimmIE
|
||||||
from .vimple import VimpleIE
|
from .vimple import VimpleIE
|
||||||
from .vine import (
|
from .vine import (
|
||||||
VineIE,
|
VineIE,
|
||||||
@@ -1930,6 +2002,7 @@ from .youtube import (
|
|||||||
YoutubeFavouritesIE,
|
YoutubeFavouritesIE,
|
||||||
YoutubeHistoryIE,
|
YoutubeHistoryIE,
|
||||||
YoutubeTabIE,
|
YoutubeTabIE,
|
||||||
|
YoutubeLivestreamEmbedIE,
|
||||||
YoutubePlaylistIE,
|
YoutubePlaylistIE,
|
||||||
YoutubeRecommendedIE,
|
YoutubeRecommendedIE,
|
||||||
YoutubeSearchDateIE,
|
YoutubeSearchDateIE,
|
||||||
|
|||||||
@@ -13,23 +13,25 @@ from ..compat import (
|
|||||||
)
|
)
|
||||||
from ..utils import (
|
from ..utils import (
|
||||||
clean_html,
|
clean_html,
|
||||||
|
determine_ext,
|
||||||
error_to_compat_str,
|
error_to_compat_str,
|
||||||
ExtractorError,
|
ExtractorError,
|
||||||
float_or_none,
|
float_or_none,
|
||||||
get_element_by_id,
|
get_element_by_id,
|
||||||
int_or_none,
|
int_or_none,
|
||||||
js_to_json,
|
js_to_json,
|
||||||
limit_length,
|
|
||||||
merge_dicts,
|
merge_dicts,
|
||||||
network_exceptions,
|
network_exceptions,
|
||||||
parse_count,
|
parse_count,
|
||||||
parse_qs,
|
parse_qs,
|
||||||
qualities,
|
qualities,
|
||||||
sanitized_Request,
|
sanitized_Request,
|
||||||
|
traverse_obj,
|
||||||
try_get,
|
try_get,
|
||||||
url_or_none,
|
url_or_none,
|
||||||
urlencode_postdata,
|
urlencode_postdata,
|
||||||
urljoin,
|
urljoin,
|
||||||
|
variadic,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -163,7 +165,7 @@ class FacebookIE(InfoExtractor):
|
|||||||
'info_dict': {
|
'info_dict': {
|
||||||
'id': '1417995061575415',
|
'id': '1417995061575415',
|
||||||
'ext': 'mp4',
|
'ext': 'mp4',
|
||||||
'title': 'Yaroslav Korpan - Довгоочікуване відео',
|
'title': 'Ukrainian Scientists Worldwide | Довгоочікуване відео',
|
||||||
'description': 'Довгоочікуване відео',
|
'description': 'Довгоочікуване відео',
|
||||||
'timestamp': 1486648771,
|
'timestamp': 1486648771,
|
||||||
'upload_date': '20170209',
|
'upload_date': '20170209',
|
||||||
@@ -194,8 +196,8 @@ class FacebookIE(InfoExtractor):
|
|||||||
'info_dict': {
|
'info_dict': {
|
||||||
'id': '202882990186699',
|
'id': '202882990186699',
|
||||||
'ext': 'mp4',
|
'ext': 'mp4',
|
||||||
'title': 'Elisabeth Ahtn - Hello? Yes your uber ride is here\n* Jukin...',
|
'title': 'birb (O v O") | Hello? Yes your uber ride is here',
|
||||||
'description': 'Hello? Yes your uber ride is here\n* Jukin Media Verified *\nFind this video and others like it by visiting...',
|
'description': 'Hello? Yes your uber ride is here * Jukin Media Verified * Find this video and others like it by visiting...',
|
||||||
'timestamp': 1486035513,
|
'timestamp': 1486035513,
|
||||||
'upload_date': '20170202',
|
'upload_date': '20170202',
|
||||||
'uploader': 'Elisabeth Ahtn',
|
'uploader': 'Elisabeth Ahtn',
|
||||||
@@ -397,28 +399,31 @@ class FacebookIE(InfoExtractor):
|
|||||||
url.replace('://m.facebook.com/', '://www.facebook.com/'), video_id)
|
url.replace('://m.facebook.com/', '://www.facebook.com/'), video_id)
|
||||||
|
|
||||||
def extract_metadata(webpage):
|
def extract_metadata(webpage):
|
||||||
video_title = self._html_search_regex(
|
post_data = [self._parse_json(j, video_id, fatal=False) for j in re.findall(
|
||||||
r'<h2\s+[^>]*class="uiHeaderTitle"[^>]*>([^<]*)</h2>', webpage,
|
r'handleWithCustomApplyEach\(\s*ScheduledApplyEach\s*,\s*(\{.+?\})\s*\);', webpage)]
|
||||||
'title', default=None)
|
post = traverse_obj(post_data, (
|
||||||
if not video_title:
|
..., 'require', ..., ..., ..., '__bbox', 'result', 'data'), expected_type=dict) or []
|
||||||
video_title = self._html_search_regex(
|
media = [m for m in traverse_obj(post, (..., 'attachments', ..., 'media'), expected_type=dict) or []
|
||||||
r'(?s)<span class="fbPhotosPhotoCaption".*?id="fbPhotoPageCaption"><span class="hasCaption">(.*?)</span>',
|
if str(m.get('id')) == video_id and m.get('__typename') == 'Video']
|
||||||
webpage, 'alternative title', default=None)
|
title = traverse_obj(media, (..., 'title', 'text'), get_all=False)
|
||||||
if not video_title:
|
description = traverse_obj(media, (
|
||||||
video_title = self._html_search_meta(
|
..., 'creation_story', 'comet_sections', 'message', 'story', 'message', 'text'), get_all=False)
|
||||||
['og:title', 'twitter:title', 'description'],
|
uploader_data = (traverse_obj(media, (..., 'owner'), get_all=False)
|
||||||
webpage, 'title', default=None)
|
or traverse_obj(post, (..., 'node', 'actors', ...), get_all=False) or {})
|
||||||
if video_title:
|
|
||||||
video_title = limit_length(video_title, 80)
|
page_title = title or self._html_search_regex((
|
||||||
else:
|
r'<h2\s+[^>]*class="uiHeaderTitle"[^>]*>(?P<content>[^<]*)</h2>',
|
||||||
video_title = 'Facebook video #%s' % video_id
|
r'(?s)<span class="fbPhotosPhotoCaption".*?id="fbPhotoPageCaption"><span class="hasCaption">(?P<content>.*?)</span>',
|
||||||
description = self._html_search_meta(
|
self._meta_regex('og:title'), self._meta_regex('twitter:title'), r'<title>(?P<content>.+?)</title>'
|
||||||
|
), webpage, 'title', default=None, group='content')
|
||||||
|
description = description or self._html_search_meta(
|
||||||
['description', 'og:description', 'twitter:description'],
|
['description', 'og:description', 'twitter:description'],
|
||||||
webpage, 'description', default=None)
|
webpage, 'description', default=None)
|
||||||
uploader = clean_html(get_element_by_id(
|
uploader = uploader_data.get('name') or (
|
||||||
'fbPhotoPageAuthorName', webpage)) or self._search_regex(
|
clean_html(get_element_by_id('fbPhotoPageAuthorName', webpage))
|
||||||
r'ownerName\s*:\s*"([^"]+)"', webpage, 'uploader',
|
or self._search_regex(
|
||||||
default=None) or self._og_search_title(webpage, fatal=False)
|
(r'ownerName\s*:\s*"([^"]+)"', *self._og_regexes('title')), webpage, 'uploader', fatal=False))
|
||||||
|
|
||||||
timestamp = int_or_none(self._search_regex(
|
timestamp = int_or_none(self._search_regex(
|
||||||
r'<abbr[^>]+data-utime=["\'](\d+)', webpage,
|
r'<abbr[^>]+data-utime=["\'](\d+)', webpage,
|
||||||
'timestamp', default=None))
|
'timestamp', default=None))
|
||||||
@@ -433,17 +438,17 @@ class FacebookIE(InfoExtractor):
|
|||||||
r'\bviewCount\s*:\s*["\']([\d,.]+)', webpage, 'view count',
|
r'\bviewCount\s*:\s*["\']([\d,.]+)', webpage, 'view count',
|
||||||
default=None))
|
default=None))
|
||||||
info_dict = {
|
info_dict = {
|
||||||
'title': video_title,
|
|
||||||
'description': description,
|
'description': description,
|
||||||
'uploader': uploader,
|
'uploader': uploader,
|
||||||
|
'uploader_id': uploader_data.get('id'),
|
||||||
'timestamp': timestamp,
|
'timestamp': timestamp,
|
||||||
'thumbnail': thumbnail,
|
'thumbnail': thumbnail,
|
||||||
'view_count': view_count,
|
'view_count': view_count,
|
||||||
}
|
}
|
||||||
|
|
||||||
info_json_ld = self._search_json_ld(webpage, video_id, default={})
|
info_json_ld = self._search_json_ld(webpage, video_id, default={})
|
||||||
if info_json_ld.get('title'):
|
info_json_ld['title'] = (re.sub(r'\s*\|\s*Facebook$', '', title or info_json_ld.get('title') or page_title or '')
|
||||||
info_json_ld['title'] = limit_length(
|
or (description or '').replace('\n', ' ') or f'Facebook video #{video_id}')
|
||||||
re.sub(r'\s*\|\s*Facebook$', '', info_json_ld['title']), 80)
|
|
||||||
return merge_dicts(info_json_ld, info_dict)
|
return merge_dicts(info_json_ld, info_dict)
|
||||||
|
|
||||||
video_data = None
|
video_data = None
|
||||||
@@ -510,15 +515,19 @@ class FacebookIE(InfoExtractor):
|
|||||||
def parse_graphql_video(video):
|
def parse_graphql_video(video):
|
||||||
formats = []
|
formats = []
|
||||||
q = qualities(['sd', 'hd'])
|
q = qualities(['sd', 'hd'])
|
||||||
for (suffix, format_id) in [('', 'sd'), ('_quality_hd', 'hd')]:
|
for key, format_id in (('playable_url', 'sd'), ('playable_url_quality_hd', 'hd'),
|
||||||
playable_url = video.get('playable_url' + suffix)
|
('playable_url_dash', '')):
|
||||||
|
playable_url = video.get(key)
|
||||||
if not playable_url:
|
if not playable_url:
|
||||||
continue
|
continue
|
||||||
formats.append({
|
if determine_ext(playable_url) == 'mpd':
|
||||||
'format_id': format_id,
|
formats.extend(self._extract_mpd_formats(playable_url, video_id))
|
||||||
'quality': q(format_id),
|
else:
|
||||||
'url': playable_url,
|
formats.append({
|
||||||
})
|
'format_id': format_id,
|
||||||
|
'quality': q(format_id),
|
||||||
|
'url': playable_url,
|
||||||
|
})
|
||||||
extract_dash_manifest(video, formats)
|
extract_dash_manifest(video, formats)
|
||||||
process_formats(formats)
|
process_formats(formats)
|
||||||
v_id = video.get('videoId') or video.get('id') or video_id
|
v_id = video.get('videoId') or video.get('id') or video_id
|
||||||
@@ -546,22 +555,15 @@ class FacebookIE(InfoExtractor):
|
|||||||
if media.get('__typename') == 'Video':
|
if media.get('__typename') == 'Video':
|
||||||
return parse_graphql_video(media)
|
return parse_graphql_video(media)
|
||||||
|
|
||||||
nodes = data.get('nodes') or []
|
nodes = variadic(traverse_obj(data, 'nodes', 'node') or [])
|
||||||
node = data.get('node') or {}
|
attachments = traverse_obj(nodes, (
|
||||||
if not nodes and node:
|
..., 'comet_sections', 'content', 'story', (None, 'attached_story'), 'attachments',
|
||||||
nodes.append(node)
|
..., ('styles', 'style_type_renderer'), 'attachment'), expected_type=dict) or []
|
||||||
for node in nodes:
|
for attachment in attachments:
|
||||||
story = try_get(node, lambda x: x['comet_sections']['content']['story'], dict) or {}
|
ns = try_get(attachment, lambda x: x['all_subattachments']['nodes'], list) or []
|
||||||
attachments = try_get(story, [
|
for n in ns:
|
||||||
lambda x: x['attached_story']['attachments'],
|
parse_attachment(n)
|
||||||
lambda x: x['attachments']
|
parse_attachment(attachment)
|
||||||
], list) or []
|
|
||||||
for attachment in attachments:
|
|
||||||
attachment = try_get(attachment, lambda x: x['style_type_renderer']['attachment'], dict)
|
|
||||||
ns = try_get(attachment, lambda x: x['all_subattachments']['nodes'], list) or []
|
|
||||||
for n in ns:
|
|
||||||
parse_attachment(n)
|
|
||||||
parse_attachment(attachment)
|
|
||||||
|
|
||||||
edges = try_get(data, lambda x: x['mediaset']['currMedia']['edges'], list) or []
|
edges = try_get(data, lambda x: x['mediaset']['currMedia']['edges'], list) or []
|
||||||
for edge in edges:
|
for edge in edges:
|
||||||
@@ -730,6 +732,7 @@ class FacebookPluginsVideoIE(InfoExtractor):
|
|||||||
'info_dict': {
|
'info_dict': {
|
||||||
'id': '10154383743583686',
|
'id': '10154383743583686',
|
||||||
'ext': 'mp4',
|
'ext': 'mp4',
|
||||||
|
# TODO: Fix title, uploader
|
||||||
'title': 'What to do during the haze?',
|
'title': 'What to do during the haze?',
|
||||||
'uploader': 'Gov.sg',
|
'uploader': 'Gov.sg',
|
||||||
'upload_date': '20160826',
|
'upload_date': '20160826',
|
||||||
|
|||||||
@@ -7,6 +7,7 @@ from ..compat import (
|
|||||||
)
|
)
|
||||||
from ..utils import (
|
from ..utils import (
|
||||||
ExtractorError,
|
ExtractorError,
|
||||||
|
format_field,
|
||||||
int_or_none,
|
int_or_none,
|
||||||
qualities,
|
qualities,
|
||||||
)
|
)
|
||||||
@@ -95,7 +96,7 @@ class FlickrIE(InfoExtractor):
|
|||||||
owner = video_info.get('owner', {})
|
owner = video_info.get('owner', {})
|
||||||
uploader_id = owner.get('nsid')
|
uploader_id = owner.get('nsid')
|
||||||
uploader_path = owner.get('path_alias') or uploader_id
|
uploader_path = owner.get('path_alias') or uploader_id
|
||||||
uploader_url = 'https://www.flickr.com/photos/%s/' % uploader_path if uploader_path else None
|
uploader_url = format_field(uploader_path, template='https://www.flickr.com/photos/%s/')
|
||||||
|
|
||||||
return {
|
return {
|
||||||
'id': video_id,
|
'id': video_id,
|
||||||
|
|||||||
+31
-8
@@ -4,7 +4,7 @@ from __future__ import unicode_literals
|
|||||||
import json
|
import json
|
||||||
import uuid
|
import uuid
|
||||||
|
|
||||||
from .adobepass import AdobePassIE
|
from .common import InfoExtractor
|
||||||
from ..compat import (
|
from ..compat import (
|
||||||
compat_HTTPError,
|
compat_HTTPError,
|
||||||
compat_str,
|
compat_str,
|
||||||
@@ -20,7 +20,7 @@ from ..utils import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
class FOXIE(AdobePassIE):
|
class FOXIE(InfoExtractor):
|
||||||
_VALID_URL = r'https?://(?:www\.)?fox\.com/watch/(?P<id>[\da-fA-F]+)'
|
_VALID_URL = r'https?://(?:www\.)?fox\.com/watch/(?P<id>[\da-fA-F]+)'
|
||||||
_TESTS = [{
|
_TESTS = [{
|
||||||
# clip
|
# clip
|
||||||
@@ -37,6 +37,7 @@ class FOXIE(AdobePassIE):
|
|||||||
'creator': 'FOX',
|
'creator': 'FOX',
|
||||||
'series': 'Gotham',
|
'series': 'Gotham',
|
||||||
'age_limit': 14,
|
'age_limit': 14,
|
||||||
|
'episode': 'Aftermath: Bruce Wayne Develops Into The Dark Knight'
|
||||||
},
|
},
|
||||||
'params': {
|
'params': {
|
||||||
'skip_download': True,
|
'skip_download': True,
|
||||||
@@ -46,14 +47,15 @@ class FOXIE(AdobePassIE):
|
|||||||
'url': 'https://www.fox.com/watch/087036ca7f33c8eb79b08152b4dd75c1/',
|
'url': 'https://www.fox.com/watch/087036ca7f33c8eb79b08152b4dd75c1/',
|
||||||
'only_matching': True,
|
'only_matching': True,
|
||||||
}, {
|
}, {
|
||||||
# episode, geo-restricted, tv provided required
|
# sports event, geo-restricted
|
||||||
'url': 'https://www.fox.com/watch/30056b295fb57f7452aeeb4920bc3024/',
|
'url': 'https://www.fox.com/watch/b057484dade738d1f373b3e46216fa2c/',
|
||||||
'only_matching': True,
|
'only_matching': True,
|
||||||
}]
|
}]
|
||||||
_GEO_BYPASS = False
|
_GEO_BYPASS = False
|
||||||
_HOME_PAGE_URL = 'https://www.fox.com/'
|
_HOME_PAGE_URL = 'https://www.fox.com/'
|
||||||
_API_KEY = 'abdcbed02c124d393b39e818a4312055'
|
_API_KEY = '6E9S4bmcoNnZwVLOHywOv8PJEdu76cM9'
|
||||||
_access_token = None
|
_access_token = None
|
||||||
|
_device_id = compat_str(uuid.uuid4())
|
||||||
|
|
||||||
def _call_api(self, path, video_id, data=None):
|
def _call_api(self, path, video_id, data=None):
|
||||||
headers = {
|
headers = {
|
||||||
@@ -63,7 +65,7 @@ class FOXIE(AdobePassIE):
|
|||||||
headers['Authorization'] = 'Bearer ' + self._access_token
|
headers['Authorization'] = 'Bearer ' + self._access_token
|
||||||
try:
|
try:
|
||||||
return self._download_json(
|
return self._download_json(
|
||||||
'https://api2.fox.com/v2.0/' + path,
|
'https://api3.fox.com/v2.0/' + path,
|
||||||
video_id, data=data, headers=headers)
|
video_id, data=data, headers=headers)
|
||||||
except ExtractorError as e:
|
except ExtractorError as e:
|
||||||
if isinstance(e.cause, compat_HTTPError) and e.cause.code == 403:
|
if isinstance(e.cause, compat_HTTPError) and e.cause.code == 403:
|
||||||
@@ -87,16 +89,37 @@ class FOXIE(AdobePassIE):
|
|||||||
if not self._access_token:
|
if not self._access_token:
|
||||||
self._access_token = self._call_api(
|
self._access_token = self._call_api(
|
||||||
'login', None, json.dumps({
|
'login', None, json.dumps({
|
||||||
'deviceId': compat_str(uuid.uuid4()),
|
'deviceId': self._device_id,
|
||||||
}).encode())['accessToken']
|
}).encode())['accessToken']
|
||||||
|
|
||||||
def _real_extract(self, url):
|
def _real_extract(self, url):
|
||||||
video_id = self._match_id(url)
|
video_id = self._match_id(url)
|
||||||
|
|
||||||
video = self._call_api('vodplayer/' + video_id, video_id)
|
self._access_token = self._call_api(
|
||||||
|
'previewpassmvpd?device_id=%s&mvpd_id=TempPass_fbcfox_60min' % self._device_id,
|
||||||
|
video_id)['accessToken']
|
||||||
|
|
||||||
|
video = self._call_api('watch', video_id, data=json.dumps({
|
||||||
|
'capabilities': ['drm/widevine', 'fsdk/yo'],
|
||||||
|
'deviceWidth': 1280,
|
||||||
|
'deviceHeight': 720,
|
||||||
|
'maxRes': '720p',
|
||||||
|
'os': 'macos',
|
||||||
|
'osv': '',
|
||||||
|
'provider': {
|
||||||
|
'freewheel': {'did': self._device_id},
|
||||||
|
'vdms': {'rays': ''},
|
||||||
|
'dmp': {'kuid': '', 'seg': ''}
|
||||||
|
},
|
||||||
|
'playlist': '',
|
||||||
|
'privacy': {'us': '1---'},
|
||||||
|
'siteSection': '',
|
||||||
|
'streamType': 'vod',
|
||||||
|
'streamId': video_id}).encode('utf-8'))
|
||||||
|
|
||||||
title = video['name']
|
title = video['name']
|
||||||
release_url = video['url']
|
release_url = video['url']
|
||||||
|
|
||||||
try:
|
try:
|
||||||
m3u8_url = self._download_json(release_url, video_id)['playURL']
|
m3u8_url = self._download_json(release_url, video_id)['playURL']
|
||||||
except ExtractorError as e:
|
except ExtractorError as e:
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ from ..utils import (
|
|||||||
|
|
||||||
|
|
||||||
class FunkIE(InfoExtractor):
|
class FunkIE(InfoExtractor):
|
||||||
_VALID_URL = r'https?://(?:www\.)?funk\.net/(?:channel|playlist)/[^/]+/(?P<display_id>[0-9a-z-]+)-(?P<id>\d+)'
|
_VALID_URL = r'https?://(?:www\.|origin\.)?funk\.net/(?:channel|playlist)/[^/]+/(?P<display_id>[0-9a-z-]+)-(?P<id>\d+)'
|
||||||
_TESTS = [{
|
_TESTS = [{
|
||||||
'url': 'https://www.funk.net/channel/ba-793/die-lustigsten-instrumente-aus-dem-internet-teil-2-1155821',
|
'url': 'https://www.funk.net/channel/ba-793/die-lustigsten-instrumente-aus-dem-internet-teil-2-1155821',
|
||||||
'md5': '8dd9d9ab59b4aa4173b3197f2ea48e81',
|
'md5': '8dd9d9ab59b4aa4173b3197f2ea48e81',
|
||||||
|
|||||||
@@ -7,6 +7,7 @@ from .common import InfoExtractor
|
|||||||
from ..compat import compat_urllib_parse_unquote
|
from ..compat import compat_urllib_parse_unquote
|
||||||
from ..utils import (
|
from ..utils import (
|
||||||
determine_ext,
|
determine_ext,
|
||||||
|
format_field,
|
||||||
int_or_none,
|
int_or_none,
|
||||||
str_or_none,
|
str_or_none,
|
||||||
traverse_obj,
|
traverse_obj,
|
||||||
@@ -86,7 +87,7 @@ class GameJoltBaseIE(InfoExtractor):
|
|||||||
'display_id': post_data.get('slug'),
|
'display_id': post_data.get('slug'),
|
||||||
'uploader': user_data.get('display_name') or user_data.get('name'),
|
'uploader': user_data.get('display_name') or user_data.get('name'),
|
||||||
'uploader_id': user_data.get('username'),
|
'uploader_id': user_data.get('username'),
|
||||||
'uploader_url': 'https://gamejolt.com' + user_data['url'] if user_data.get('url') else None,
|
'uploader_url': format_field(user_data, 'url', 'https://gamejolt.com%s'),
|
||||||
'categories': [try_get(category, lambda x: '%s - %s' % (x['community']['name'], x['channel'].get('display_title') or x['channel']['title']))
|
'categories': [try_get(category, lambda x: '%s - %s' % (x['community']['name'], x['channel'].get('display_title') or x['channel']['title']))
|
||||||
for category in post_data.get('communities' or [])],
|
for category in post_data.get('communities' or [])],
|
||||||
'tags': traverse_obj(
|
'tags': traverse_obj(
|
||||||
|
|||||||
+168
-19
@@ -28,6 +28,7 @@ from ..utils import (
|
|||||||
mimetype2ext,
|
mimetype2ext,
|
||||||
orderedSet,
|
orderedSet,
|
||||||
parse_duration,
|
parse_duration,
|
||||||
|
parse_resolution,
|
||||||
sanitized_Request,
|
sanitized_Request,
|
||||||
smuggle_url,
|
smuggle_url,
|
||||||
unescapeHTML,
|
unescapeHTML,
|
||||||
@@ -100,6 +101,8 @@ from .ustream import UstreamIE
|
|||||||
from .arte import ArteTVEmbedIE
|
from .arte import ArteTVEmbedIE
|
||||||
from .videopress import VideoPressIE
|
from .videopress import VideoPressIE
|
||||||
from .rutube import RutubeIE
|
from .rutube import RutubeIE
|
||||||
|
from .glomex import GlomexEmbedIE
|
||||||
|
from .megatvcom import MegaTVComEmbedIE
|
||||||
from .limelight import LimelightBaseIE
|
from .limelight import LimelightBaseIE
|
||||||
from .anvato import AnvatoIE
|
from .anvato import AnvatoIE
|
||||||
from .washingtonpost import WashingtonPostIE
|
from .washingtonpost import WashingtonPostIE
|
||||||
@@ -112,6 +115,7 @@ from .channel9 import Channel9IE
|
|||||||
from .vshare import VShareIE
|
from .vshare import VShareIE
|
||||||
from .mediasite import MediasiteIE
|
from .mediasite import MediasiteIE
|
||||||
from .springboardplatform import SpringboardPlatformIE
|
from .springboardplatform import SpringboardPlatformIE
|
||||||
|
from .ted import TedEmbedIE
|
||||||
from .yapfiles import YapFilesIE
|
from .yapfiles import YapFilesIE
|
||||||
from .vice import ViceIE
|
from .vice import ViceIE
|
||||||
from .xfileshare import XFileShareIE
|
from .xfileshare import XFileShareIE
|
||||||
@@ -135,8 +139,11 @@ from .arcpublishing import ArcPublishingIE
|
|||||||
from .medialaan import MedialaanIE
|
from .medialaan import MedialaanIE
|
||||||
from .simplecast import SimplecastIE
|
from .simplecast import SimplecastIE
|
||||||
from .wimtv import WimTVIE
|
from .wimtv import WimTVIE
|
||||||
|
from .tvopengr import TVOpenGrEmbedIE
|
||||||
from .tvp import TVPEmbedIE
|
from .tvp import TVPEmbedIE
|
||||||
from .blogger import BloggerIE
|
from .blogger import BloggerIE
|
||||||
|
from .mainstreaming import MainStreamingIE
|
||||||
|
from .gfycat import GfycatIE
|
||||||
|
|
||||||
|
|
||||||
class GenericIE(InfoExtractor):
|
class GenericIE(InfoExtractor):
|
||||||
@@ -1869,6 +1876,53 @@ class GenericIE(InfoExtractor):
|
|||||||
},
|
},
|
||||||
'add_ie': [RutubeIE.ie_key()],
|
'add_ie': [RutubeIE.ie_key()],
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
# glomex:embed
|
||||||
|
'url': 'https://www.skai.gr/news/world/iatrikos-syllogos-tourkias-to-turkovac-aplo-dialyma-erntogan-eiste-apateones-kai-pseytes',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'v-ch2nkhcirwc9-sf',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'md5:786e1e24e06c55993cee965ef853a0c1',
|
||||||
|
'description': 'md5:8b517a61d577efe7e36fde72fd535995',
|
||||||
|
'timestamp': 1641885019,
|
||||||
|
'upload_date': '20220111',
|
||||||
|
'duration': 460000,
|
||||||
|
'thumbnail': 'https://i3thumbs.glomex.com/dC1idjJwdndiMjRzeGwvMjAyMi8wMS8xMS8wNy8xMF8zNV82MWRkMmQ2YmU5ZTgyLmpwZw==/profile:player-960x540',
|
||||||
|
},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
# megatvcom:embed
|
||||||
|
'url': 'https://www.in.gr/2021/12/18/greece/apokalypsi-mega-poios-parelave-tin-ereyna-tsiodra-ek-merous-tis-kyvernisis-o-prothypourgos-telika-gnorize/',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'apokalypsi-mega-poios-parelave-tin-ereyna-tsiodra-ek-merous-tis-kyvernisis-o-prothypourgos-telika-gnorize',
|
||||||
|
'title': 'md5:5e569cf996ec111057c2764ec272848f',
|
||||||
|
},
|
||||||
|
'playlist': [{
|
||||||
|
'md5': '1afa26064ff00ccb91617957dbc73dc1',
|
||||||
|
'info_dict': {
|
||||||
|
'ext': 'mp4',
|
||||||
|
'id': '564916',
|
||||||
|
'display_id': 'md5:6cdf22d3a2e7bacb274b7295089a1770',
|
||||||
|
'title': 'md5:33b9dd39584685b62873043670eb52a6',
|
||||||
|
'description': 'md5:c1db7310f390518ac36dd69d947ef1a1',
|
||||||
|
'timestamp': 1639753145,
|
||||||
|
'upload_date': '20211217',
|
||||||
|
'thumbnail': 'https://www.megatv.com/wp-content/uploads/2021/12/prezerakos-1024x597.jpg',
|
||||||
|
},
|
||||||
|
}, {
|
||||||
|
'md5': '4a1c220695f1ef865a8b7966a53e2474',
|
||||||
|
'info_dict': {
|
||||||
|
'ext': 'mp4',
|
||||||
|
'id': '564905',
|
||||||
|
'display_id': 'md5:ead15695e485e649aed2b81ebd699b88',
|
||||||
|
'title': 'md5:2b71fd54249a3ca34609fe39ae31c47b',
|
||||||
|
'description': 'md5:c42e12f638d0a97d6de4508e2c4df982',
|
||||||
|
'timestamp': 1639753047,
|
||||||
|
'upload_date': '20211217',
|
||||||
|
'thumbnail': 'https://www.megatv.com/wp-content/uploads/2021/12/tsiodras-mitsotakis-1024x545.jpg',
|
||||||
|
},
|
||||||
|
}]
|
||||||
|
},
|
||||||
{
|
{
|
||||||
# ThePlatform embedded with whitespaces in URLs
|
# ThePlatform embedded with whitespaces in URLs
|
||||||
'url': 'http://www.golfchannel.com/topics/shows/golftalkcentral.htm',
|
'url': 'http://www.golfchannel.com/topics/shows/golftalkcentral.htm',
|
||||||
@@ -2174,6 +2228,22 @@ class GenericIE(InfoExtractor):
|
|||||||
'skip_download': True,
|
'skip_download': True,
|
||||||
},
|
},
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
# tvopengr:embed
|
||||||
|
'url': 'https://www.ethnos.gr/World/article/190604/hparosiaxekinoynoisynomiliessthgeneyhmethskiatoypolemoypanoapothnoykrania',
|
||||||
|
'md5': 'eb0c3995d0a6f18f6538c8e057865d7d',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '101119',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'display_id': 'oikarpoitondiapragmateyseonhparosias',
|
||||||
|
'title': 'md5:b979f4d640c568617d6547035528a149',
|
||||||
|
'description': 'md5:e54fc1977c7159b01cc11cd7d9d85550',
|
||||||
|
'timestamp': 1641772800,
|
||||||
|
'upload_date': '20220110',
|
||||||
|
'thumbnail': 'https://opentv-static.siliconweb.com/imgHandler/1920/70bc39fa-895b-4918-a364-c39d2135fc6d.jpg',
|
||||||
|
|
||||||
|
}
|
||||||
|
},
|
||||||
{
|
{
|
||||||
# blogger embed
|
# blogger embed
|
||||||
'url': 'https://blog.tomeuvizoso.net/2019/01/a-panfrost-milestone.html',
|
'url': 'https://blog.tomeuvizoso.net/2019/01/a-panfrost-milestone.html',
|
||||||
@@ -2344,6 +2414,18 @@ class GenericIE(InfoExtractor):
|
|||||||
'thumbnail': 'https://bogmedia.org/contents/videos_screenshots/21000/21217/preview_480p.mp4.jpg',
|
'thumbnail': 'https://bogmedia.org/contents/videos_screenshots/21000/21217/preview_480p.mp4.jpg',
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
# KVS Player (for sites that serve kt_player.js via non-https urls)
|
||||||
|
'url': 'http://www.camhub.world/embed/389508',
|
||||||
|
'md5': 'fbe89af4cfb59c8fd9f34a202bb03e32',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '389508',
|
||||||
|
'display_id': 'syren-de-mer-onlyfans-05-07-2020have-a-happy-safe-holiday5f014e68a220979bdb8cd-source',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'Syren De Mer onlyfans_05-07-2020Have_a_happy_safe_holiday5f014e68a220979bdb8cd_source / Embed плеер',
|
||||||
|
'thumbnail': 'http://www.camhub.world/contents/videos_screenshots/389000/389508/preview.mp4.jpg',
|
||||||
|
}
|
||||||
|
},
|
||||||
{
|
{
|
||||||
# Reddit-hosted video that will redirect and be processed by RedditIE
|
# Reddit-hosted video that will redirect and be processed by RedditIE
|
||||||
# Redirects to https://www.reddit.com/r/videos/comments/6rrwyj/that_small_heart_attack/
|
# Redirects to https://www.reddit.com/r/videos/comments/6rrwyj/that_small_heart_attack/
|
||||||
@@ -2370,8 +2452,47 @@ class GenericIE(InfoExtractor):
|
|||||||
'timestamp': 1636788683.0,
|
'timestamp': 1636788683.0,
|
||||||
'upload_date': '20211113'
|
'upload_date': '20211113'
|
||||||
}
|
}
|
||||||
|
},
|
||||||
|
{
|
||||||
|
# MainStreaming player
|
||||||
|
'url': 'https://www.lactv.it/2021/10/03/lac-news24-la-settimana-03-10-2021/',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'EUlZfGWkGpOd',
|
||||||
|
'title': 'La Settimana ',
|
||||||
|
'description': '03 Ottobre ore 02:00',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'live_status': 'not_live',
|
||||||
|
'thumbnail': r're:https?://[A-Za-z0-9-]*\.msvdn.net/image/\w+/poster',
|
||||||
|
'duration': 1512
|
||||||
|
}
|
||||||
|
},
|
||||||
|
{
|
||||||
|
# Multiple gfycat iframe embeds
|
||||||
|
'url': 'https://www.gezip.net/bbs/board.php?bo_table=entertaine&wr_id=613422',
|
||||||
|
'info_dict': {
|
||||||
|
'title': '재이, 윤, 세은 황금 드레스를 입고 빛난다',
|
||||||
|
'id': 'board'
|
||||||
|
},
|
||||||
|
'playlist_count': 8,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
# Multiple gfycat gifs (direct links)
|
||||||
|
'url': 'https://www.gezip.net/bbs/board.php?bo_table=entertaine&wr_id=612199',
|
||||||
|
'info_dict': {
|
||||||
|
'title': '옳게 된 크롭 니트 스테이씨 아이사',
|
||||||
|
'id': 'board'
|
||||||
|
},
|
||||||
|
'playlist_count': 6
|
||||||
|
},
|
||||||
|
{
|
||||||
|
# Multiple gfycat embeds, with uppercase "IFR" in urls
|
||||||
|
'url': 'https://kkzz.kr/?vid=2295',
|
||||||
|
'info_dict': {
|
||||||
|
'title': '지방시 앰버서더 에스파 카리나 움짤',
|
||||||
|
'id': '?vid=2295'
|
||||||
|
},
|
||||||
|
'playlist_count': 9
|
||||||
}
|
}
|
||||||
#
|
|
||||||
]
|
]
|
||||||
|
|
||||||
def report_following_redirect(self, new_url):
|
def report_following_redirect(self, new_url):
|
||||||
@@ -3071,10 +3192,9 @@ class GenericIE(InfoExtractor):
|
|||||||
return self.url_result(mobj.group('url'), 'Tvigle')
|
return self.url_result(mobj.group('url'), 'Tvigle')
|
||||||
|
|
||||||
# Look for embedded TED player
|
# Look for embedded TED player
|
||||||
mobj = re.search(
|
ted_urls = TedEmbedIE._extract_urls(webpage)
|
||||||
r'<iframe[^>]+?src=(["\'])(?P<url>https?://embed(?:-ssl)?\.ted\.com/.+?)\1', webpage)
|
if ted_urls:
|
||||||
if mobj is not None:
|
return self.playlist_from_matches(ted_urls, video_id, video_title, ie=TedEmbedIE.ie_key())
|
||||||
return self.url_result(mobj.group('url'), 'TED')
|
|
||||||
|
|
||||||
# Look for embedded Ustream videos
|
# Look for embedded Ustream videos
|
||||||
ustream_url = UstreamIE._extract_url(webpage)
|
ustream_url = UstreamIE._extract_url(webpage)
|
||||||
@@ -3410,6 +3530,18 @@ class GenericIE(InfoExtractor):
|
|||||||
return self.playlist_from_matches(
|
return self.playlist_from_matches(
|
||||||
rutube_urls, video_id, video_title, ie=RutubeIE.ie_key())
|
rutube_urls, video_id, video_title, ie=RutubeIE.ie_key())
|
||||||
|
|
||||||
|
# Look for Glomex embeds
|
||||||
|
glomex_urls = list(GlomexEmbedIE._extract_urls(webpage, url))
|
||||||
|
if glomex_urls:
|
||||||
|
return self.playlist_from_matches(
|
||||||
|
glomex_urls, video_id, video_title, ie=GlomexEmbedIE.ie_key())
|
||||||
|
|
||||||
|
# Look for megatv.com embeds
|
||||||
|
megatvcom_urls = list(MegaTVComEmbedIE._extract_urls(webpage))
|
||||||
|
if megatvcom_urls:
|
||||||
|
return self.playlist_from_matches(
|
||||||
|
megatvcom_urls, video_id, video_title, ie=MegaTVComEmbedIE.ie_key())
|
||||||
|
|
||||||
# Look for WashingtonPost embeds
|
# Look for WashingtonPost embeds
|
||||||
wapo_urls = WashingtonPostIE._extract_urls(webpage)
|
wapo_urls = WashingtonPostIE._extract_urls(webpage)
|
||||||
if wapo_urls:
|
if wapo_urls:
|
||||||
@@ -3556,10 +3688,25 @@ class GenericIE(InfoExtractor):
|
|||||||
return self.playlist_from_matches(
|
return self.playlist_from_matches(
|
||||||
rumble_urls, video_id, video_title, ie=RumbleEmbedIE.ie_key())
|
rumble_urls, video_id, video_title, ie=RumbleEmbedIE.ie_key())
|
||||||
|
|
||||||
|
# Look for (tvopen|ethnos).gr embeds
|
||||||
|
tvopengr_urls = list(TVOpenGrEmbedIE._extract_urls(webpage))
|
||||||
|
if tvopengr_urls:
|
||||||
|
return self.playlist_from_matches(tvopengr_urls, video_id, video_title, ie=TVOpenGrEmbedIE.ie_key())
|
||||||
|
|
||||||
tvp_urls = TVPEmbedIE._extract_urls(webpage)
|
tvp_urls = TVPEmbedIE._extract_urls(webpage)
|
||||||
if tvp_urls:
|
if tvp_urls:
|
||||||
return self.playlist_from_matches(tvp_urls, video_id, video_title, ie=TVPEmbedIE.ie_key())
|
return self.playlist_from_matches(tvp_urls, video_id, video_title, ie=TVPEmbedIE.ie_key())
|
||||||
|
|
||||||
|
# Look for MainStreaming embeds
|
||||||
|
mainstreaming_urls = MainStreamingIE._extract_urls(webpage)
|
||||||
|
if mainstreaming_urls:
|
||||||
|
return self.playlist_from_matches(mainstreaming_urls, video_id, video_title, ie=MainStreamingIE.ie_key())
|
||||||
|
|
||||||
|
# Look for Gfycat Embeds
|
||||||
|
gfycat_urls = GfycatIE._extract_urls(webpage)
|
||||||
|
if gfycat_urls:
|
||||||
|
return self.playlist_from_matches(gfycat_urls, video_id, video_title, ie=GfycatIE.ie_key())
|
||||||
|
|
||||||
# Look for HTML5 media
|
# Look for HTML5 media
|
||||||
entries = self._parse_html5_media_entries(url, webpage, video_id, m3u8_id='hls')
|
entries = self._parse_html5_media_entries(url, webpage, video_id, m3u8_id='hls')
|
||||||
if entries:
|
if entries:
|
||||||
@@ -3657,6 +3804,7 @@ class GenericIE(InfoExtractor):
|
|||||||
json_ld['formats'], json_ld['subtitles'] = self._extract_m3u8_formats_and_subtitles(
|
json_ld['formats'], json_ld['subtitles'] = self._extract_m3u8_formats_and_subtitles(
|
||||||
json_ld['url'], video_id, 'mp4')
|
json_ld['url'], video_id, 'mp4')
|
||||||
json_ld.pop('url')
|
json_ld.pop('url')
|
||||||
|
self._sort_formats(json_ld['formats'])
|
||||||
return merge_dicts(json_ld, info_dict)
|
return merge_dicts(json_ld, info_dict)
|
||||||
|
|
||||||
def check_video(vurl):
|
def check_video(vurl):
|
||||||
@@ -3689,7 +3837,7 @@ class GenericIE(InfoExtractor):
|
|||||||
self.report_detected('JW Player embed')
|
self.report_detected('JW Player embed')
|
||||||
if not found:
|
if not found:
|
||||||
# Look for generic KVS player
|
# Look for generic KVS player
|
||||||
found = re.search(r'<script [^>]*?src="https://.+?/kt_player\.js\?v=(?P<ver>(?P<maj_ver>\d+)(\.\d+)+)".*?>', webpage)
|
found = re.search(r'<script [^>]*?src="https?://.+?/kt_player\.js\?v=(?P<ver>(?P<maj_ver>\d+)(\.\d+)+)".*?>', webpage)
|
||||||
if found:
|
if found:
|
||||||
self.report_detected('KWS Player')
|
self.report_detected('KWS Player')
|
||||||
if found.group('maj_ver') not in ['4', '5']:
|
if found.group('maj_ver') not in ['4', '5']:
|
||||||
@@ -3711,20 +3859,21 @@ class GenericIE(InfoExtractor):
|
|||||||
protocol, _, _ = url.partition('/')
|
protocol, _, _ = url.partition('/')
|
||||||
thumbnail = protocol + thumbnail
|
thumbnail = protocol + thumbnail
|
||||||
|
|
||||||
|
url_keys = list(filter(re.compile(r'video_url|video_alt_url\d*').fullmatch, flashvars.keys()))
|
||||||
formats = []
|
formats = []
|
||||||
for key in ('video_url', 'video_alt_url', 'video_alt_url2'):
|
for key in url_keys:
|
||||||
if key in flashvars and '/get_file/' in flashvars[key]:
|
if '/get_file/' not in flashvars[key]:
|
||||||
next_format = {
|
continue
|
||||||
'url': self._kvs_getrealurl(flashvars[key], flashvars['license_code']),
|
format_id = flashvars.get(f'{key}_text', key)
|
||||||
'format_id': flashvars.get(key + '_text', key),
|
formats.append({
|
||||||
'ext': 'mp4',
|
'url': self._kvs_getrealurl(flashvars[key], flashvars['license_code']),
|
||||||
}
|
'format_id': format_id,
|
||||||
height = re.search(r'%s_(\d+)p\.mp4(?:/[?].*)?$' % flashvars['video_id'], flashvars[key])
|
'ext': 'mp4',
|
||||||
if height:
|
**(parse_resolution(format_id) or parse_resolution(flashvars[key]))
|
||||||
next_format['height'] = int(height.group(1))
|
})
|
||||||
else:
|
if not formats[-1].get('height'):
|
||||||
next_format['quality'] = 1
|
formats[-1]['quality'] = 1
|
||||||
formats.append(next_format)
|
|
||||||
self._sort_formats(formats)
|
self._sort_formats(formats)
|
||||||
|
|
||||||
return {
|
return {
|
||||||
|
|||||||
@@ -1,6 +1,8 @@
|
|||||||
# coding: utf-8
|
# coding: utf-8
|
||||||
from __future__ import unicode_literals
|
from __future__ import unicode_literals
|
||||||
|
|
||||||
|
import re
|
||||||
|
|
||||||
from .common import InfoExtractor
|
from .common import InfoExtractor
|
||||||
from ..utils import (
|
from ..utils import (
|
||||||
int_or_none,
|
int_or_none,
|
||||||
@@ -11,7 +13,7 @@ from ..utils import (
|
|||||||
|
|
||||||
|
|
||||||
class GfycatIE(InfoExtractor):
|
class GfycatIE(InfoExtractor):
|
||||||
_VALID_URL = r'https?://(?:(?:www|giant|thumbs)\.)?gfycat\.com/(?:ru/|ifr/|gifs/detail/)?(?P<id>[^-/?#\.]+)'
|
_VALID_URL = r'(?i)https?://(?:(?:www|giant|thumbs)\.)?gfycat\.com/(?:ru/|ifr/|gifs/detail/)?(?P<id>[^-/?#\."\']+)'
|
||||||
_TESTS = [{
|
_TESTS = [{
|
||||||
'url': 'http://gfycat.com/DeadlyDecisiveGermanpinscher',
|
'url': 'http://gfycat.com/DeadlyDecisiveGermanpinscher',
|
||||||
'info_dict': {
|
'info_dict': {
|
||||||
@@ -24,9 +26,10 @@ class GfycatIE(InfoExtractor):
|
|||||||
'duration': 10.4,
|
'duration': 10.4,
|
||||||
'view_count': int,
|
'view_count': int,
|
||||||
'like_count': int,
|
'like_count': int,
|
||||||
'dislike_count': int,
|
|
||||||
'categories': list,
|
'categories': list,
|
||||||
'age_limit': 0,
|
'age_limit': 0,
|
||||||
|
'uploader_id': 'anonymous',
|
||||||
|
'description': '',
|
||||||
}
|
}
|
||||||
}, {
|
}, {
|
||||||
'url': 'http://gfycat.com/ifr/JauntyTimelyAmazontreeboa',
|
'url': 'http://gfycat.com/ifr/JauntyTimelyAmazontreeboa',
|
||||||
@@ -40,9 +43,27 @@ class GfycatIE(InfoExtractor):
|
|||||||
'duration': 3.52,
|
'duration': 3.52,
|
||||||
'view_count': int,
|
'view_count': int,
|
||||||
'like_count': int,
|
'like_count': int,
|
||||||
'dislike_count': int,
|
|
||||||
'categories': list,
|
'categories': list,
|
||||||
'age_limit': 0,
|
'age_limit': 0,
|
||||||
|
'uploader_id': 'anonymous',
|
||||||
|
'description': '',
|
||||||
|
}
|
||||||
|
}, {
|
||||||
|
'url': 'https://gfycat.com/alienatedsolidgreathornedowl',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'alienatedsolidgreathornedowl',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'upload_date': '20211226',
|
||||||
|
'uploader_id': 'reactions',
|
||||||
|
'timestamp': 1640536930,
|
||||||
|
'like_count': int,
|
||||||
|
'description': '',
|
||||||
|
'title': 'Ingrid Michaelson, Zooey Deschanel - Merry Christmas Happy New Year',
|
||||||
|
'categories': list,
|
||||||
|
'age_limit': 0,
|
||||||
|
'duration': 2.9583333333333335,
|
||||||
|
'uploader': 'Reaction GIFs',
|
||||||
|
'view_count': int,
|
||||||
}
|
}
|
||||||
}, {
|
}, {
|
||||||
'url': 'https://gfycat.com/ru/RemarkableDrearyAmurstarfish',
|
'url': 'https://gfycat.com/ru/RemarkableDrearyAmurstarfish',
|
||||||
@@ -59,8 +80,19 @@ class GfycatIE(InfoExtractor):
|
|||||||
}, {
|
}, {
|
||||||
'url': 'https://giant.gfycat.com/acceptablehappygoluckyharborporpoise.mp4',
|
'url': 'https://giant.gfycat.com/acceptablehappygoluckyharborporpoise.mp4',
|
||||||
'only_matching': True
|
'only_matching': True
|
||||||
|
}, {
|
||||||
|
'url': 'http://gfycat.com/IFR/JauntyTimelyAmazontreeboa',
|
||||||
|
'only_matching': True
|
||||||
}]
|
}]
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _extract_urls(webpage):
|
||||||
|
return [
|
||||||
|
mobj.group('url')
|
||||||
|
for mobj in re.finditer(
|
||||||
|
r'<(?:iframe|source)[^>]+\bsrc=["\'](?P<url>%s)' % GfycatIE._VALID_URL,
|
||||||
|
webpage)]
|
||||||
|
|
||||||
def _real_extract(self, url):
|
def _real_extract(self, url):
|
||||||
video_id = self._match_id(url)
|
video_id = self._match_id(url)
|
||||||
|
|
||||||
@@ -74,7 +106,7 @@ class GfycatIE(InfoExtractor):
|
|||||||
title = gfy.get('title') or gfy['gfyName']
|
title = gfy.get('title') or gfy['gfyName']
|
||||||
description = gfy.get('description')
|
description = gfy.get('description')
|
||||||
timestamp = int_or_none(gfy.get('createDate'))
|
timestamp = int_or_none(gfy.get('createDate'))
|
||||||
uploader = gfy.get('userName')
|
uploader = gfy.get('userName') or gfy.get('username')
|
||||||
view_count = int_or_none(gfy.get('views'))
|
view_count = int_or_none(gfy.get('views'))
|
||||||
like_count = int_or_none(gfy.get('likes'))
|
like_count = int_or_none(gfy.get('likes'))
|
||||||
dislike_count = int_or_none(gfy.get('dislikes'))
|
dislike_count = int_or_none(gfy.get('dislikes'))
|
||||||
@@ -114,7 +146,8 @@ class GfycatIE(InfoExtractor):
|
|||||||
'title': title,
|
'title': title,
|
||||||
'description': description,
|
'description': description,
|
||||||
'timestamp': timestamp,
|
'timestamp': timestamp,
|
||||||
'uploader': uploader,
|
'uploader': gfy.get('userDisplayName') or uploader,
|
||||||
|
'uploader_id': uploader,
|
||||||
'duration': duration,
|
'duration': duration,
|
||||||
'view_count': view_count,
|
'view_count': view_count,
|
||||||
'like_count': like_count,
|
'like_count': like_count,
|
||||||
|
|||||||
@@ -0,0 +1,225 @@
|
|||||||
|
# coding: utf-8
|
||||||
|
from __future__ import unicode_literals
|
||||||
|
|
||||||
|
import re
|
||||||
|
import urllib.parse
|
||||||
|
|
||||||
|
from .common import InfoExtractor
|
||||||
|
from ..utils import (
|
||||||
|
determine_ext,
|
||||||
|
ExtractorError,
|
||||||
|
int_or_none,
|
||||||
|
parse_qs,
|
||||||
|
smuggle_url,
|
||||||
|
unescapeHTML,
|
||||||
|
unsmuggle_url,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class GlomexBaseIE(InfoExtractor):
|
||||||
|
_DEFAULT_ORIGIN_URL = 'https://player.glomex.com/'
|
||||||
|
_API_URL = 'https://integration-cloudfront-eu-west-1.mes.glomex.cloud/'
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _smuggle_origin_url(url, origin_url):
|
||||||
|
if origin_url is None:
|
||||||
|
return url
|
||||||
|
return smuggle_url(url, {'origin': origin_url})
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def _unsmuggle_origin_url(cls, url, fallback_origin_url=None):
|
||||||
|
defaults = {'origin': fallback_origin_url or cls._DEFAULT_ORIGIN_URL}
|
||||||
|
unsmuggled_url, data = unsmuggle_url(url, default=defaults)
|
||||||
|
return unsmuggled_url, data['origin']
|
||||||
|
|
||||||
|
def _get_videoid_type(self, video_id):
|
||||||
|
_VIDEOID_TYPES = {
|
||||||
|
'v': 'video',
|
||||||
|
'pl': 'playlist',
|
||||||
|
'rl': 'related videos playlist',
|
||||||
|
'cl': 'curated playlist',
|
||||||
|
}
|
||||||
|
prefix = video_id.split('-')[0]
|
||||||
|
return _VIDEOID_TYPES.get(prefix, 'unknown type')
|
||||||
|
|
||||||
|
def _download_api_data(self, video_id, integration, current_url=None):
|
||||||
|
query = {
|
||||||
|
'integration_id': integration,
|
||||||
|
'playlist_id': video_id,
|
||||||
|
'current_url': current_url or self._DEFAULT_ORIGIN_URL,
|
||||||
|
}
|
||||||
|
video_id_type = self._get_videoid_type(video_id)
|
||||||
|
return self._download_json(
|
||||||
|
self._API_URL,
|
||||||
|
video_id, 'Downloading %s JSON' % video_id_type,
|
||||||
|
'Unable to download %s JSON' % video_id_type,
|
||||||
|
query=query)
|
||||||
|
|
||||||
|
def _download_and_extract_api_data(self, video_id, integration, current_url):
|
||||||
|
api_data = self._download_api_data(video_id, integration, current_url)
|
||||||
|
videos = api_data['videos']
|
||||||
|
if not videos:
|
||||||
|
raise ExtractorError('no videos found for %s' % video_id)
|
||||||
|
videos = [self._extract_api_data(video, video_id) for video in videos]
|
||||||
|
return videos[0] if len(videos) == 1 else self.playlist_result(videos, video_id)
|
||||||
|
|
||||||
|
def _extract_api_data(self, video, video_id):
|
||||||
|
if video.get('error_code') == 'contentGeoblocked':
|
||||||
|
self.raise_geo_restricted(countries=video['geo_locations'])
|
||||||
|
|
||||||
|
formats, subs = [], {}
|
||||||
|
for format_id, format_url in video['source'].items():
|
||||||
|
ext = determine_ext(format_url)
|
||||||
|
if ext == 'm3u8':
|
||||||
|
formats_, subs_ = self._extract_m3u8_formats_and_subtitles(
|
||||||
|
format_url, video_id, 'mp4', m3u8_id=format_id,
|
||||||
|
fatal=False)
|
||||||
|
formats.extend(formats_)
|
||||||
|
self._merge_subtitles(subs_, target=subs)
|
||||||
|
else:
|
||||||
|
formats.append({
|
||||||
|
'url': format_url,
|
||||||
|
'format_id': format_id,
|
||||||
|
})
|
||||||
|
if video.get('language'):
|
||||||
|
for fmt in formats:
|
||||||
|
fmt['language'] = video['language']
|
||||||
|
self._sort_formats(formats)
|
||||||
|
|
||||||
|
images = (video.get('images') or []) + [video.get('image') or {}]
|
||||||
|
thumbnails = [{
|
||||||
|
'id': image.get('id'),
|
||||||
|
'url': f'{image["url"]}/profile:player-960x540',
|
||||||
|
'width': 960,
|
||||||
|
'height': 540,
|
||||||
|
} for image in images if image.get('url')]
|
||||||
|
self._remove_duplicate_formats(thumbnails)
|
||||||
|
|
||||||
|
return {
|
||||||
|
'id': video.get('clip_id') or video_id,
|
||||||
|
'title': video.get('title'),
|
||||||
|
'description': video.get('description'),
|
||||||
|
'thumbnails': thumbnails,
|
||||||
|
'duration': int_or_none(video.get('clip_duration')),
|
||||||
|
'timestamp': video.get('created_at'),
|
||||||
|
'formats': formats,
|
||||||
|
'subtitles': subs,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class GlomexIE(GlomexBaseIE):
|
||||||
|
IE_NAME = 'glomex'
|
||||||
|
IE_DESC = 'Glomex videos'
|
||||||
|
_VALID_URL = r'https?://video\.glomex\.com/[^/]+/(?P<id>v-[^-]+)'
|
||||||
|
_INTEGRATION_ID = '19syy24xjn1oqlpc'
|
||||||
|
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://video.glomex.com/sport/v-cb24uwg77hgh-nach-2-0-sieg-guardiola-mit-mancity-vor-naechstem-titel',
|
||||||
|
'md5': 'cec33a943c4240c9cb33abea8c26242e',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'v-cb24uwg77hgh',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'md5:38a90cedcfadd72982c81acf13556e0c',
|
||||||
|
'description': 'md5:1ea6b6caff1443fcbbba159e432eedb8',
|
||||||
|
'duration': 29600,
|
||||||
|
'timestamp': 1619895017,
|
||||||
|
'upload_date': '20210501',
|
||||||
|
},
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
video_id = self._match_id(url)
|
||||||
|
return self.url_result(
|
||||||
|
GlomexEmbedIE.build_player_url(video_id, self._INTEGRATION_ID, url),
|
||||||
|
GlomexEmbedIE.ie_key(), video_id)
|
||||||
|
|
||||||
|
|
||||||
|
class GlomexEmbedIE(GlomexBaseIE):
|
||||||
|
IE_NAME = 'glomex:embed'
|
||||||
|
IE_DESC = 'Glomex embedded videos'
|
||||||
|
_BASE_PLAYER_URL = '//player.glomex.com/integration/1/iframe-player.html'
|
||||||
|
_BASE_PLAYER_URL_RE = re.escape(_BASE_PLAYER_URL).replace('/1/', r'/[^/]/')
|
||||||
|
_VALID_URL = rf'https?:{_BASE_PLAYER_URL_RE}\?([^#]+&)?playlistId=(?P<id>[^#&]+)'
|
||||||
|
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://player.glomex.com/integration/1/iframe-player.html?integrationId=4059a013k56vb2yd&playlistId=v-cfa6lye0dkdd-sf',
|
||||||
|
'md5': '68f259b98cc01918ac34180142fce287',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'v-cfa6lye0dkdd-sf',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'timestamp': 1635337199,
|
||||||
|
'duration': 133080,
|
||||||
|
'upload_date': '20211027',
|
||||||
|
'description': 'md5:e741185fc309310ff5d0c789b437be66',
|
||||||
|
'title': 'md5:35647293513a6c92363817a0fb0a7961',
|
||||||
|
},
|
||||||
|
}, {
|
||||||
|
'url': 'https://player.glomex.com/integration/1/iframe-player.html?origin=fullpage&integrationId=19syy24xjn1oqlpc&playlistId=rl-vcb49w1fb592p&playlistIndex=0',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'rl-vcb49w1fb592p',
|
||||||
|
},
|
||||||
|
'playlist_count': 100,
|
||||||
|
}, {
|
||||||
|
'url': 'https://player.glomex.com/integration/1/iframe-player.html?playlistId=cl-bgqaata6aw8x&integrationId=19syy24xjn1oqlpc',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'cl-bgqaata6aw8x',
|
||||||
|
},
|
||||||
|
'playlist_mincount': 2,
|
||||||
|
}]
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def build_player_url(cls, video_id, integration, origin_url=None):
|
||||||
|
query_string = urllib.parse.urlencode({
|
||||||
|
'playlistId': video_id,
|
||||||
|
'integrationId': integration,
|
||||||
|
})
|
||||||
|
return cls._smuggle_origin_url(f'https:{cls._BASE_PLAYER_URL}?{query_string}', origin_url)
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def _extract_urls(cls, webpage, origin_url):
|
||||||
|
VALID_SRC = rf'(?:https?:)?{cls._BASE_PLAYER_URL_RE}\?(?:(?!(?P=_q1)).)+'
|
||||||
|
|
||||||
|
# https://docs.glomex.com/publisher/video-player-integration/javascript-api/
|
||||||
|
EMBED_RE = r'''(?x)(?:
|
||||||
|
<iframe[^>]+?src=(?P<_q1>%(quot_re)s)(?P<url>%(url_re)s)(?P=_q1)|
|
||||||
|
<(?P<html_tag>glomex-player|div)(?:
|
||||||
|
data-integration-id=(?P<_q2>%(quot_re)s)(?P<integration_html>(?:(?!(?P=_q2)).)+)(?P=_q2)|
|
||||||
|
data-playlist-id=(?P<_q3>%(quot_re)s)(?P<id_html>(?:(?!(?P=_q3)).)+)(?P=_q3)|
|
||||||
|
data-glomex-player=(?P<_q4>%(quot_re)s)(?P<glomex_player>true)(?P=_q4)|
|
||||||
|
[^>]*?
|
||||||
|
)+>|
|
||||||
|
# naive parsing of inline scripts for hard-coded integration parameters
|
||||||
|
<(?P<script_tag>script)[^<]*?>(?:
|
||||||
|
(?P<_stjs1>dataset\.)?integrationId\s*(?(_stjs1)=|:)\s*
|
||||||
|
(?P<_q5>%(quot_re)s)(?P<integration_js>(?:(?!(?P=_q5)).)+)(?P=_q5)\s*(?(_stjs1);|,)?|
|
||||||
|
(?P<_stjs2>dataset\.)?playlistId\s*(?(_stjs2)=|:)\s*
|
||||||
|
(?P<_q6>%(quot_re)s)(?P<id_js>(?:(?!(?P=_q6)).)+)(?P=_q6)\s*(?(_stjs2);|,)?|
|
||||||
|
(?:\s|.)*?
|
||||||
|
)+</script>
|
||||||
|
)''' % {'quot_re': r'["\']', 'url_re': VALID_SRC}
|
||||||
|
|
||||||
|
for mobj in re.finditer(EMBED_RE, webpage):
|
||||||
|
mdict = mobj.groupdict()
|
||||||
|
if mdict.get('url'):
|
||||||
|
url = unescapeHTML(mdict['url'])
|
||||||
|
if not cls.suitable(url):
|
||||||
|
continue
|
||||||
|
yield cls._smuggle_origin_url(url, origin_url)
|
||||||
|
elif mdict.get('html_tag'):
|
||||||
|
if mdict['html_tag'] == 'div' and not mdict.get('glomex_player'):
|
||||||
|
continue
|
||||||
|
if not mdict.get('video_id_html') or not mdict.get('integration_html'):
|
||||||
|
continue
|
||||||
|
yield cls.build_player_url(mdict['video_id_html'], mdict['integration_html'], origin_url)
|
||||||
|
elif mdict.get('script_tag'):
|
||||||
|
if not mdict.get('video_id_js') or not mdict.get('integration_js'):
|
||||||
|
continue
|
||||||
|
yield cls.build_player_url(mdict['video_id_js'], mdict['integration_js'], origin_url)
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
url, origin_url = self._unsmuggle_origin_url(url)
|
||||||
|
playlist_id = self._match_id(url)
|
||||||
|
integration = parse_qs(url).get('integrationId', [None])[0]
|
||||||
|
if not integration:
|
||||||
|
raise ExtractorError('No integrationId in URL', expected=True)
|
||||||
|
return self._download_and_extract_api_data(playlist_id, integration, origin_url)
|
||||||
@@ -203,6 +203,9 @@ class HotStarIE(HotStarBaseIE):
|
|||||||
format_url = re.sub(
|
format_url = re.sub(
|
||||||
r'(?<=//staragvod)(\d)', r'web\1', format_url)
|
r'(?<=//staragvod)(\d)', r'web\1', format_url)
|
||||||
tags = str_or_none(playback_set.get('tagsCombination')) or ''
|
tags = str_or_none(playback_set.get('tagsCombination')) or ''
|
||||||
|
ingored_res, ignored_vcodec, ignored_dr = self._configuration_arg('res'), self._configuration_arg('vcodec'), self._configuration_arg('dr')
|
||||||
|
if any(f'resolution:{ig_res}' in tags for ig_res in ingored_res) or any(f'video_codec:{ig_vc}' in tags for ig_vc in ignored_vcodec) or any(f'dynamic_range:{ig_dr}' in tags for ig_dr in ignored_dr):
|
||||||
|
continue
|
||||||
ext = determine_ext(format_url)
|
ext = determine_ext(format_url)
|
||||||
current_formats, current_subs = [], {}
|
current_formats, current_subs = [], {}
|
||||||
try:
|
try:
|
||||||
|
|||||||
@@ -26,13 +26,7 @@ class HRFernsehenIE(InfoExtractor):
|
|||||||
}]},
|
}]},
|
||||||
'timestamp': 1598470200,
|
'timestamp': 1598470200,
|
||||||
'upload_date': '20200826',
|
'upload_date': '20200826',
|
||||||
'thumbnails': [{
|
'thumbnail': 'https://www.hessenschau.de/tv-sendung/hs_ganz-1554~_t-1598465545029_v-16to9__medium.jpg',
|
||||||
'url': 'https://www.hessenschau.de/tv-sendung/hs_ganz-1554~_t-1598465545029_v-16to9.jpg',
|
|
||||||
'id': '0'
|
|
||||||
}, {
|
|
||||||
'url': 'https://www.hessenschau.de/tv-sendung/hs_ganz-1554~_t-1598465545029_v-16to9__medium.jpg',
|
|
||||||
'id': '1'
|
|
||||||
}],
|
|
||||||
'title': 'hessenschau vom 26.08.2020'
|
'title': 'hessenschau vom 26.08.2020'
|
||||||
}
|
}
|
||||||
}, {
|
}, {
|
||||||
@@ -81,7 +75,7 @@ class HRFernsehenIE(InfoExtractor):
|
|||||||
description = self._html_search_meta(
|
description = self._html_search_meta(
|
||||||
['description'], webpage)
|
['description'], webpage)
|
||||||
|
|
||||||
loader_str = unescapeHTML(self._search_regex(r"data-hr-mediaplayer-loader='([^']*)'", webpage, "ardloader"))
|
loader_str = unescapeHTML(self._search_regex(r"data-new-hr-mediaplayer-loader='([^']*)'", webpage, "ardloader"))
|
||||||
loader_data = json.loads(loader_str)
|
loader_data = json.loads(loader_str)
|
||||||
|
|
||||||
info = {
|
info = {
|
||||||
|
|||||||
@@ -12,6 +12,7 @@ from ..compat import (
|
|||||||
)
|
)
|
||||||
from ..utils import (
|
from ..utils import (
|
||||||
ExtractorError,
|
ExtractorError,
|
||||||
|
format_field,
|
||||||
float_or_none,
|
float_or_none,
|
||||||
get_element_by_attribute,
|
get_element_by_attribute,
|
||||||
int_or_none,
|
int_or_none,
|
||||||
@@ -341,7 +342,7 @@ class InstagramIE(InstagramBaseIE):
|
|||||||
if nodes:
|
if nodes:
|
||||||
return self.playlist_result(
|
return self.playlist_result(
|
||||||
self._extract_nodes(nodes, True), video_id,
|
self._extract_nodes(nodes, True), video_id,
|
||||||
'Post by %s' % uploader_id if uploader_id else None, description)
|
format_field(uploader_id, template='Post by %s'), description)
|
||||||
|
|
||||||
video_url = self._og_search_video_url(webpage, secure=False)
|
video_url = self._og_search_video_url(webpage, secure=False)
|
||||||
|
|
||||||
@@ -542,3 +543,90 @@ class InstagramTagIE(InstagramPlaylistBaseIE):
|
|||||||
'tag_name':
|
'tag_name':
|
||||||
data['entry_data']['TagPage'][0]['graphql']['hashtag']['name']
|
data['entry_data']['TagPage'][0]['graphql']['hashtag']['name']
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class InstagramStoryIE(InstagramBaseIE):
|
||||||
|
_VALID_URL = r'https?://(?:www\.)?instagram\.com/stories/(?P<user>[^/]+)/(?P<id>\d+)'
|
||||||
|
IE_NAME = 'instagram:story'
|
||||||
|
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://www.instagram.com/stories/highlights/18090946048123978/',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '18090946048123978',
|
||||||
|
'title': 'Rare',
|
||||||
|
},
|
||||||
|
'playlist_mincount': 50
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
username, story_id = self._match_valid_url(url).groups()
|
||||||
|
|
||||||
|
story_info_url = f'{username}/{story_id}/?__a=1' if username == 'highlights' else f'{username}/?__a=1'
|
||||||
|
story_info = self._download_json(f'https://www.instagram.com/stories/{story_info_url}', story_id, headers={
|
||||||
|
'X-IG-App-ID': 936619743392459,
|
||||||
|
'X-ASBD-ID': 198387,
|
||||||
|
'X-IG-WWW-Claim': 0,
|
||||||
|
'X-Requested-With': 'XMLHttpRequest',
|
||||||
|
'Referer': url,
|
||||||
|
})
|
||||||
|
user_id = story_info['user']['id']
|
||||||
|
highlight_title = traverse_obj(story_info, ('highlight', 'title'))
|
||||||
|
|
||||||
|
story_info_url = user_id if username != 'highlights' else f'highlight:{story_id}'
|
||||||
|
videos = self._download_json(f'https://i.instagram.com/api/v1/feed/reels_media/?reel_ids={story_info_url}', story_id, headers={
|
||||||
|
'X-IG-App-ID': 936619743392459,
|
||||||
|
'X-ASBD-ID': 198387,
|
||||||
|
'X-IG-WWW-Claim': 0,
|
||||||
|
})['reels']
|
||||||
|
entites = []
|
||||||
|
|
||||||
|
full_name = traverse_obj(videos, ('user', 'full_name'))
|
||||||
|
|
||||||
|
user_info = {}
|
||||||
|
if not (username and username != 'highlights' and full_name):
|
||||||
|
user_info = self._download_json(
|
||||||
|
f'https://i.instagram.com/api/v1/users/{user_id}/info/', story_id, headers={
|
||||||
|
'User-Agent': 'Mozilla/5.0 (Linux; Android 11; SM-A505F Build/RP1A.200720.012; wv) AppleWebKit/537.36 (KHTML, like Gecko) Version/4.0 Chrome/96.0.4664.45 Mobile Safari/537.36 Instagram 214.1.0.29.120 Android (30/11; 450dpi; 1080x2122; samsung; SM-A505F; a50; exynos9610; en_US; 333717274)',
|
||||||
|
}, note='Downloading user info')
|
||||||
|
|
||||||
|
username = traverse_obj(user_info, ('user', 'username')) or username
|
||||||
|
full_name = traverse_obj(user_info, ('user', 'full_name')) or full_name
|
||||||
|
|
||||||
|
videos = traverse_obj(videos, (f'highlight:{story_id}', 'items'), (str(user_id), 'items'))
|
||||||
|
for video_info in videos:
|
||||||
|
formats = []
|
||||||
|
if isinstance(video_info, list):
|
||||||
|
video_info = video_info[0]
|
||||||
|
vcodec = video_info.get('video_codec')
|
||||||
|
dash_manifest_raw = video_info.get('video_dash_manifest')
|
||||||
|
videos_list = video_info.get('video_versions')
|
||||||
|
if not (dash_manifest_raw or videos_list):
|
||||||
|
continue
|
||||||
|
for format in videos_list:
|
||||||
|
formats.append({
|
||||||
|
'url': format.get('url'),
|
||||||
|
'width': format.get('width'),
|
||||||
|
'height': format.get('height'),
|
||||||
|
'vcodec': vcodec,
|
||||||
|
})
|
||||||
|
if dash_manifest_raw:
|
||||||
|
formats.extend(self._parse_mpd_formats(self._parse_xml(dash_manifest_raw, story_id), mpd_id='dash'))
|
||||||
|
self._sort_formats(formats)
|
||||||
|
thumbnails = [{
|
||||||
|
'url': thumbnail.get('url'),
|
||||||
|
'width': thumbnail.get('width'),
|
||||||
|
'height': thumbnail.get('height')
|
||||||
|
} for thumbnail in traverse_obj(video_info, ('image_versions2', 'candidates')) or []]
|
||||||
|
entites.append({
|
||||||
|
'id': video_info.get('id'),
|
||||||
|
'title': f'Story by {username}',
|
||||||
|
'timestamp': int_or_none(video_info.get('taken_at')),
|
||||||
|
'channel': username,
|
||||||
|
'uploader': full_name,
|
||||||
|
'duration': float_or_none(video_info.get('video_duration')),
|
||||||
|
'uploader_id': user_id,
|
||||||
|
'thumbnails': thumbnails,
|
||||||
|
'formats': formats,
|
||||||
|
})
|
||||||
|
|
||||||
|
return self.playlist_result(entites, playlist_id=story_id, playlist_title=highlight_title)
|
||||||
|
|||||||
+342
-1
@@ -11,14 +11,26 @@ from ..compat import (
|
|||||||
compat_str,
|
compat_str,
|
||||||
compat_urllib_parse_urlencode,
|
compat_urllib_parse_urlencode,
|
||||||
)
|
)
|
||||||
|
from .openload import PhantomJSwrapper
|
||||||
from ..utils import (
|
from ..utils import (
|
||||||
clean_html,
|
clean_html,
|
||||||
decode_packed_codes,
|
decode_packed_codes,
|
||||||
|
ExtractorError,
|
||||||
|
float_or_none,
|
||||||
get_element_by_id,
|
get_element_by_id,
|
||||||
get_element_by_attribute,
|
get_element_by_attribute,
|
||||||
ExtractorError,
|
int_or_none,
|
||||||
|
js_to_json,
|
||||||
ohdave_rsa_encrypt,
|
ohdave_rsa_encrypt,
|
||||||
|
parse_age_limit,
|
||||||
|
parse_duration,
|
||||||
|
parse_iso8601,
|
||||||
|
parse_resolution,
|
||||||
|
qualities,
|
||||||
remove_start,
|
remove_start,
|
||||||
|
str_or_none,
|
||||||
|
traverse_obj,
|
||||||
|
urljoin,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -392,3 +404,332 @@ class IqiyiIE(InfoExtractor):
|
|||||||
'title': title,
|
'title': title,
|
||||||
'formats': formats,
|
'formats': formats,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class IqIE(InfoExtractor):
|
||||||
|
IE_NAME = 'iq.com'
|
||||||
|
IE_DESC = 'International version of iQiyi'
|
||||||
|
_VALID_URL = r'https?://(?:www\.)?iq\.com/play/(?:[\w%-]*-)?(?P<id>\w+)'
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://www.iq.com/play/one-piece-episode-1000-1ma1i6ferf4',
|
||||||
|
'md5': '2d7caf6eeca8a32b407094b33b757d39',
|
||||||
|
'info_dict': {
|
||||||
|
'ext': 'mp4',
|
||||||
|
'id': '1ma1i6ferf4',
|
||||||
|
'title': '航海王 第1000集',
|
||||||
|
'description': 'Subtitle available on Sunday 4PM(GMT+8).',
|
||||||
|
'duration': 1430,
|
||||||
|
'timestamp': 1637488203,
|
||||||
|
'upload_date': '20211121',
|
||||||
|
'episode_number': 1000,
|
||||||
|
'episode': 'Episode 1000',
|
||||||
|
'series': 'One Piece',
|
||||||
|
'age_limit': 13,
|
||||||
|
'average_rating': float,
|
||||||
|
},
|
||||||
|
'params': {
|
||||||
|
'format': '500',
|
||||||
|
},
|
||||||
|
'expected_warnings': ['format is restricted']
|
||||||
|
}]
|
||||||
|
_BID_TAGS = {
|
||||||
|
'100': '240P',
|
||||||
|
'200': '360P',
|
||||||
|
'300': '480P',
|
||||||
|
'500': '720P',
|
||||||
|
'600': '1080P',
|
||||||
|
'610': '1080P50',
|
||||||
|
'700': '2K',
|
||||||
|
'800': '4K',
|
||||||
|
}
|
||||||
|
_LID_TAGS = {
|
||||||
|
'1': 'zh_CN',
|
||||||
|
'2': 'zh_TW',
|
||||||
|
'3': 'en',
|
||||||
|
'18': 'th',
|
||||||
|
'21': 'my',
|
||||||
|
'23': 'vi',
|
||||||
|
'24': 'id',
|
||||||
|
'26': 'es',
|
||||||
|
'28': 'ar',
|
||||||
|
}
|
||||||
|
|
||||||
|
_DASH_JS = '''
|
||||||
|
console.log(page.evaluate(function() {
|
||||||
|
var tvid = "%(tvid)s"; var vid = "%(vid)s"; var src = "%(src)s";
|
||||||
|
var dfp = "%(dfp)s"; var mode = "%(mode)s"; var lang = "%(lang)s"; var bid_list = %(bid_list)s;
|
||||||
|
var tm = new Date().getTime();
|
||||||
|
var cmd5x_func = %(cmd5x_func)s; var cmd5x_exporter = {}; cmd5x_func({}, cmd5x_exporter, {}); var cmd5x = cmd5x_exporter.cmd5x;
|
||||||
|
var authKey = cmd5x(cmd5x('') + tm + '' + tvid);
|
||||||
|
var k_uid = Array.apply(null, Array(32)).map(function() {return Math.floor(Math.random() * 15).toString(16)}).join('');
|
||||||
|
var dash_paths = {};
|
||||||
|
bid_list.forEach(function(bid) {
|
||||||
|
var query = {
|
||||||
|
'tvid': tvid,
|
||||||
|
'bid': bid,
|
||||||
|
'ds': 1,
|
||||||
|
'vid': vid,
|
||||||
|
'src': src,
|
||||||
|
'vt': 0,
|
||||||
|
'rs': 1,
|
||||||
|
'uid': 0,
|
||||||
|
'ori': 'pcw',
|
||||||
|
'ps': 1,
|
||||||
|
'k_uid': k_uid,
|
||||||
|
'pt': 0,
|
||||||
|
'd': 0,
|
||||||
|
's': '',
|
||||||
|
'lid': '',
|
||||||
|
'slid': 0,
|
||||||
|
'cf': '',
|
||||||
|
'ct': '',
|
||||||
|
'authKey': authKey,
|
||||||
|
'k_tag': 1,
|
||||||
|
'ost': 0,
|
||||||
|
'ppt': 0,
|
||||||
|
'dfp': dfp,
|
||||||
|
'prio': JSON.stringify({
|
||||||
|
'ff': 'f4v',
|
||||||
|
'code': 2
|
||||||
|
}),
|
||||||
|
'k_err_retries': 0,
|
||||||
|
'up': '',
|
||||||
|
'su': 2,
|
||||||
|
'applang': lang,
|
||||||
|
'sver': 2,
|
||||||
|
'X-USER-MODE': mode,
|
||||||
|
'qd_v': 2,
|
||||||
|
'tm': tm,
|
||||||
|
'qdy': 'a',
|
||||||
|
'qds': 0,
|
||||||
|
'k_ft1': 141287244169348,
|
||||||
|
'k_ft4': 34359746564,
|
||||||
|
'k_ft5': 1,
|
||||||
|
'bop': JSON.stringify({
|
||||||
|
'version': '10.0',
|
||||||
|
'dfp': dfp
|
||||||
|
}),
|
||||||
|
'ut': 0, // TODO: Set ut param for VIP members
|
||||||
|
};
|
||||||
|
var enc_params = [];
|
||||||
|
for (var prop in query) {
|
||||||
|
enc_params.push(encodeURIComponent(prop) + '=' + encodeURIComponent(query[prop]));
|
||||||
|
}
|
||||||
|
var dash_path = '/dash?' + enc_params.join('&'); dash_path += '&vf=' + cmd5x(dash_path);
|
||||||
|
dash_paths[bid] = dash_path;
|
||||||
|
});
|
||||||
|
return JSON.stringify(dash_paths);
|
||||||
|
}));
|
||||||
|
saveAndExit();
|
||||||
|
'''
|
||||||
|
|
||||||
|
def _extract_vms_player_js(self, webpage, video_id):
|
||||||
|
player_js_cache = self._downloader.cache.load('iq', 'player_js')
|
||||||
|
if player_js_cache:
|
||||||
|
return player_js_cache
|
||||||
|
webpack_js_url = self._proto_relative_url(self._search_regex(
|
||||||
|
r'<script src="((?:https?)?//stc.iqiyipic.com/_next/static/chunks/webpack-\w+\.js)"', webpage, 'webpack URL'))
|
||||||
|
webpack_js = self._download_webpage(webpack_js_url, video_id, note='Downloading webpack JS', errnote='Unable to download webpack JS')
|
||||||
|
webpack_map1, webpack_map2 = [self._parse_json(js_map, video_id, transform_source=js_to_json) for js_map in self._search_regex(
|
||||||
|
r'\(({[^}]*})\[\w+\][^\)]*\)\s*\+\s*["\']\.["\']\s*\+\s*({[^}]*})\[\w+\]\+["\']\.js', webpack_js, 'JS locations', group=(1, 2))]
|
||||||
|
for module_index in reversed(list(webpack_map2.keys())):
|
||||||
|
module_js = self._download_webpage(
|
||||||
|
f'https://stc.iqiyipic.com/_next/static/chunks/{webpack_map1.get(module_index, module_index)}.{webpack_map2[module_index]}.js',
|
||||||
|
video_id, note=f'Downloading #{module_index} module JS', errnote='Unable to download module JS', fatal=False) or ''
|
||||||
|
if 'vms request' in module_js:
|
||||||
|
self._downloader.cache.store('iq', 'player_js', module_js)
|
||||||
|
return module_js
|
||||||
|
raise ExtractorError('Unable to extract player JS')
|
||||||
|
|
||||||
|
def _extract_cmd5x_function(self, webpage, video_id):
|
||||||
|
return self._search_regex(r',\s*(function\s*\([^\)]*\)\s*{\s*var _qda.+_qdc\(\)\s*})\s*,',
|
||||||
|
self._extract_vms_player_js(webpage, video_id), 'signature function')
|
||||||
|
|
||||||
|
def _update_bid_tags(self, webpage, video_id):
|
||||||
|
extracted_bid_tags = self._parse_json(
|
||||||
|
self._search_regex(
|
||||||
|
r'arguments\[1\][^,]*,\s*function\s*\([^\)]*\)\s*{\s*"use strict";?\s*var \w=({.+}})\s*,\s*\w\s*=\s*{\s*getNewVd',
|
||||||
|
self._extract_vms_player_js(webpage, video_id), 'video tags', default=''),
|
||||||
|
video_id, transform_source=js_to_json, fatal=False)
|
||||||
|
if not extracted_bid_tags:
|
||||||
|
return
|
||||||
|
self._BID_TAGS = {
|
||||||
|
bid: traverse_obj(extracted_bid_tags, (bid, 'value'), expected_type=str, default=self._BID_TAGS.get(bid))
|
||||||
|
for bid in extracted_bid_tags.keys()
|
||||||
|
}
|
||||||
|
|
||||||
|
def _get_cookie(self, name, default=None):
|
||||||
|
cookie = self._get_cookies('https://iq.com/').get(name)
|
||||||
|
return cookie.value if cookie else default
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
video_id = self._match_id(url)
|
||||||
|
webpage = self._download_webpage(url, video_id)
|
||||||
|
self._update_bid_tags(webpage, video_id)
|
||||||
|
|
||||||
|
next_props = self._search_nextjs_data(webpage, video_id)['props']
|
||||||
|
page_data = next_props['initialState']['play']
|
||||||
|
video_info = page_data['curVideoInfo']
|
||||||
|
|
||||||
|
# bid 0 as an initial format checker
|
||||||
|
dash_paths = self._parse_json(PhantomJSwrapper(self).get(
|
||||||
|
url, html='<!DOCTYPE html>', video_id=video_id, note2='Executing signature code', jscode=self._DASH_JS % {
|
||||||
|
'tvid': video_info['tvId'],
|
||||||
|
'vid': video_info['vid'],
|
||||||
|
'src': traverse_obj(next_props, ('initialProps', 'pageProps', 'ptid'),
|
||||||
|
expected_type=str, default='01010031010018000000'),
|
||||||
|
'dfp': self._get_cookie('dfp', ''),
|
||||||
|
'mode': self._get_cookie('mod', 'intl'),
|
||||||
|
'lang': self._get_cookie('lang', 'en_us'),
|
||||||
|
'bid_list': '[' + ','.join(['0', *self._BID_TAGS.keys()]) + ']',
|
||||||
|
'cmd5x_func': self._extract_cmd5x_function(webpage, video_id),
|
||||||
|
})[1].strip(), video_id)
|
||||||
|
|
||||||
|
formats, subtitles = [], {}
|
||||||
|
initial_format_data = self._download_json(
|
||||||
|
urljoin('https://cache-video.iq.com', dash_paths['0']), video_id,
|
||||||
|
note='Downloading initial video format info', errnote='Unable to download initial video format info')['data']
|
||||||
|
|
||||||
|
preview_time = traverse_obj(initial_format_data, ('boss_ts', 'data', 'previewTime'), expected_type=float_or_none)
|
||||||
|
if preview_time:
|
||||||
|
self.report_warning(f'This preview video is limited to {preview_time} seconds')
|
||||||
|
|
||||||
|
# TODO: Extract audio-only formats
|
||||||
|
for bid in set(traverse_obj(initial_format_data, ('program', 'video', ..., 'bid'), expected_type=str_or_none, default=[])):
|
||||||
|
dash_path = dash_paths.get(bid)
|
||||||
|
if not dash_path:
|
||||||
|
self.report_warning(f'Unknown format id: {bid}. It is currently not being extracted')
|
||||||
|
continue
|
||||||
|
format_data = traverse_obj(self._download_json(
|
||||||
|
urljoin('https://cache-video.iq.com', dash_path), video_id,
|
||||||
|
note=f'Downloading format data for {self._BID_TAGS[bid]}', errnote='Unable to download format data',
|
||||||
|
fatal=False), 'data', expected_type=dict)
|
||||||
|
|
||||||
|
video_format = next((video_format for video_format in traverse_obj(
|
||||||
|
format_data, ('program', 'video', ...), expected_type=dict, default=[]) if str(video_format['bid']) == bid), {})
|
||||||
|
extracted_formats = []
|
||||||
|
if video_format.get('m3u8Url'):
|
||||||
|
extracted_formats.extend(self._extract_m3u8_formats(
|
||||||
|
urljoin(format_data.get('dm3u8', 'https://cache-m.iq.com/dc/dt/'), video_format['m3u8Url']),
|
||||||
|
'mp4', m3u8_id=bid, fatal=False))
|
||||||
|
if video_format.get('mpdUrl'):
|
||||||
|
# TODO: Properly extract mpd hostname
|
||||||
|
extracted_formats.extend(self._extract_mpd_formats(
|
||||||
|
urljoin(format_data.get('dm3u8', 'https://cache-m.iq.com/dc/dt/'), video_format['mpdUrl']),
|
||||||
|
mpd_id=bid, fatal=False))
|
||||||
|
if video_format.get('m3u8'):
|
||||||
|
ff = video_format.get('ff', 'ts')
|
||||||
|
if ff == 'ts':
|
||||||
|
m3u8_formats, _ = self._parse_m3u8_formats_and_subtitles(
|
||||||
|
video_format['m3u8'], ext='mp4', m3u8_id=bid, fatal=False)
|
||||||
|
extracted_formats.extend(m3u8_formats)
|
||||||
|
elif ff == 'm4s':
|
||||||
|
mpd_data = traverse_obj(
|
||||||
|
self._parse_json(video_format['m3u8'], video_id, fatal=False), ('payload', ..., 'data'), expected_type=str)
|
||||||
|
if not mpd_data:
|
||||||
|
continue
|
||||||
|
mpd_formats, _ = self._parse_mpd_formats_and_subtitles(
|
||||||
|
mpd_data, bid, format_data.get('dm3u8', 'https://cache-m.iq.com/dc/dt/'))
|
||||||
|
extracted_formats.extend(mpd_formats)
|
||||||
|
else:
|
||||||
|
self.report_warning(f'{ff} formats are currently not supported')
|
||||||
|
|
||||||
|
if not extracted_formats:
|
||||||
|
if video_format.get('s'):
|
||||||
|
self.report_warning(f'{self._BID_TAGS[bid]} format is restricted')
|
||||||
|
else:
|
||||||
|
self.report_warning(f'Unable to extract {self._BID_TAGS[bid]} format')
|
||||||
|
for f in extracted_formats:
|
||||||
|
f.update({
|
||||||
|
'quality': qualities(list(self._BID_TAGS.keys()))(bid),
|
||||||
|
'format_note': self._BID_TAGS[bid],
|
||||||
|
**parse_resolution(video_format.get('scrsz'))
|
||||||
|
})
|
||||||
|
formats.extend(extracted_formats)
|
||||||
|
|
||||||
|
self._sort_formats(formats)
|
||||||
|
|
||||||
|
for sub_format in traverse_obj(initial_format_data, ('program', 'stl', ...), expected_type=dict, default=[]):
|
||||||
|
lang = self._LID_TAGS.get(str_or_none(sub_format.get('lid')), sub_format.get('_name'))
|
||||||
|
subtitles.setdefault(lang, []).extend([{
|
||||||
|
'ext': format_ext,
|
||||||
|
'url': urljoin(initial_format_data.get('dstl', 'http://meta.video.iqiyi.com'), sub_format[format_key])
|
||||||
|
} for format_key, format_ext in [('srt', 'srt'), ('webvtt', 'vtt')] if sub_format.get(format_key)])
|
||||||
|
|
||||||
|
extra_metadata = page_data.get('albumInfo') if video_info.get('albumId') and page_data.get('albumInfo') else video_info
|
||||||
|
return {
|
||||||
|
'id': video_id,
|
||||||
|
'title': video_info['name'],
|
||||||
|
'formats': formats,
|
||||||
|
'subtitles': subtitles,
|
||||||
|
'description': video_info.get('mergeDesc'),
|
||||||
|
'duration': parse_duration(video_info.get('len')),
|
||||||
|
'age_limit': parse_age_limit(video_info.get('rating')),
|
||||||
|
'average_rating': traverse_obj(page_data, ('playScoreInfo', 'score'), expected_type=float_or_none),
|
||||||
|
'timestamp': parse_iso8601(video_info.get('isoUploadDate')),
|
||||||
|
'categories': traverse_obj(extra_metadata, ('videoTagMap', ..., ..., 'name'), expected_type=str),
|
||||||
|
'cast': traverse_obj(extra_metadata, ('actorArr', ..., 'name'), expected_type=str),
|
||||||
|
'episode_number': int_or_none(video_info.get('order')) or None,
|
||||||
|
'series': video_info.get('albumName'),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class IqAlbumIE(InfoExtractor):
|
||||||
|
IE_NAME = 'iq.com:album'
|
||||||
|
_VALID_URL = r'https?://(?:www\.)?iq\.com/album/(?:[\w%-]*-)?(?P<id>\w+)'
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://www.iq.com/album/one-piece-1999-1bk9icvr331',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '1bk9icvr331',
|
||||||
|
'title': 'One Piece',
|
||||||
|
'description': 'Subtitle available on Sunday 4PM(GMT+8).'
|
||||||
|
},
|
||||||
|
'playlist_mincount': 238
|
||||||
|
}, {
|
||||||
|
# Movie/single video
|
||||||
|
'url': 'https://www.iq.com/album/九龙城寨-2021-22yjnij099k',
|
||||||
|
'info_dict': {
|
||||||
|
'ext': 'mp4',
|
||||||
|
'id': '22yjnij099k',
|
||||||
|
'title': '九龙城寨',
|
||||||
|
'description': 'md5:8a09f50b8ba0db4dc69bc7c844228044',
|
||||||
|
'duration': 5000,
|
||||||
|
'timestamp': 1641911371,
|
||||||
|
'upload_date': '20220111',
|
||||||
|
'series': '九龙城寨',
|
||||||
|
'cast': ['Shi Yan Neng', 'Yu Lang', 'Peter lv', 'Sun Zi Jun', 'Yang Xiao Bo'],
|
||||||
|
'age_limit': 13,
|
||||||
|
'average_rating': float,
|
||||||
|
},
|
||||||
|
'expected_warnings': ['format is restricted']
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _entries(self, album_id_num, page_ranges, album_id=None, mode_code='intl', lang_code='en_us'):
|
||||||
|
for page_range in page_ranges:
|
||||||
|
page = self._download_json(
|
||||||
|
f'https://pcw-api.iq.com/api/episodeListSource/{album_id_num}', album_id,
|
||||||
|
note=f'Downloading video list episodes {page_range.get("msg", "")}',
|
||||||
|
errnote='Unable to download video list', query={
|
||||||
|
'platformId': 3,
|
||||||
|
'modeCode': mode_code,
|
||||||
|
'langCode': lang_code,
|
||||||
|
'endOrder': page_range['to'],
|
||||||
|
'startOrder': page_range['from']
|
||||||
|
})
|
||||||
|
for video in page['data']['epg']:
|
||||||
|
yield self.url_result('https://www.iq.com/play/%s' % (video.get('playLocSuffix') or video['qipuIdStr']),
|
||||||
|
IqIE.ie_key(), video.get('qipuIdStr'), video.get('name'))
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
album_id = self._match_id(url)
|
||||||
|
webpage = self._download_webpage(url, album_id)
|
||||||
|
next_data = self._search_nextjs_data(webpage, album_id)
|
||||||
|
album_data = next_data['props']['initialState']['album']['videoAlbumInfo']
|
||||||
|
|
||||||
|
if album_data.get('videoType') == 'singleVideo':
|
||||||
|
return self.url_result('https://www.iq.com/play/%s' % album_id, IqIE.ie_key())
|
||||||
|
return self.playlist_result(
|
||||||
|
self._entries(album_data['albumId'], album_data['totalPageRange'], album_id,
|
||||||
|
traverse_obj(next_data, ('props', 'initialProps', 'pageProps', 'modeCode')),
|
||||||
|
traverse_obj(next_data, ('props', 'initialProps', 'pageProps', 'langCode'))),
|
||||||
|
album_id, album_data.get('name'), album_data.get('desc'))
|
||||||
|
|||||||
@@ -243,8 +243,8 @@ class ITVBTCCIE(InfoExtractor):
|
|||||||
|
|
||||||
webpage = self._download_webpage(url, playlist_id)
|
webpage = self._download_webpage(url, playlist_id)
|
||||||
|
|
||||||
json_map = try_get(self._parse_json(self._html_search_regex(
|
json_map = try_get(
|
||||||
'(?s)<script[^>]+id=[\'"]__NEXT_DATA__[^>]*>([^<]+)</script>', webpage, 'json_map'), playlist_id),
|
self._search_nextjs_data(webpage, playlist_id),
|
||||||
lambda x: x['props']['pageProps']['article']['body']['content']) or []
|
lambda x: x['props']['pageProps']['article']['body']['content']) or []
|
||||||
|
|
||||||
entries = []
|
entries = []
|
||||||
|
|||||||
@@ -6,6 +6,7 @@ import re
|
|||||||
from .common import InfoExtractor
|
from .common import InfoExtractor
|
||||||
from ..compat import compat_str
|
from ..compat import compat_str
|
||||||
from ..utils import (
|
from ..utils import (
|
||||||
|
format_field,
|
||||||
int_or_none,
|
int_or_none,
|
||||||
js_to_json,
|
js_to_json,
|
||||||
try_get,
|
try_get,
|
||||||
@@ -72,7 +73,7 @@ class JojIE(InfoExtractor):
|
|||||||
r'(\d+)[pP]\.', format_url, 'height', default=None)
|
r'(\d+)[pP]\.', format_url, 'height', default=None)
|
||||||
formats.append({
|
formats.append({
|
||||||
'url': format_url,
|
'url': format_url,
|
||||||
'format_id': '%sp' % height if height else None,
|
'format_id': format_field(height, template='%sp'),
|
||||||
'height': int(height),
|
'height': int(height),
|
||||||
})
|
})
|
||||||
if not formats:
|
if not formats:
|
||||||
|
|||||||
+35
-11
@@ -3,10 +3,12 @@
|
|||||||
from __future__ import unicode_literals
|
from __future__ import unicode_literals
|
||||||
|
|
||||||
from .common import InfoExtractor
|
from .common import InfoExtractor
|
||||||
from ..compat import compat_str
|
from ..compat import compat_HTTPError
|
||||||
from ..utils import (
|
from ..utils import (
|
||||||
|
ExtractorError,
|
||||||
int_or_none,
|
int_or_none,
|
||||||
strip_or_none,
|
strip_or_none,
|
||||||
|
str_or_none,
|
||||||
traverse_obj,
|
traverse_obj,
|
||||||
unified_timestamp,
|
unified_timestamp,
|
||||||
)
|
)
|
||||||
@@ -24,10 +26,17 @@ class KakaoIE(InfoExtractor):
|
|||||||
'id': '301965083',
|
'id': '301965083',
|
||||||
'ext': 'mp4',
|
'ext': 'mp4',
|
||||||
'title': '乃木坂46 バナナマン 「3期生紹介コーナーが始動!顔高低差GPも!」 『乃木坂工事中』',
|
'title': '乃木坂46 バナナマン 「3期生紹介コーナーが始動!顔高低差GPも!」 『乃木坂工事中』',
|
||||||
'uploader_id': 2671005,
|
'description': '',
|
||||||
|
'uploader_id': '2671005',
|
||||||
'uploader': '그랑그랑이',
|
'uploader': '그랑그랑이',
|
||||||
'timestamp': 1488160199,
|
'timestamp': 1488160199,
|
||||||
'upload_date': '20170227',
|
'upload_date': '20170227',
|
||||||
|
'like_count': int,
|
||||||
|
'thumbnail': r're:http://.+/thumb\.png',
|
||||||
|
'tags': ['乃木坂'],
|
||||||
|
'view_count': int,
|
||||||
|
'duration': 1503,
|
||||||
|
'comment_count': int,
|
||||||
}
|
}
|
||||||
}, {
|
}, {
|
||||||
'url': 'http://tv.kakao.com/channel/2653210/cliplink/300103180',
|
'url': 'http://tv.kakao.com/channel/2653210/cliplink/300103180',
|
||||||
@@ -37,11 +46,21 @@ class KakaoIE(InfoExtractor):
|
|||||||
'ext': 'mp4',
|
'ext': 'mp4',
|
||||||
'description': '러블리즈 - Destiny (나의 지구) (Lovelyz - Destiny)\r\n\r\n[쇼! 음악중심] 20160611, 507회',
|
'description': '러블리즈 - Destiny (나의 지구) (Lovelyz - Destiny)\r\n\r\n[쇼! 음악중심] 20160611, 507회',
|
||||||
'title': '러블리즈 - Destiny (나의 지구) (Lovelyz - Destiny)',
|
'title': '러블리즈 - Destiny (나의 지구) (Lovelyz - Destiny)',
|
||||||
'uploader_id': 2653210,
|
'uploader_id': '2653210',
|
||||||
'uploader': '쇼! 음악중심',
|
'uploader': '쇼! 음악중심',
|
||||||
'timestamp': 1485684628,
|
'timestamp': 1485684628,
|
||||||
'upload_date': '20170129',
|
'upload_date': '20170129',
|
||||||
|
'like_count': int,
|
||||||
|
'thumbnail': r're:http://.+/thumb\.png',
|
||||||
|
'tags': 'count:28',
|
||||||
|
'view_count': int,
|
||||||
|
'duration': 184,
|
||||||
|
'comment_count': int,
|
||||||
}
|
}
|
||||||
|
}, {
|
||||||
|
# geo restricted
|
||||||
|
'url': 'https://tv.kakao.com/channel/3643855/cliplink/412069491',
|
||||||
|
'only_matching': True,
|
||||||
}]
|
}]
|
||||||
|
|
||||||
def _real_extract(self, url):
|
def _real_extract(self, url):
|
||||||
@@ -73,19 +92,24 @@ class KakaoIE(InfoExtractor):
|
|||||||
title = clip.get('title') or clip_link.get('displayTitle')
|
title = clip.get('title') or clip_link.get('displayTitle')
|
||||||
|
|
||||||
formats = []
|
formats = []
|
||||||
for fmt in clip.get('videoOutputList', []):
|
for fmt in clip.get('videoOutputList') or []:
|
||||||
profile_name = fmt.get('profile')
|
profile_name = fmt.get('profile')
|
||||||
if not profile_name or profile_name == 'AUDIO':
|
if not profile_name or profile_name == 'AUDIO':
|
||||||
continue
|
continue
|
||||||
query.update({
|
query.update({
|
||||||
'profile': profile_name,
|
'profile': profile_name,
|
||||||
'fields': '-*,url',
|
'fields': '-*,code,message,url',
|
||||||
})
|
})
|
||||||
|
try:
|
||||||
|
fmt_url_json = self._download_json(
|
||||||
|
cdn_api_base, video_id, query=query,
|
||||||
|
note='Downloading video URL for profile %s' % profile_name)
|
||||||
|
except ExtractorError as e:
|
||||||
|
if isinstance(e.cause, compat_HTTPError) and e.cause.code == 403:
|
||||||
|
resp = self._parse_json(e.cause.read().decode(), video_id)
|
||||||
|
if resp.get('code') == 'GeoBlocked':
|
||||||
|
self.raise_geo_restricted()
|
||||||
|
|
||||||
fmt_url_json = self._download_json(
|
|
||||||
cdn_api_base, video_id,
|
|
||||||
'Downloading video URL for profile %s' % profile_name,
|
|
||||||
query=query, fatal=False)
|
|
||||||
fmt_url = traverse_obj(fmt_url_json, ('videoLocation', 'url'))
|
fmt_url = traverse_obj(fmt_url_json, ('videoLocation', 'url'))
|
||||||
if not fmt_url:
|
if not fmt_url:
|
||||||
continue
|
continue
|
||||||
@@ -105,7 +129,7 @@ class KakaoIE(InfoExtractor):
|
|||||||
for thumb in clip.get('clipChapterThumbnailList') or []:
|
for thumb in clip.get('clipChapterThumbnailList') or []:
|
||||||
thumbs.append({
|
thumbs.append({
|
||||||
'url': thumb.get('thumbnailUrl'),
|
'url': thumb.get('thumbnailUrl'),
|
||||||
'id': compat_str(thumb.get('timeInSec')),
|
'id': str(thumb.get('timeInSec')),
|
||||||
'preference': -1 if thumb.get('isDefault') else 0
|
'preference': -1 if thumb.get('isDefault') else 0
|
||||||
})
|
})
|
||||||
top_thumbnail = clip.get('thumbnailUrl')
|
top_thumbnail = clip.get('thumbnailUrl')
|
||||||
@@ -120,7 +144,7 @@ class KakaoIE(InfoExtractor):
|
|||||||
'title': title,
|
'title': title,
|
||||||
'description': strip_or_none(clip.get('description')),
|
'description': strip_or_none(clip.get('description')),
|
||||||
'uploader': traverse_obj(clip_link, ('channel', 'name')),
|
'uploader': traverse_obj(clip_link, ('channel', 'name')),
|
||||||
'uploader_id': clip_link.get('channelId'),
|
'uploader_id': str_or_none(clip_link.get('channelId')),
|
||||||
'thumbnails': thumbs,
|
'thumbnails': thumbs,
|
||||||
'timestamp': unified_timestamp(clip_link.get('createTime')),
|
'timestamp': unified_timestamp(clip_link.get('createTime')),
|
||||||
'duration': int_or_none(clip.get('duration')),
|
'duration': int_or_none(clip.get('duration')),
|
||||||
|
|||||||
@@ -12,6 +12,7 @@ from ..compat import (
|
|||||||
from ..utils import (
|
from ..utils import (
|
||||||
clean_html,
|
clean_html,
|
||||||
ExtractorError,
|
ExtractorError,
|
||||||
|
format_field,
|
||||||
int_or_none,
|
int_or_none,
|
||||||
unsmuggle_url,
|
unsmuggle_url,
|
||||||
smuggle_url,
|
smuggle_url,
|
||||||
@@ -372,6 +373,6 @@ class KalturaIE(InfoExtractor):
|
|||||||
'thumbnail': info.get('thumbnailUrl'),
|
'thumbnail': info.get('thumbnailUrl'),
|
||||||
'duration': info.get('duration'),
|
'duration': info.get('duration'),
|
||||||
'timestamp': info.get('createdAt'),
|
'timestamp': info.get('createdAt'),
|
||||||
'uploader_id': info.get('userId') if info.get('userId') != 'None' else None,
|
'uploader_id': format_field(info, 'userId', ignore=('None', None)),
|
||||||
'view_count': info.get('plays'),
|
'view_count': info.get('plays'),
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -8,6 +8,7 @@ from ..compat import compat_urllib_parse_unquote
|
|||||||
from ..utils import (
|
from ..utils import (
|
||||||
determine_ext,
|
determine_ext,
|
||||||
ExtractorError,
|
ExtractorError,
|
||||||
|
format_field,
|
||||||
int_or_none,
|
int_or_none,
|
||||||
str_to_int,
|
str_to_int,
|
||||||
strip_or_none,
|
strip_or_none,
|
||||||
@@ -69,7 +70,7 @@ class KeezMoviesIE(InfoExtractor):
|
|||||||
video_url, title, 32).decode('utf-8')
|
video_url, title, 32).decode('utf-8')
|
||||||
formats.append({
|
formats.append({
|
||||||
'url': format_url,
|
'url': format_url,
|
||||||
'format_id': '%dp' % height if height else None,
|
'format_id': format_field(height, template='%dp'),
|
||||||
'height': height,
|
'height': height,
|
||||||
'tbr': tbr,
|
'tbr': tbr,
|
||||||
})
|
})
|
||||||
|
|||||||
@@ -0,0 +1,84 @@
|
|||||||
|
# coding: utf-8
|
||||||
|
from __future__ import unicode_literals
|
||||||
|
|
||||||
|
from .common import InfoExtractor
|
||||||
|
from ..utils import int_or_none
|
||||||
|
|
||||||
|
|
||||||
|
class KelbyOneIE(InfoExtractor):
|
||||||
|
_VALID_URL = r'https?://members\.kelbyone\.com/course/(?P<id>[^$&?#/]+)'
|
||||||
|
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://members.kelbyone.com/course/glyn-dewis-mastering-selections/',
|
||||||
|
'playlist_mincount': 1,
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'glyn-dewis-mastering-selections',
|
||||||
|
'title': 'Trailer - Mastering Selections in Photoshop',
|
||||||
|
},
|
||||||
|
'playlist': [{
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'MkiOnLqK',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'Trailer - Mastering Selections in Photoshop',
|
||||||
|
'description': 'md5:d41d8cd98f00b204e9800998ecf8427e',
|
||||||
|
'thumbnail': 'https://content.jwplatform.com/v2/media/MkiOnLqK/poster.jpg?width=720',
|
||||||
|
'timestamp': 1601568639,
|
||||||
|
'duration': 90,
|
||||||
|
'upload_date': '20201001',
|
||||||
|
},
|
||||||
|
}]
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _entries(self, playlist):
|
||||||
|
for item in playlist:
|
||||||
|
video_id = item['mediaid']
|
||||||
|
thumbnails = [{
|
||||||
|
'url': image.get('src'),
|
||||||
|
'width': int_or_none(image.get('width')),
|
||||||
|
} for image in item.get('images') or []]
|
||||||
|
formats, subtitles = [], {}
|
||||||
|
for source in item.get('sources') or []:
|
||||||
|
if not source.get('file'):
|
||||||
|
continue
|
||||||
|
if source.get('type') == 'application/vnd.apple.mpegurl':
|
||||||
|
fmts, subs = self._extract_m3u8_formats_and_subtitles(source['file'], video_id)
|
||||||
|
formats.extend(fmts)
|
||||||
|
subtitles = self._merge_subtitles(subs, subtitles)
|
||||||
|
elif source.get('type') == 'audio/mp4':
|
||||||
|
formats.append({
|
||||||
|
'format_id': source.get('label'),
|
||||||
|
'url': source['file'],
|
||||||
|
'vcodec': 'none',
|
||||||
|
})
|
||||||
|
else:
|
||||||
|
formats.append({
|
||||||
|
'format_id': source.get('label'),
|
||||||
|
'height': source.get('height'),
|
||||||
|
'width': source.get('width'),
|
||||||
|
'url': source['file'],
|
||||||
|
})
|
||||||
|
for track in item.get('tracks'):
|
||||||
|
if track.get('kind') == 'captions' and track.get('file'):
|
||||||
|
subtitles.setdefault('en', []).append({
|
||||||
|
'url': track['file'],
|
||||||
|
})
|
||||||
|
self._sort_formats(formats)
|
||||||
|
yield {
|
||||||
|
'id': video_id,
|
||||||
|
'title': item['title'],
|
||||||
|
'description': item.get('description'),
|
||||||
|
'thumbnails': thumbnails,
|
||||||
|
'thumbnail': item.get('image'),
|
||||||
|
'timestamp': item.get('pubdate'),
|
||||||
|
'duration': item.get('duration'),
|
||||||
|
'formats': formats,
|
||||||
|
'subtitles': subtitles,
|
||||||
|
}
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
item_id = self._match_id(url)
|
||||||
|
webpage = self._download_webpage(url, item_id)
|
||||||
|
playlist_url = self._html_search_regex(r'playlist"\:"(https.*content\.jwplatform\.com.*json)"', webpage, 'playlist url').replace('\\', '')
|
||||||
|
course_data = self._download_json(playlist_url, item_id)
|
||||||
|
return self.playlist_result(self._entries(course_data['playlist']), item_id,
|
||||||
|
course_data.get('title'), course_data.get('description'))
|
||||||
+15
-95
@@ -5,95 +5,12 @@ from __future__ import unicode_literals
|
|||||||
from .common import InfoExtractor
|
from .common import InfoExtractor
|
||||||
from ..compat import compat_str
|
from ..compat import compat_str
|
||||||
from ..utils import (
|
from ..utils import (
|
||||||
|
format_field,
|
||||||
int_or_none,
|
int_or_none,
|
||||||
js_to_json,
|
|
||||||
str_or_none,
|
str_or_none,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
class LineTVIE(InfoExtractor):
|
|
||||||
_VALID_URL = r'https?://tv\.line\.me/v/(?P<id>\d+)_[^/]+-(?P<segment>ep\d+-\d+)'
|
|
||||||
|
|
||||||
_TESTS = [{
|
|
||||||
'url': 'https://tv.line.me/v/793123_goodbye-mrblack-ep1-1/list/69246',
|
|
||||||
'info_dict': {
|
|
||||||
'id': '793123_ep1-1',
|
|
||||||
'ext': 'mp4',
|
|
||||||
'title': 'Goodbye Mr.Black | EP.1-1',
|
|
||||||
'thumbnail': r're:^https?://.*\.jpg$',
|
|
||||||
'duration': 998.509,
|
|
||||||
'view_count': int,
|
|
||||||
},
|
|
||||||
}, {
|
|
||||||
'url': 'https://tv.line.me/v/2587507_%E6%B4%BE%E9%81%A3%E5%A5%B3%E9%86%ABx-ep1-02/list/185245',
|
|
||||||
'only_matching': True,
|
|
||||||
}]
|
|
||||||
|
|
||||||
def _real_extract(self, url):
|
|
||||||
series_id, segment = self._match_valid_url(url).groups()
|
|
||||||
video_id = '%s_%s' % (series_id, segment)
|
|
||||||
|
|
||||||
webpage = self._download_webpage(url, video_id)
|
|
||||||
|
|
||||||
player_params = self._parse_json(self._search_regex(
|
|
||||||
r'naver\.WebPlayer\(({[^}]+})\)', webpage, 'player parameters'),
|
|
||||||
video_id, transform_source=js_to_json)
|
|
||||||
|
|
||||||
video_info = self._download_json(
|
|
||||||
'https://global-nvapis.line.me/linetv/rmcnmv/vod_play_videoInfo.json',
|
|
||||||
video_id, query={
|
|
||||||
'videoId': player_params['videoId'],
|
|
||||||
'key': player_params['key'],
|
|
||||||
})
|
|
||||||
|
|
||||||
stream = video_info['streams'][0]
|
|
||||||
extra_query = '?__gda__=' + stream['key']['value']
|
|
||||||
formats = self._extract_m3u8_formats(
|
|
||||||
stream['source'] + extra_query, video_id, ext='mp4',
|
|
||||||
entry_protocol='m3u8_native', m3u8_id='hls')
|
|
||||||
|
|
||||||
for a_format in formats:
|
|
||||||
a_format['url'] += extra_query
|
|
||||||
|
|
||||||
duration = None
|
|
||||||
for video in video_info.get('videos', {}).get('list', []):
|
|
||||||
encoding_option = video.get('encodingOption', {})
|
|
||||||
abr = video['bitrate']['audio']
|
|
||||||
vbr = video['bitrate']['video']
|
|
||||||
tbr = abr + vbr
|
|
||||||
formats.append({
|
|
||||||
'url': video['source'],
|
|
||||||
'format_id': 'http-%d' % int(tbr),
|
|
||||||
'height': encoding_option.get('height'),
|
|
||||||
'width': encoding_option.get('width'),
|
|
||||||
'abr': abr,
|
|
||||||
'vbr': vbr,
|
|
||||||
'filesize': video.get('size'),
|
|
||||||
})
|
|
||||||
if video.get('duration') and duration is None:
|
|
||||||
duration = video['duration']
|
|
||||||
|
|
||||||
self._sort_formats(formats)
|
|
||||||
|
|
||||||
if formats and not formats[0].get('width'):
|
|
||||||
formats[0]['vcodec'] = 'none'
|
|
||||||
|
|
||||||
title = self._og_search_title(webpage)
|
|
||||||
|
|
||||||
# like_count requires an additional API request https://tv.line.me/api/likeit/getCount
|
|
||||||
|
|
||||||
return {
|
|
||||||
'id': video_id,
|
|
||||||
'title': title,
|
|
||||||
'formats': formats,
|
|
||||||
'extra_param_to_segment_url': extra_query[1:],
|
|
||||||
'duration': duration,
|
|
||||||
'thumbnails': [{'url': thumbnail['source']}
|
|
||||||
for thumbnail in video_info.get('thumbnails', {}).get('list', [])],
|
|
||||||
'view_count': video_info.get('meta', {}).get('count'),
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
class LineLiveBaseIE(InfoExtractor):
|
class LineLiveBaseIE(InfoExtractor):
|
||||||
_API_BASE_URL = 'https://live-api.line-apps.com/web/v4.0/channel/'
|
_API_BASE_URL = 'https://live-api.line-apps.com/web/v4.0/channel/'
|
||||||
|
|
||||||
@@ -121,7 +38,7 @@ class LineLiveBaseIE(InfoExtractor):
|
|||||||
'timestamp': int_or_none(item.get('createdAt')),
|
'timestamp': int_or_none(item.get('createdAt')),
|
||||||
'channel': channel.get('name'),
|
'channel': channel.get('name'),
|
||||||
'channel_id': channel_id,
|
'channel_id': channel_id,
|
||||||
'channel_url': 'https://live.line.me/channels/' + channel_id if channel_id else None,
|
'channel_url': format_field(channel_id, template='https://live.line.me/channels/%s'),
|
||||||
'duration': int_or_none(item.get('archiveDuration')),
|
'duration': int_or_none(item.get('archiveDuration')),
|
||||||
'view_count': int_or_none(item.get('viewerCount')),
|
'view_count': int_or_none(item.get('viewerCount')),
|
||||||
'comment_count': int_or_none(item.get('chatCount')),
|
'comment_count': int_or_none(item.get('chatCount')),
|
||||||
@@ -132,16 +49,19 @@ class LineLiveBaseIE(InfoExtractor):
|
|||||||
class LineLiveIE(LineLiveBaseIE):
|
class LineLiveIE(LineLiveBaseIE):
|
||||||
_VALID_URL = r'https?://live\.line\.me/channels/(?P<channel_id>\d+)/broadcast/(?P<id>\d+)'
|
_VALID_URL = r'https?://live\.line\.me/channels/(?P<channel_id>\d+)/broadcast/(?P<id>\d+)'
|
||||||
_TESTS = [{
|
_TESTS = [{
|
||||||
'url': 'https://live.line.me/channels/4867368/broadcast/16331360',
|
'url': 'https://live.line.me/channels/5833718/broadcast/18373277',
|
||||||
'md5': 'bc931f26bf1d4f971e3b0982b3fab4a3',
|
'md5': '2c15843b8cb3acd55009ddcb2db91f7c',
|
||||||
'info_dict': {
|
'info_dict': {
|
||||||
'id': '16331360',
|
'id': '18373277',
|
||||||
'title': '振りコピ講座😙😙😙',
|
'title': '2021/12/05 (15分犬)定例譲渡会🐶',
|
||||||
'ext': 'mp4',
|
'ext': 'mp4',
|
||||||
'timestamp': 1617095132,
|
'timestamp': 1638674925,
|
||||||
'upload_date': '20210330',
|
'upload_date': '20211205',
|
||||||
'channel': '白川ゆめか',
|
'thumbnail': 'md5:e1f5817e60f4a72b7e43377cf308d7ef',
|
||||||
'channel_id': '4867368',
|
'channel_url': 'https://live.line.me/channels/5833718',
|
||||||
|
'channel': 'Yahooニュース掲載🗞プロフ見てね🐕🐕',
|
||||||
|
'channel_id': '5833718',
|
||||||
|
'duration': 937,
|
||||||
'view_count': int,
|
'view_count': int,
|
||||||
'comment_count': int,
|
'comment_count': int,
|
||||||
'is_live': False,
|
'is_live': False,
|
||||||
@@ -193,8 +113,8 @@ class LineLiveChannelIE(LineLiveBaseIE):
|
|||||||
'url': 'https://live.line.me/channels/5893542',
|
'url': 'https://live.line.me/channels/5893542',
|
||||||
'info_dict': {
|
'info_dict': {
|
||||||
'id': '5893542',
|
'id': '5893542',
|
||||||
'title': 'いくらちゃん',
|
'title': 'いくらちゃんだよぉ🦒',
|
||||||
'description': 'md5:c3a4af801f43b2fac0b02294976580be',
|
'description': 'md5:4d418087973ad081ceb1b3481f0b1816',
|
||||||
},
|
},
|
||||||
'playlist_mincount': 29
|
'playlist_mincount': 29
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -6,8 +6,10 @@ from .common import InfoExtractor
|
|||||||
from ..utils import (
|
from ..utils import (
|
||||||
clean_html,
|
clean_html,
|
||||||
compat_str,
|
compat_str,
|
||||||
|
format_field,
|
||||||
int_or_none,
|
int_or_none,
|
||||||
parse_iso8601,
|
parse_iso8601,
|
||||||
|
unified_strdate,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -71,17 +73,97 @@ class LnkGoIE(InfoExtractor):
|
|||||||
video_id, 'mp4', 'm3u8_native')
|
video_id, 'mp4', 'm3u8_native')
|
||||||
self._sort_formats(formats)
|
self._sort_formats(formats)
|
||||||
|
|
||||||
poster_image = video_info.get('posterImage')
|
|
||||||
|
|
||||||
return {
|
return {
|
||||||
'id': video_id,
|
'id': video_id,
|
||||||
'display_id': display_id,
|
'display_id': display_id,
|
||||||
'title': title,
|
'title': title,
|
||||||
'formats': formats,
|
'formats': formats,
|
||||||
'thumbnail': 'https://lnk.lt/all-images/' + poster_image if poster_image else None,
|
'thumbnail': format_field(video_info, 'posterImage', 'https://lnk.lt/all-images/%s'),
|
||||||
'duration': int_or_none(video_info.get('duration')),
|
'duration': int_or_none(video_info.get('duration')),
|
||||||
'description': clean_html(video_info.get('htmlDescription')),
|
'description': clean_html(video_info.get('htmlDescription')),
|
||||||
'age_limit': self._AGE_LIMITS.get(video_info.get('pgRating'), 0),
|
'age_limit': self._AGE_LIMITS.get(video_info.get('pgRating'), 0),
|
||||||
'timestamp': parse_iso8601(video_info.get('airDate')),
|
'timestamp': parse_iso8601(video_info.get('airDate')),
|
||||||
'view_count': int_or_none(video_info.get('viewsCount')),
|
'view_count': int_or_none(video_info.get('viewsCount')),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class LnkIE(InfoExtractor):
|
||||||
|
_VALID_URL = r'https?://(?:www\.)?lnk\.lt/[^/]+/(?P<id>\d+)'
|
||||||
|
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://lnk.lt/zinios/79791',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '79791',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'LNK.lt: Viešintų gyventojai sukilo prieš radijo bangų siųstuvą',
|
||||||
|
'description': 'Svarbiausios naujienos trumpai, LNK žinios ir Info dienos pokalbiai.',
|
||||||
|
'view_count': int,
|
||||||
|
'duration': 233,
|
||||||
|
'upload_date': '20191123',
|
||||||
|
'thumbnail': r're:^https?://.*\.jpg$',
|
||||||
|
'episode_number': 13431,
|
||||||
|
'series': 'Naujausi žinių reportažai',
|
||||||
|
'episode': 'Episode 13431'
|
||||||
|
},
|
||||||
|
'params': {'skip_download': True}
|
||||||
|
}, {
|
||||||
|
'url': 'https://lnk.lt/istorijos-trumpai/152546',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '152546',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'Radžio koncertas gaisre ',
|
||||||
|
'description': 'md5:0666b5b85cb9fc7c1238dec96f71faba',
|
||||||
|
'view_count': int,
|
||||||
|
'duration': 54,
|
||||||
|
'upload_date': '20220105',
|
||||||
|
'thumbnail': r're:^https?://.*\.jpg$',
|
||||||
|
'episode_number': 1036,
|
||||||
|
'series': 'Istorijos trumpai',
|
||||||
|
'episode': 'Episode 1036'
|
||||||
|
},
|
||||||
|
'params': {'skip_download': True}
|
||||||
|
}, {
|
||||||
|
'url': 'https://lnk.lt/gyvunu-pasaulis/151549',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '151549',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'Gyvūnų pasaulis',
|
||||||
|
'description': '',
|
||||||
|
'view_count': int,
|
||||||
|
'duration': 1264,
|
||||||
|
'upload_date': '20220108',
|
||||||
|
'thumbnail': r're:^https?://.*\.jpg$',
|
||||||
|
'episode_number': 16,
|
||||||
|
'series': 'Gyvūnų pasaulis',
|
||||||
|
'episode': 'Episode 16'
|
||||||
|
},
|
||||||
|
'params': {'skip_download': True}
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
id = self._match_id(url)
|
||||||
|
video_json = self._download_json(f'https://lnk.lt/api/video/video-config/{id}', id)['videoInfo']
|
||||||
|
formats, subtitles = [], {}
|
||||||
|
if video_json.get('videoUrl'):
|
||||||
|
fmts, subs = self._extract_m3u8_formats_and_subtitles(video_json['videoUrl'], id)
|
||||||
|
formats.extend(fmts)
|
||||||
|
subtitles = self._merge_subtitles(subtitles, subs)
|
||||||
|
if video_json.get('videoFairplayUrl') and not video_json.get('drm'):
|
||||||
|
fmts, subs = self._extract_m3u8_formats_and_subtitles(video_json['videoFairplayUrl'], id)
|
||||||
|
formats.extend(fmts)
|
||||||
|
subtitles = self._merge_subtitles(subtitles, subs)
|
||||||
|
|
||||||
|
self._sort_formats(formats)
|
||||||
|
return {
|
||||||
|
'id': id,
|
||||||
|
'title': video_json.get('title'),
|
||||||
|
'description': video_json.get('description'),
|
||||||
|
'view_count': video_json.get('viewsCount'),
|
||||||
|
'duration': video_json.get('duration'),
|
||||||
|
'upload_date': unified_strdate(video_json.get('airDate')),
|
||||||
|
'thumbnail': format_field(video_json, 'posterImage', 'https://lnk.lt/all-images/%s'),
|
||||||
|
'episode_number': int_or_none(video_json.get('episodeNumber')),
|
||||||
|
'series': video_json.get('programTitle'),
|
||||||
|
'formats': formats,
|
||||||
|
'subtitles': subtitles,
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,219 @@
|
|||||||
|
# coding: utf-8
|
||||||
|
import re
|
||||||
|
|
||||||
|
from .common import InfoExtractor
|
||||||
|
|
||||||
|
from ..utils import (
|
||||||
|
int_or_none,
|
||||||
|
js_to_json,
|
||||||
|
parse_duration,
|
||||||
|
traverse_obj,
|
||||||
|
try_get,
|
||||||
|
urljoin
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class MainStreamingIE(InfoExtractor):
|
||||||
|
_VALID_URL = r'https?://(?:webtools-?)?(?P<host>[A-Za-z0-9-]*\.msvdn.net)/(?:embed|amp_embed|content)/(?P<id>\w+)'
|
||||||
|
IE_DESC = 'MainStreaming Player'
|
||||||
|
|
||||||
|
_TESTS = [
|
||||||
|
{
|
||||||
|
# Live stream offline, has alternative content id
|
||||||
|
'url': 'https://webtools-e18da6642b684f8aa9ae449862783a56.msvdn.net/embed/53EN6GxbWaJC',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '53EN6GxbWaJC',
|
||||||
|
'title': 'Diretta homepage 2021-12-31 12:00',
|
||||||
|
'description': '',
|
||||||
|
'live_status': 'was_live',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'thumbnail': r're:https?://[A-Za-z0-9-]*\.msvdn.net/image/\w+/poster',
|
||||||
|
},
|
||||||
|
'expected_warnings': [
|
||||||
|
'Ignoring alternative content ID: WDAF1KOWUpH3',
|
||||||
|
'MainStreaming said: Live event is OFFLINE'
|
||||||
|
],
|
||||||
|
'skip': 'live stream offline'
|
||||||
|
}, {
|
||||||
|
# playlist
|
||||||
|
'url': 'https://webtools-e18da6642b684f8aa9ae449862783a56.msvdn.net/embed/WDAF1KOWUpH3',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'WDAF1KOWUpH3',
|
||||||
|
'title': 'Playlist homepage',
|
||||||
|
},
|
||||||
|
'playlist_mincount': 2
|
||||||
|
}, {
|
||||||
|
# livestream
|
||||||
|
'url': 'https://webtools-859c1818ed614cc5b0047439470927b0.msvdn.net/embed/tDoFkZD3T1Lw',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'tDoFkZD3T1Lw',
|
||||||
|
'title': r're:Class CNBC Live \d{4}-\d{2}-\d{2} \d{2}:\d{2}$',
|
||||||
|
'live_status': 'is_live',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'thumbnail': r're:https?://[A-Za-z0-9-]*\.msvdn.net/image/\w+/poster',
|
||||||
|
},
|
||||||
|
'skip': 'live stream'
|
||||||
|
}, {
|
||||||
|
'url': 'https://webtools-f5842579ff984c1c98d63b8d789673eb.msvdn.net/embed/EUlZfGWkGpOd?autoPlay=false',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'EUlZfGWkGpOd',
|
||||||
|
'title': 'La Settimana ',
|
||||||
|
'description': '03 Ottobre ore 02:00',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'live_status': 'not_live',
|
||||||
|
'thumbnail': r're:https?://[A-Za-z0-9-]*\.msvdn.net/image/\w+/poster',
|
||||||
|
'duration': 1512
|
||||||
|
}
|
||||||
|
}, {
|
||||||
|
# video without webtools- prefix
|
||||||
|
'url': 'https://f5842579ff984c1c98d63b8d789673eb.msvdn.net/embed/MfuWmzL2lGkA?autoplay=false&T=1635860445',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'MfuWmzL2lGkA',
|
||||||
|
'title': 'TG Mattina',
|
||||||
|
'description': '06 Ottobre ore 08:00',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'live_status': 'not_live',
|
||||||
|
'thumbnail': r're:https?://[A-Za-z0-9-]*\.msvdn.net/image/\w+/poster',
|
||||||
|
'duration': 789.04
|
||||||
|
}
|
||||||
|
}, {
|
||||||
|
# always-on livestream with DVR
|
||||||
|
'url': 'https://webtools-f5842579ff984c1c98d63b8d789673eb.msvdn.net/embed/HVvPMzy',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'HVvPMzy',
|
||||||
|
'title': r're:^Diretta LaC News24 \d{4}-\d{2}-\d{2} \d{2}:\d{2}$',
|
||||||
|
'description': 'canale all news',
|
||||||
|
'live_status': 'is_live',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'thumbnail': r're:https?://[A-Za-z0-9-]*\.msvdn.net/image/\w+/poster',
|
||||||
|
},
|
||||||
|
'params': {
|
||||||
|
'skip_download': True,
|
||||||
|
},
|
||||||
|
}, {
|
||||||
|
# no host
|
||||||
|
'url': 'https://webtools.msvdn.net/embed/MfuWmzL2lGkA',
|
||||||
|
'only_matching': True
|
||||||
|
}, {
|
||||||
|
'url': 'https://859c1818ed614cc5b0047439470927b0.msvdn.net/amp_embed/tDoFkZD3T1Lw',
|
||||||
|
'only_matching': True
|
||||||
|
}, {
|
||||||
|
'url': 'https://859c1818ed614cc5b0047439470927b0.msvdn.net/content/tDoFkZD3T1Lw#',
|
||||||
|
'only_matching': True
|
||||||
|
}
|
||||||
|
]
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _extract_urls(webpage):
|
||||||
|
mobj = re.findall(
|
||||||
|
r'<iframe[^>]+?src=["\']?(?P<url>%s)["\']?' % MainStreamingIE._VALID_URL, webpage)
|
||||||
|
if mobj:
|
||||||
|
return [group[0] for group in mobj]
|
||||||
|
|
||||||
|
def _playlist_entries(self, host, playlist_content):
|
||||||
|
for entry in playlist_content:
|
||||||
|
content_id = entry.get('contentID')
|
||||||
|
yield {
|
||||||
|
'_type': 'url',
|
||||||
|
'ie_key': MainStreamingIE.ie_key(),
|
||||||
|
'id': content_id,
|
||||||
|
'duration': int_or_none(traverse_obj(entry, ('duration', 'totalSeconds'))),
|
||||||
|
'title': entry.get('title'),
|
||||||
|
'url': f'https://{host}/embed/{content_id}'
|
||||||
|
}
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _get_webtools_host(host):
|
||||||
|
if not host.startswith('webtools'):
|
||||||
|
host = 'webtools' + ('-' if not host.startswith('.') else '') + host
|
||||||
|
return host
|
||||||
|
|
||||||
|
def _get_webtools_base_url(self, host):
|
||||||
|
return f'{self.http_scheme()}//{self._get_webtools_host(host)}'
|
||||||
|
|
||||||
|
def _call_api(self, host: str, path: str, item_id: str, query=None, note='Downloading API JSON', fatal=False):
|
||||||
|
# JSON API, does not appear to be documented
|
||||||
|
return self._call_webtools_api(host, '/api/v2/' + path, item_id, query, note, fatal)
|
||||||
|
|
||||||
|
def _call_webtools_api(self, host: str, path: str, item_id: str, query=None, note='Downloading webtools API JSON', fatal=False):
|
||||||
|
# webtools docs: https://webtools.msvdn.net/
|
||||||
|
return self._download_json(
|
||||||
|
urljoin(self._get_webtools_base_url(host), path), item_id, query=query, note=note, fatal=fatal)
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
host, video_id = self._match_valid_url(url).groups()
|
||||||
|
content_info = try_get(
|
||||||
|
self._call_api(
|
||||||
|
host, f'content/{video_id}', video_id, note='Downloading content info API JSON'), lambda x: x['playerContentInfo'])
|
||||||
|
# Fallback
|
||||||
|
if not content_info:
|
||||||
|
webpage = self._download_webpage(url, video_id)
|
||||||
|
player_config = self._parse_json(
|
||||||
|
self._search_regex(
|
||||||
|
r'config\s*=\s*({.+?})\s*;', webpage, 'mainstreaming player config',
|
||||||
|
default='{}', flags=re.DOTALL),
|
||||||
|
video_id, transform_source=js_to_json, fatal=False) or {}
|
||||||
|
content_info = player_config['contentInfo']
|
||||||
|
|
||||||
|
host = content_info.get('host') or host
|
||||||
|
video_id = content_info.get('contentID') or video_id
|
||||||
|
title = content_info.get('title')
|
||||||
|
description = traverse_obj(content_info, 'longDescription', 'shortDescription', expected_type=str)
|
||||||
|
live_status = 'not_live'
|
||||||
|
if content_info.get('drmEnabled'):
|
||||||
|
self.report_drm(video_id)
|
||||||
|
|
||||||
|
alternative_content_id = content_info.get('alternativeContentID')
|
||||||
|
if alternative_content_id:
|
||||||
|
self.report_warning(f'Ignoring alternative content ID: {alternative_content_id}')
|
||||||
|
|
||||||
|
content_type = int_or_none(content_info.get('contentType'))
|
||||||
|
format_base_url = None
|
||||||
|
formats = []
|
||||||
|
subtitles = {}
|
||||||
|
# Live content
|
||||||
|
if content_type == 20:
|
||||||
|
dvr_enabled = traverse_obj(content_info, ('playerSettings', 'dvrEnabled'), expected_type=bool)
|
||||||
|
format_base_url = f"https://{host}/live/{content_info['liveSourceID']}/{video_id}/%s{'?DVR' if dvr_enabled else ''}"
|
||||||
|
live_status = 'is_live'
|
||||||
|
heartbeat = self._call_api(host, f'heartbeat/{video_id}', video_id, note='Checking stream status') or {}
|
||||||
|
if heartbeat.get('heartBeatUp') is False:
|
||||||
|
self.raise_no_formats(f'MainStreaming said: {heartbeat.get("responseMessage")}', expected=True)
|
||||||
|
live_status = 'was_live'
|
||||||
|
|
||||||
|
# Playlist
|
||||||
|
elif content_type == 31:
|
||||||
|
return self.playlist_result(
|
||||||
|
self._playlist_entries(host, content_info.get('playlistContents')), video_id, title, description)
|
||||||
|
# Normal video content?
|
||||||
|
elif content_type == 10:
|
||||||
|
format_base_url = f'https://{host}/vod/{video_id}/%s'
|
||||||
|
# Progressive format
|
||||||
|
# Note: in https://webtools.msvdn.net/loader/playerV2.js there is mention of original.mp3 format,
|
||||||
|
# however it seems to be the same as original.mp4?
|
||||||
|
formats.append({'url': format_base_url % 'original.mp4', 'format_note': 'original', 'quality': 1})
|
||||||
|
else:
|
||||||
|
self.raise_no_formats(f'Unknown content type {content_type}')
|
||||||
|
|
||||||
|
if format_base_url:
|
||||||
|
m3u8_formats, m3u8_subs = self._extract_m3u8_formats_and_subtitles(
|
||||||
|
format_base_url % 'playlist.m3u8', video_id=video_id, fatal=False)
|
||||||
|
mpd_formats, mpd_subs = self._extract_mpd_formats_and_subtitles(
|
||||||
|
format_base_url % 'manifest.mpd', video_id=video_id, fatal=False)
|
||||||
|
|
||||||
|
subtitles = self._merge_subtitles(m3u8_subs, mpd_subs)
|
||||||
|
formats.extend(m3u8_formats + mpd_formats)
|
||||||
|
|
||||||
|
self._sort_formats(formats)
|
||||||
|
|
||||||
|
return {
|
||||||
|
'id': video_id,
|
||||||
|
'title': title,
|
||||||
|
'description': description,
|
||||||
|
'formats': formats,
|
||||||
|
'live_status': live_status,
|
||||||
|
'duration': parse_duration(content_info.get('duration')),
|
||||||
|
'tags': content_info.get('tags'),
|
||||||
|
'subtitles': subtitles,
|
||||||
|
'thumbnail': urljoin(self._get_webtools_base_url(host), f'image/{video_id}/poster')
|
||||||
|
}
|
||||||
@@ -7,6 +7,7 @@ from .common import InfoExtractor
|
|||||||
from ..compat import compat_str
|
from ..compat import compat_str
|
||||||
from ..utils import (
|
from ..utils import (
|
||||||
ExtractorError,
|
ExtractorError,
|
||||||
|
format_field,
|
||||||
float_or_none,
|
float_or_none,
|
||||||
int_or_none,
|
int_or_none,
|
||||||
str_or_none,
|
str_or_none,
|
||||||
@@ -118,7 +119,7 @@ class MedalTVIE(InfoExtractor):
|
|||||||
author = try_get(
|
author = try_get(
|
||||||
hydration_data, lambda x: list(x['profiles'].values())[0], dict) or {}
|
hydration_data, lambda x: list(x['profiles'].values())[0], dict) or {}
|
||||||
author_id = str_or_none(author.get('id'))
|
author_id = str_or_none(author.get('id'))
|
||||||
author_url = 'https://medal.tv/users/{0}'.format(author_id) if author_id else None
|
author_url = format_field(author_id, template='https://medal.tv/users/%s')
|
||||||
|
|
||||||
return {
|
return {
|
||||||
'id': video_id,
|
'id': video_id,
|
||||||
|
|||||||
@@ -0,0 +1,173 @@
|
|||||||
|
# coding: utf-8
|
||||||
|
from __future__ import unicode_literals
|
||||||
|
|
||||||
|
import re
|
||||||
|
|
||||||
|
from .common import InfoExtractor
|
||||||
|
from ..utils import (
|
||||||
|
clean_html,
|
||||||
|
determine_ext,
|
||||||
|
ExtractorError,
|
||||||
|
extract_attributes,
|
||||||
|
get_element_by_class,
|
||||||
|
get_element_html_by_id,
|
||||||
|
HEADRequest,
|
||||||
|
parse_qs,
|
||||||
|
unescapeHTML,
|
||||||
|
unified_timestamp,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class MegaTVComBaseIE(InfoExtractor):
|
||||||
|
_PLAYER_DIV_ID = 'player_div_id'
|
||||||
|
|
||||||
|
def _extract_player_attrs(self, webpage):
|
||||||
|
player_el = get_element_html_by_id(self._PLAYER_DIV_ID, webpage)
|
||||||
|
return {
|
||||||
|
re.sub(r'^data-(?:kwik_)?', '', k): v
|
||||||
|
for k, v in extract_attributes(player_el).items()
|
||||||
|
if k not in ('id',)
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class MegaTVComIE(MegaTVComBaseIE):
|
||||||
|
IE_NAME = 'megatvcom'
|
||||||
|
IE_DESC = 'megatv.com videos'
|
||||||
|
_VALID_URL = r'https?://(?:www\.)?megatv\.com/(?:\d{4}/\d{2}/\d{2}|[^/]+/(?P<id>\d+))/(?P<slug>[^/]+)'
|
||||||
|
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://www.megatv.com/2021/10/23/egkainia-gia-ti-nea-skini-omega-tou-dimotikou-theatrou-peiraia/',
|
||||||
|
'md5': '6546a1a37fff0dd51c9dce5f490b7d7d',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '520979',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'md5:70eef71a9cd2c1ecff7ee428354dded2',
|
||||||
|
'description': 'md5:0209fa8d318128569c0d256a5c404db1',
|
||||||
|
'timestamp': 1634975747,
|
||||||
|
'upload_date': '20211023',
|
||||||
|
'display_id': 'egkainia-gia-ti-nea-skini-omega-tou-dimotikou-theatrou-peiraia',
|
||||||
|
'thumbnail': 'https://www.megatv.com/wp-content/uploads/2021/10/ΠΕΙΡΑΙΑΣ-1024x450.jpg',
|
||||||
|
},
|
||||||
|
}, {
|
||||||
|
'url': 'https://www.megatv.com/tvshows/527800/epeisodio-65-12/',
|
||||||
|
'md5': 'cba2085d45c1abeb8e7e9b7e1d6c0072',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '527800',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'md5:fc322cb51f682eecfe2f54cd5ab3a157',
|
||||||
|
'description': 'md5:b2b7ed3690a78f2a0156eb790fdc00df',
|
||||||
|
'timestamp': 1636048859,
|
||||||
|
'upload_date': '20211104',
|
||||||
|
'display_id': 'epeisodio-65-12',
|
||||||
|
'thumbnail': 'https://www.megatv.com/wp-content/uploads/2021/11/16-1-1.jpg',
|
||||||
|
},
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
video_id, display_id = self._match_valid_url(url).group('id', 'slug')
|
||||||
|
_is_article = video_id is None
|
||||||
|
webpage = self._download_webpage(url, video_id or display_id)
|
||||||
|
if _is_article:
|
||||||
|
video_id = self._search_regex(
|
||||||
|
r'<article[^>]*\sid=["\']Article_(\d+)["\']', webpage, 'article id')
|
||||||
|
player_attrs = self._extract_player_attrs(webpage)
|
||||||
|
title = player_attrs.get('label') or self._og_search_title(webpage)
|
||||||
|
description = get_element_by_class(
|
||||||
|
'article-wrapper' if _is_article else 'story_content',
|
||||||
|
webpage)
|
||||||
|
description = clean_html(re.sub(r'<script[^>]*>[^<]+</script>', '', description))
|
||||||
|
if not description:
|
||||||
|
description = self._og_search_description(webpage)
|
||||||
|
thumbnail = player_attrs.get('image') or self._og_search_thumbnail(webpage)
|
||||||
|
timestamp = unified_timestamp(self._html_search_meta(
|
||||||
|
'article:published_time', webpage))
|
||||||
|
source = player_attrs.get('source')
|
||||||
|
if not source:
|
||||||
|
raise ExtractorError('No source found', video_id=video_id)
|
||||||
|
if determine_ext(source) == 'm3u8':
|
||||||
|
formats, subs = self._extract_m3u8_formats_and_subtitles(source, video_id, 'mp4')
|
||||||
|
else:
|
||||||
|
formats, subs = [{'url': source}], {}
|
||||||
|
if player_attrs.get('subs'):
|
||||||
|
self._merge_subtitles({'und': [{'url': player_attrs['subs']}]}, target=subs)
|
||||||
|
self._sort_formats(formats)
|
||||||
|
return {
|
||||||
|
'id': video_id,
|
||||||
|
'display_id': display_id,
|
||||||
|
'title': title,
|
||||||
|
'description': description,
|
||||||
|
'thumbnail': thumbnail,
|
||||||
|
'timestamp': timestamp,
|
||||||
|
'formats': formats,
|
||||||
|
'subtitles': subs,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class MegaTVComEmbedIE(MegaTVComBaseIE):
|
||||||
|
IE_NAME = 'megatvcom:embed'
|
||||||
|
IE_DESC = 'megatv.com embedded videos'
|
||||||
|
_VALID_URL = r'(?:https?:)?//(?:www\.)?megatv\.com/embed/?\?p=(?P<id>\d+)'
|
||||||
|
_EMBED_RE = re.compile(rf'''<iframe[^>]+?src=(?P<_q1>["'])(?P<url>{_VALID_URL})(?P=_q1)''')
|
||||||
|
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://www.megatv.com/embed/?p=2020520979',
|
||||||
|
'md5': '6546a1a37fff0dd51c9dce5f490b7d7d',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '520979',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'md5:70eef71a9cd2c1ecff7ee428354dded2',
|
||||||
|
'description': 'md5:0209fa8d318128569c0d256a5c404db1',
|
||||||
|
'timestamp': 1634975747,
|
||||||
|
'upload_date': '20211023',
|
||||||
|
'display_id': 'egkainia-gia-ti-nea-skini-omega-tou-dimotikou-theatrou-peiraia',
|
||||||
|
'thumbnail': 'https://www.megatv.com/wp-content/uploads/2021/10/ΠΕΙΡΑΙΑΣ-1024x450.jpg',
|
||||||
|
},
|
||||||
|
}, {
|
||||||
|
'url': 'https://www.megatv.com/embed/?p=2020534081',
|
||||||
|
'md5': '6ac8b3ce4dc6120c802f780a1e6b3812',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '534081',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'md5:062e9d5976ef854d8bdc1f5724d9b2d0',
|
||||||
|
'description': 'md5:36dbe4c3762d2ede9513eea8d07f6d52',
|
||||||
|
'timestamp': 1636376351,
|
||||||
|
'upload_date': '20211108',
|
||||||
|
'display_id': 'neo-rekor-stin-timi-tou-ilektrikou-reymatos-pano-apo-ta-200e-i-xondriki-timi-tou-ilektrikou',
|
||||||
|
'thumbnail': 'https://www.megatv.com/wp-content/uploads/2021/11/Capture-266.jpg',
|
||||||
|
},
|
||||||
|
}]
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def _extract_urls(cls, webpage):
|
||||||
|
for mobj in cls._EMBED_RE.finditer(webpage):
|
||||||
|
yield unescapeHTML(mobj.group('url'))
|
||||||
|
|
||||||
|
def _match_canonical_url(self, webpage):
|
||||||
|
LINK_RE = r'''(?x)
|
||||||
|
<link(?:
|
||||||
|
rel=(?P<_q1>["'])(?P<canonical>canonical)(?P=_q1)|
|
||||||
|
href=(?P<_q2>["'])(?P<href>(?:(?!(?P=_q2)).)+)(?P=_q2)|
|
||||||
|
[^>]*?
|
||||||
|
)+>
|
||||||
|
'''
|
||||||
|
for mobj in re.finditer(LINK_RE, webpage):
|
||||||
|
canonical, href = mobj.group('canonical', 'href')
|
||||||
|
if canonical and href:
|
||||||
|
return unescapeHTML(href)
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
video_id = self._match_id(url)
|
||||||
|
webpage = self._download_webpage(url, video_id)
|
||||||
|
player_attrs = self._extract_player_attrs(webpage)
|
||||||
|
canonical_url = player_attrs.get('share_url') or self._match_canonical_url(webpage)
|
||||||
|
if not canonical_url:
|
||||||
|
raise ExtractorError('canonical URL not found')
|
||||||
|
video_id = parse_qs(canonical_url)['p'][0]
|
||||||
|
|
||||||
|
# Defer to megatvcom as the metadata extracted from the embeddable page some
|
||||||
|
# times are slightly different, for the same video
|
||||||
|
canonical_url = self._request_webpage(
|
||||||
|
HEADRequest(canonical_url), video_id,
|
||||||
|
note='Resolve canonical URL',
|
||||||
|
errnote='Could not resolve canonical URL').geturl()
|
||||||
|
return self.url_result(canonical_url, MegaTVComIE.ie_key(), video_id)
|
||||||
@@ -5,6 +5,7 @@ from .common import InfoExtractor
|
|||||||
from ..compat import compat_str
|
from ..compat import compat_str
|
||||||
from ..utils import (
|
from ..utils import (
|
||||||
clean_html,
|
clean_html,
|
||||||
|
format_field,
|
||||||
int_or_none,
|
int_or_none,
|
||||||
str_or_none,
|
str_or_none,
|
||||||
strip_or_none,
|
strip_or_none,
|
||||||
@@ -120,7 +121,7 @@ class MindsIE(MindsBaseIE):
|
|||||||
'timestamp': int_or_none(entity.get('time_created')),
|
'timestamp': int_or_none(entity.get('time_created')),
|
||||||
'uploader': strip_or_none(owner.get('name')),
|
'uploader': strip_or_none(owner.get('name')),
|
||||||
'uploader_id': uploader_id,
|
'uploader_id': uploader_id,
|
||||||
'uploader_url': 'https://www.minds.com/' + uploader_id if uploader_id else None,
|
'uploader_url': format_field(uploader_id, template='https://www.minds.com/%s'),
|
||||||
'view_count': int_or_none(entity.get('play:count')),
|
'view_count': int_or_none(entity.get('play:count')),
|
||||||
'like_count': int_or_none(entity.get('thumbs:up:count')),
|
'like_count': int_or_none(entity.get('thumbs:up:count')),
|
||||||
'dislike_count': int_or_none(entity.get('thumbs:down:count')),
|
'dislike_count': int_or_none(entity.get('thumbs:down:count')),
|
||||||
|
|||||||
@@ -11,7 +11,7 @@ class MixchIE(InfoExtractor):
|
|||||||
IE_NAME = 'mixch'
|
IE_NAME = 'mixch'
|
||||||
_VALID_URL = r'https?://(?:www\.)?mixch\.tv/u/(?P<id>\d+)'
|
_VALID_URL = r'https?://(?:www\.)?mixch\.tv/u/(?P<id>\d+)'
|
||||||
|
|
||||||
TESTS = [{
|
_TESTS = [{
|
||||||
'url': 'https://mixch.tv/u/16236849/live',
|
'url': 'https://mixch.tv/u/16236849/live',
|
||||||
'skip': 'don\'t know if this live persists',
|
'skip': 'don\'t know if this live persists',
|
||||||
'info_dict': {
|
'info_dict': {
|
||||||
@@ -53,3 +53,33 @@ class MixchIE(InfoExtractor):
|
|||||||
}],
|
}],
|
||||||
'is_live': True,
|
'is_live': True,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class MixchArchiveIE(InfoExtractor):
|
||||||
|
IE_NAME = 'mixch:archive'
|
||||||
|
_VALID_URL = r'https?://(?:www\.)?mixch\.tv/archive/(?P<id>\d+)'
|
||||||
|
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://mixch.tv/archive/421',
|
||||||
|
'skip': 'paid video, no DRM. expires at Jan 23',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '421',
|
||||||
|
'title': '96NEKO SHOW TIME',
|
||||||
|
}
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
video_id = self._match_id(url)
|
||||||
|
webpage = self._download_webpage(url, video_id)
|
||||||
|
|
||||||
|
html5_videos = self._parse_html5_media_entries(
|
||||||
|
url, webpage.replace('video-js', 'video'), video_id, 'hls')
|
||||||
|
if not html5_videos:
|
||||||
|
self.raise_login_required(method='cookies')
|
||||||
|
infodict = html5_videos[0]
|
||||||
|
infodict.update({
|
||||||
|
'id': video_id,
|
||||||
|
'title': self._html_search_regex(r'class="archive-title">(.+?)</', webpage, 'title')
|
||||||
|
})
|
||||||
|
|
||||||
|
return infodict
|
||||||
|
|||||||
@@ -12,6 +12,7 @@ from ..compat import (
|
|||||||
compat_zip
|
compat_zip
|
||||||
)
|
)
|
||||||
from ..utils import (
|
from ..utils import (
|
||||||
|
ExtractorError,
|
||||||
int_or_none,
|
int_or_none,
|
||||||
parse_iso8601,
|
parse_iso8601,
|
||||||
strip_or_none,
|
strip_or_none,
|
||||||
@@ -125,7 +126,20 @@ class MixcloudIE(MixcloudBaseIE):
|
|||||||
tag {
|
tag {
|
||||||
name
|
name
|
||||||
}
|
}
|
||||||
}''', track_id, username, slug)
|
}
|
||||||
|
restrictedReason
|
||||||
|
id''', track_id, username, slug)
|
||||||
|
|
||||||
|
if not cloudcast:
|
||||||
|
raise ExtractorError('Track not found', expected=True)
|
||||||
|
|
||||||
|
reason = cloudcast.get('restrictedReason')
|
||||||
|
if reason == 'tracklist':
|
||||||
|
raise ExtractorError('Track unavailable in your country due to licensing restrictions', expected=True)
|
||||||
|
elif reason == 'repeat_play':
|
||||||
|
raise ExtractorError('You have reached your play limit for this track', expected=True)
|
||||||
|
elif reason:
|
||||||
|
raise ExtractorError('Track is restricted', expected=True)
|
||||||
|
|
||||||
title = cloudcast['name']
|
title = cloudcast['name']
|
||||||
|
|
||||||
|
|||||||
+13
-10
@@ -197,9 +197,12 @@ class NBCSportsVPlayerIE(InfoExtractor):
|
|||||||
'timestamp': 1426270238,
|
'timestamp': 1426270238,
|
||||||
'upload_date': '20150313',
|
'upload_date': '20150313',
|
||||||
'uploader': 'NBCU-SPORTS',
|
'uploader': 'NBCU-SPORTS',
|
||||||
|
'duration': 72.818,
|
||||||
|
'chapters': [],
|
||||||
|
'thumbnail': r're:^https?://.*\.jpg$'
|
||||||
}
|
}
|
||||||
}, {
|
}, {
|
||||||
'url': 'https://vplayer.nbcsports.com/p/BxmELC/nbcsports_embed/select/media/_hqLjQ95yx8Z',
|
'url': 'https://vplayer.nbcsports.com/p/BxmELC/nbcsports_embed/select/media/PEgOtlNcC_y2',
|
||||||
'only_matching': True,
|
'only_matching': True,
|
||||||
}, {
|
}, {
|
||||||
'url': 'https://www.nbcsports.com/vplayer/p/BxmELC/nbcsports/select/PHJSaFWbrTY9?form=html&autoPlay=true',
|
'url': 'https://www.nbcsports.com/vplayer/p/BxmELC/nbcsports/select/PHJSaFWbrTY9?form=html&autoPlay=true',
|
||||||
@@ -208,16 +211,15 @@ class NBCSportsVPlayerIE(InfoExtractor):
|
|||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def _extract_url(webpage):
|
def _extract_url(webpage):
|
||||||
iframe_m = re.search(
|
video_urls = re.search(
|
||||||
r'<(?:iframe[^>]+|div[^>]+data-(?:mpx-)?)src="(?P<url>%s[^"]+)"' % NBCSportsVPlayerIE._VALID_URL_BASE, webpage)
|
r'(?:iframe[^>]+|var video|div[^>]+data-(?:mpx-)?)[sS]rc\s?=\s?"(?P<url>%s[^\"]+)' % NBCSportsVPlayerIE._VALID_URL_BASE, webpage)
|
||||||
if iframe_m:
|
if video_urls:
|
||||||
return iframe_m.group('url')
|
return video_urls.group('url')
|
||||||
|
|
||||||
def _real_extract(self, url):
|
def _real_extract(self, url):
|
||||||
video_id = self._match_id(url)
|
video_id = self._match_id(url)
|
||||||
webpage = self._download_webpage(url, video_id)
|
webpage = self._download_webpage(url, video_id)
|
||||||
theplatform_url = self._og_search_video_url(webpage).replace(
|
theplatform_url = self._html_search_regex(r'tp:releaseUrl="(.+?)"', webpage, 'url')
|
||||||
'vplayer.nbcsports.com', 'player.theplatform.com')
|
|
||||||
return self.url_result(theplatform_url, 'ThePlatform')
|
return self.url_result(theplatform_url, 'ThePlatform')
|
||||||
|
|
||||||
|
|
||||||
@@ -235,6 +237,9 @@ class NBCSportsIE(InfoExtractor):
|
|||||||
'uploader': 'NBCU-SPORTS',
|
'uploader': 'NBCU-SPORTS',
|
||||||
'upload_date': '20150330',
|
'upload_date': '20150330',
|
||||||
'timestamp': 1427726529,
|
'timestamp': 1427726529,
|
||||||
|
'chapters': [],
|
||||||
|
'thumbnail': 'https://hdliveextra-a.akamaihd.net/HD/image_sports/NBCU_Sports_Group_-_nbcsports/253/303/izzodps.jpg',
|
||||||
|
'duration': 528.395,
|
||||||
}
|
}
|
||||||
}, {
|
}, {
|
||||||
# data-mpx-src
|
# data-mpx-src
|
||||||
@@ -403,9 +408,7 @@ class NBCNewsIE(ThePlatformIE):
|
|||||||
video_id = self._match_id(url)
|
video_id = self._match_id(url)
|
||||||
webpage = self._download_webpage(url, video_id)
|
webpage = self._download_webpage(url, video_id)
|
||||||
|
|
||||||
data = self._parse_json(self._search_regex(
|
data = self._search_nextjs_data(webpage, video_id)['props']['initialState']
|
||||||
r'<script[^>]+id="__NEXT_DATA__"[^>]*>({.+?})</script>',
|
|
||||||
webpage, 'bootstrap json'), video_id)['props']['initialState']
|
|
||||||
video_data = try_get(data, lambda x: x['video']['current'], dict)
|
video_data = try_get(data, lambda x: x['video']['current'], dict)
|
||||||
if not video_data:
|
if not video_data:
|
||||||
video_data = data['article']['content'][0]['primaryMedia']['video']
|
video_data = data['article']['content'][0]['primaryMedia']['video']
|
||||||
|
|||||||
@@ -0,0 +1,51 @@
|
|||||||
|
# coding: utf-8
|
||||||
|
from __future__ import unicode_literals
|
||||||
|
|
||||||
|
from .common import InfoExtractor
|
||||||
|
from ..utils import (
|
||||||
|
js_to_json,
|
||||||
|
merge_dicts,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class NewsyIE(InfoExtractor):
|
||||||
|
_VALID_URL = r'https?://(?:www\.)?newsy\.com/stories/(?P<id>[^/?#$&]+)'
|
||||||
|
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://www.newsy.com/stories/nft-trend-leads-to-fraudulent-art-auctions/',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '609d65125b086c24fb529312',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'NFT Art Auctions Have A Piracy Problem',
|
||||||
|
'description': 'md5:971e52ab8bc97e50305475cde8284c83',
|
||||||
|
'display_id': 'nft-trend-leads-to-fraudulent-art-auctions',
|
||||||
|
'timestamp': 1621339200,
|
||||||
|
'duration': 339630,
|
||||||
|
'thumbnail': 'https://cdn.newsy.com/images/videos/x/1620927824_xyrrP4.jpg',
|
||||||
|
'upload_date': '20210518'
|
||||||
|
},
|
||||||
|
'params': {'skip_download': True}
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
display_id = self._match_id(url)
|
||||||
|
webpage = self._download_webpage(url, display_id)
|
||||||
|
data_json = self._parse_json(self._html_search_regex(
|
||||||
|
r'data-video-player\s?=\s?"({[^"]+})">', webpage, 'data'), display_id, js_to_json)
|
||||||
|
ld_json = self._search_json_ld(webpage, display_id, fatal=False)
|
||||||
|
|
||||||
|
formats, subtitles = [], {}
|
||||||
|
if data_json.get('stream'):
|
||||||
|
fmts, subs = self._extract_m3u8_formats_and_subtitles(data_json['stream'], display_id)
|
||||||
|
formats.extend(fmts)
|
||||||
|
subtitles = self._merge_subtitles(subtitles, subs)
|
||||||
|
self._sort_formats(formats)
|
||||||
|
return merge_dicts(ld_json, {
|
||||||
|
'id': data_json['id'],
|
||||||
|
'display_id': display_id,
|
||||||
|
'title': data_json.get('headline'),
|
||||||
|
'duration': data_json.get('duration'),
|
||||||
|
'thumbnail': data_json.get('image'),
|
||||||
|
'formats': formats,
|
||||||
|
'subtitles': subtitles,
|
||||||
|
})
|
||||||
+117
-26
@@ -12,6 +12,8 @@ from ..utils import (
|
|||||||
ExtractorError,
|
ExtractorError,
|
||||||
int_or_none,
|
int_or_none,
|
||||||
parse_duration,
|
parse_duration,
|
||||||
|
srt_subtitles_timecode,
|
||||||
|
traverse_obj,
|
||||||
try_get,
|
try_get,
|
||||||
urlencode_postdata,
|
urlencode_postdata,
|
||||||
)
|
)
|
||||||
@@ -20,7 +22,7 @@ from ..utils import (
|
|||||||
class NexxIE(InfoExtractor):
|
class NexxIE(InfoExtractor):
|
||||||
_VALID_URL = r'''(?x)
|
_VALID_URL = r'''(?x)
|
||||||
(?:
|
(?:
|
||||||
https?://api\.nexx(?:\.cloud|cdn\.com)/v3/(?P<domain_id>\d+)/videos/byid/|
|
https?://api\.nexx(?:\.cloud|cdn\.com)/v3(?:\.\d)?/(?P<domain_id>\d+)/videos/byid/|
|
||||||
nexx:(?:(?P<domain_id_s>\d+):)?|
|
nexx:(?:(?P<domain_id_s>\d+):)?|
|
||||||
https?://arc\.nexx\.cloud/api/video/
|
https?://arc\.nexx\.cloud/api/video/
|
||||||
)
|
)
|
||||||
@@ -42,35 +44,37 @@ class NexxIE(InfoExtractor):
|
|||||||
'timestamp': 1384264416,
|
'timestamp': 1384264416,
|
||||||
'upload_date': '20131112',
|
'upload_date': '20131112',
|
||||||
},
|
},
|
||||||
|
'skip': 'Spiegel nexx CDNs are now disabled'
|
||||||
}, {
|
}, {
|
||||||
# episode
|
# episode with captions
|
||||||
'url': 'https://api.nexx.cloud/v3/741/videos/byid/247858',
|
'url': 'https://api.nexx.cloud/v3.1/741/videos/byid/1701834',
|
||||||
'info_dict': {
|
'info_dict': {
|
||||||
'id': '247858',
|
'id': '1701834',
|
||||||
'ext': 'mp4',
|
'ext': 'mp4',
|
||||||
'title': 'Return of the Golden Child (OV)',
|
'title': 'Mein Leben mit \'nem TikTok E-Boy 😤',
|
||||||
'description': 'md5:5d969537509a92b733de21bae249dc63',
|
'alt_title': 'Mein Leben mit \'nem TikTok E-Boy 😤',
|
||||||
'release_year': 2017,
|
'description': 'md5:f84f395a881fd143f952c892deab528d',
|
||||||
'thumbnail': r're:^https?://.*\.jpg$',
|
'thumbnail': r're:^https?://.*\.jpg$',
|
||||||
'duration': 1397,
|
'duration': 770,
|
||||||
'timestamp': 1495033267,
|
'timestamp': 1595600027,
|
||||||
'upload_date': '20170517',
|
'upload_date': '20200724',
|
||||||
'episode_number': 2,
|
'episode_number': 2,
|
||||||
'season_number': 2,
|
'season_number': 2,
|
||||||
|
'episode': 'Episode 2',
|
||||||
|
'season': 'Season 2',
|
||||||
},
|
},
|
||||||
'params': {
|
'params': {
|
||||||
'skip_download': True,
|
'skip_download': True,
|
||||||
},
|
},
|
||||||
'skip': 'HTTP Error 404: Not Found',
|
|
||||||
}, {
|
}, {
|
||||||
# does not work via arc
|
|
||||||
'url': 'nexx:741:1269984',
|
'url': 'nexx:741:1269984',
|
||||||
'md5': 'c714b5b238b2958dc8d5642addba6886',
|
'md5': 'd5f14e14b592501e51addd5abef95a7f',
|
||||||
'info_dict': {
|
'info_dict': {
|
||||||
'id': '1269984',
|
'id': '1269984',
|
||||||
'ext': 'mp4',
|
'ext': 'mp4',
|
||||||
'title': '1 TAG ohne KLO... wortwörtlich! 😑',
|
'title': '1 TAG ohne KLO... wortwörtlich! ?',
|
||||||
'alt_title': '1 TAG ohne KLO... wortwörtlich! 😑',
|
'alt_title': '1 TAG ohne KLO... wortwörtlich! ?',
|
||||||
|
'description': 'md5:2016393a31991a900946432ccdd09a6f',
|
||||||
'thumbnail': r're:^https?://.*\.jpg$',
|
'thumbnail': r're:^https?://.*\.jpg$',
|
||||||
'duration': 607,
|
'duration': 607,
|
||||||
'timestamp': 1518614955,
|
'timestamp': 1518614955,
|
||||||
@@ -91,6 +95,7 @@ class NexxIE(InfoExtractor):
|
|||||||
'timestamp': 1527874460,
|
'timestamp': 1527874460,
|
||||||
'upload_date': '20180601',
|
'upload_date': '20180601',
|
||||||
},
|
},
|
||||||
|
'skip': 'Spiegel nexx CDNs are now disabled'
|
||||||
}, {
|
}, {
|
||||||
'url': 'https://api.nexxcdn.com/v3/748/videos/byid/128907',
|
'url': 'https://api.nexxcdn.com/v3/748/videos/byid/128907',
|
||||||
'only_matching': True,
|
'only_matching': True,
|
||||||
@@ -138,6 +143,8 @@ class NexxIE(InfoExtractor):
|
|||||||
return NexxIE._extract_urls(webpage)[0]
|
return NexxIE._extract_urls(webpage)[0]
|
||||||
|
|
||||||
def _handle_error(self, response):
|
def _handle_error(self, response):
|
||||||
|
if traverse_obj(response, ('metadata', 'notice'), expected_type=str):
|
||||||
|
self.report_warning('%s said: %s' % (self.IE_NAME, response['metadata']['notice']))
|
||||||
status = int_or_none(try_get(
|
status = int_or_none(try_get(
|
||||||
response, lambda x: x['metadata']['status']) or 200)
|
response, lambda x: x['metadata']['status']) or 200)
|
||||||
if 200 <= status < 300:
|
if 200 <= status < 300:
|
||||||
@@ -220,6 +227,65 @@ class NexxIE(InfoExtractor):
|
|||||||
|
|
||||||
return formats
|
return formats
|
||||||
|
|
||||||
|
def _extract_3q_formats(self, video, video_id):
|
||||||
|
stream_data = video['streamdata']
|
||||||
|
cdn = stream_data['cdnType']
|
||||||
|
assert cdn == '3q'
|
||||||
|
|
||||||
|
q_acc, q_prefix, q_locator, q_hash = stream_data['qAccount'], stream_data['qPrefix'], stream_data['qLocator'], stream_data['qHash']
|
||||||
|
protection_key = traverse_obj(
|
||||||
|
video, ('protectiondata', 'key'), expected_type=str)
|
||||||
|
|
||||||
|
def get_cdn_shield_base(shield_type=''):
|
||||||
|
for secure in ('', 's'):
|
||||||
|
cdn_shield = stream_data.get('cdnShield%sHTTP%s' % (shield_type, secure.upper()))
|
||||||
|
if cdn_shield:
|
||||||
|
return 'http%s://%s' % (secure, cdn_shield)
|
||||||
|
return f'http://sdn-global-{"prog" if shield_type.lower() == "prog" else "streaming"}-cache.3qsdn.com/' + (f's/{protection_key}/' if protection_key else '')
|
||||||
|
|
||||||
|
stream_base = get_cdn_shield_base()
|
||||||
|
|
||||||
|
formats = []
|
||||||
|
formats.extend(self._extract_m3u8_formats(
|
||||||
|
f'{stream_base}{q_acc}/files/{q_prefix}/{q_locator}/{q_acc}-{stream_data.get("qHEVCHash") or q_hash}.ism/manifest.m3u8',
|
||||||
|
video_id, 'mp4', m3u8_id=f'{cdn}-hls', fatal=False))
|
||||||
|
formats.extend(self._extract_mpd_formats(
|
||||||
|
f'{stream_base}{q_acc}/files/{q_prefix}/{q_locator}/{q_acc}-{q_hash}.ism/manifest.mpd',
|
||||||
|
video_id, mpd_id=f'{cdn}-dash', fatal=False))
|
||||||
|
|
||||||
|
progressive_base = get_cdn_shield_base('Prog')
|
||||||
|
q_references = stream_data.get('qReferences') or ''
|
||||||
|
fds = q_references.split(',')
|
||||||
|
for fd in fds:
|
||||||
|
ss = fd.split(':')
|
||||||
|
if len(ss) != 3:
|
||||||
|
continue
|
||||||
|
tbr = int_or_none(ss[1], scale=1000)
|
||||||
|
formats.append({
|
||||||
|
'url': f'{progressive_base}{q_acc}/uploads/{q_acc}-{ss[2]}.webm',
|
||||||
|
'format_id': f'{cdn}-{ss[0]}{"-%s" % tbr if tbr else ""}',
|
||||||
|
'tbr': tbr,
|
||||||
|
})
|
||||||
|
|
||||||
|
azure_file_distribution = stream_data.get('azureFileDistribution') or ''
|
||||||
|
fds = azure_file_distribution.split(',')
|
||||||
|
for fd in fds:
|
||||||
|
ss = fd.split(':')
|
||||||
|
if len(ss) != 3:
|
||||||
|
continue
|
||||||
|
tbr = int_or_none(ss[0])
|
||||||
|
width, height = ss[1].split('x') if len(ss[1].split('x')) == 2 else (None, None)
|
||||||
|
f = {
|
||||||
|
'url': f'{progressive_base}{q_acc}/files/{q_prefix}/{q_locator}/{ss[2]}.mp4',
|
||||||
|
'format_id': f'{cdn}-http-{"-%s" % tbr if tbr else ""}',
|
||||||
|
'tbr': tbr,
|
||||||
|
'width': int_or_none(width),
|
||||||
|
'height': int_or_none(height),
|
||||||
|
}
|
||||||
|
formats.append(f)
|
||||||
|
|
||||||
|
return formats
|
||||||
|
|
||||||
def _extract_azure_formats(self, video, video_id):
|
def _extract_azure_formats(self, video, video_id):
|
||||||
stream_data = video['streamdata']
|
stream_data = video['streamdata']
|
||||||
cdn = stream_data['cdnType']
|
cdn = stream_data['cdnType']
|
||||||
@@ -345,10 +411,11 @@ class NexxIE(InfoExtractor):
|
|||||||
# md5( operation + domain_id + domain_secret )
|
# md5( operation + domain_id + domain_secret )
|
||||||
# where domain_secret is a static value that will be given by nexx.tv
|
# where domain_secret is a static value that will be given by nexx.tv
|
||||||
# as per [1]. Here is how this "secret" is generated (reversed
|
# as per [1]. Here is how this "secret" is generated (reversed
|
||||||
# from _play.api.init function, search for clienttoken). So it's
|
# from _play._factory.data.getDomainData function, search for
|
||||||
# actually not static and not that much of a secret.
|
# domaintoken or enableAPIAccess). So it's actually not static
|
||||||
|
# and not that much of a secret.
|
||||||
# 1. https://nexxtvstorage.blob.core.windows.net/files/201610/27.pdf
|
# 1. https://nexxtvstorage.blob.core.windows.net/files/201610/27.pdf
|
||||||
secret = result['device']['clienttoken'][int(device_id[0]):]
|
secret = result['device']['domaintoken'][int(device_id[0]):]
|
||||||
secret = secret[0:len(secret) - int(device_id[-1])]
|
secret = secret[0:len(secret) - int(device_id[-1])]
|
||||||
|
|
||||||
op = 'byid'
|
op = 'byid'
|
||||||
@@ -360,15 +427,18 @@ class NexxIE(InfoExtractor):
|
|||||||
|
|
||||||
result = self._call_api(
|
result = self._call_api(
|
||||||
domain_id, 'videos/%s/%s' % (op, video_id), video_id, data={
|
domain_id, 'videos/%s/%s' % (op, video_id), video_id, data={
|
||||||
'additionalfields': 'language,channel,actors,studio,licenseby,slug,subtitle,teaser,description',
|
'additionalfields': 'language,channel,format,licenseby,slug,fileversion,episode,season',
|
||||||
'addInteractionOptions': '1',
|
'addInteractionOptions': '1',
|
||||||
'addStatusDetails': '1',
|
'addStatusDetails': '1',
|
||||||
'addStreamDetails': '1',
|
'addStreamDetails': '1',
|
||||||
'addCaptions': '1',
|
'addFeatures': '1',
|
||||||
|
# Caption format selection doesn't seem to be enforced?
|
||||||
|
'addCaptions': 'vtt',
|
||||||
'addScenes': '1',
|
'addScenes': '1',
|
||||||
|
'addChapters': '1',
|
||||||
'addHotSpots': '1',
|
'addHotSpots': '1',
|
||||||
|
'addConnectedMedia': 'persons',
|
||||||
'addBumpers': '1',
|
'addBumpers': '1',
|
||||||
'captionFormat': 'data',
|
|
||||||
}, headers={
|
}, headers={
|
||||||
'X-Request-CID': cid,
|
'X-Request-CID': cid,
|
||||||
'X-Request-Token': request_token,
|
'X-Request-Token': request_token,
|
||||||
@@ -384,27 +454,48 @@ class NexxIE(InfoExtractor):
|
|||||||
formats = self._extract_azure_formats(video, video_id)
|
formats = self._extract_azure_formats(video, video_id)
|
||||||
elif cdn == 'free':
|
elif cdn == 'free':
|
||||||
formats = self._extract_free_formats(video, video_id)
|
formats = self._extract_free_formats(video, video_id)
|
||||||
|
elif cdn == '3q':
|
||||||
|
formats = self._extract_3q_formats(video, video_id)
|
||||||
else:
|
else:
|
||||||
self.raise_no_formats(f'{cdn} formats are currently not supported', video_id)
|
self.raise_no_formats(f'{cdn} formats are currently not supported', video_id)
|
||||||
|
|
||||||
self._sort_formats(formats)
|
self._sort_formats(formats)
|
||||||
|
|
||||||
|
subtitles = {}
|
||||||
|
for sub in video.get('captiondata') or []:
|
||||||
|
if sub.get('data'):
|
||||||
|
subtitles.setdefault(sub.get('language', 'en'), []).append({
|
||||||
|
'ext': 'srt',
|
||||||
|
'data': '\n\n'.join(
|
||||||
|
f'{i + 1}\n{srt_subtitles_timecode(line["fromms"] / 1000)} --> {srt_subtitles_timecode(line["toms"] / 1000)}\n{line["caption"]}'
|
||||||
|
for i, line in enumerate(sub['data'])),
|
||||||
|
'name': sub.get('language_long') or sub.get('title')
|
||||||
|
})
|
||||||
|
elif sub.get('url'):
|
||||||
|
subtitles.setdefault(sub.get('language', 'en'), []).append({
|
||||||
|
'url': sub['url'],
|
||||||
|
'ext': sub.get('format'),
|
||||||
|
'name': sub.get('language_long') or sub.get('title')
|
||||||
|
})
|
||||||
|
|
||||||
return {
|
return {
|
||||||
'id': video_id,
|
'id': video_id,
|
||||||
'title': title,
|
'title': title,
|
||||||
'alt_title': general.get('subtitle'),
|
'alt_title': general.get('subtitle'),
|
||||||
'description': general.get('description'),
|
'description': general.get('description'),
|
||||||
'release_year': int_or_none(general.get('year')),
|
'release_year': int_or_none(general.get('year')),
|
||||||
'creator': general.get('studio') or general.get('studio_adref'),
|
'creator': general.get('studio') or general.get('studio_adref') or None,
|
||||||
'thumbnail': try_get(
|
'thumbnail': try_get(
|
||||||
video, lambda x: x['imagedata']['thumb'], compat_str),
|
video, lambda x: x['imagedata']['thumb'], compat_str),
|
||||||
'duration': parse_duration(general.get('runtime')),
|
'duration': parse_duration(general.get('runtime')),
|
||||||
'timestamp': int_or_none(general.get('uploaded')),
|
'timestamp': int_or_none(general.get('uploaded')),
|
||||||
'episode_number': int_or_none(try_get(
|
'episode_number': traverse_obj(
|
||||||
video, lambda x: x['episodedata']['episode'])),
|
video, (('episodedata', 'general'), 'episode'), expected_type=int, get_all=False),
|
||||||
'season_number': int_or_none(try_get(
|
'season_number': traverse_obj(
|
||||||
video, lambda x: x['episodedata']['season'])),
|
video, (('episodedata', 'general'), 'season'), expected_type=int, get_all=False),
|
||||||
|
'cast': traverse_obj(video, ('connectedmedia', ..., 'title'), expected_type=str),
|
||||||
'formats': formats,
|
'formats': formats,
|
||||||
|
'subtitles': subtitles,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,67 @@
|
|||||||
|
# coding: utf-8
|
||||||
|
from __future__ import unicode_literals
|
||||||
|
|
||||||
|
from .common import InfoExtractor
|
||||||
|
from ..utils import (
|
||||||
|
parse_duration,
|
||||||
|
parse_count,
|
||||||
|
unified_strdate
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class NoodleMagazineIE(InfoExtractor):
|
||||||
|
_VALID_URL = r'https?://(?:www|adult\.)?noodlemagazine\.com/watch/(?P<id>[0-9-_]+)'
|
||||||
|
_TEST = {
|
||||||
|
'url': 'https://adult.noodlemagazine.com/watch/-67421364_456239604',
|
||||||
|
'md5': '9e02aa763612929d0b4b850591a9248b',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '-67421364_456239604',
|
||||||
|
'title': 'Aria alexander manojob',
|
||||||
|
'thumbnail': r're:^https://.*\.jpg',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'duration': 903,
|
||||||
|
'view_count': int,
|
||||||
|
'like_count': int,
|
||||||
|
'description': 'Aria alexander manojob',
|
||||||
|
'tags': ['aria', 'alexander', 'manojob'],
|
||||||
|
'upload_date': '20190218',
|
||||||
|
'age_limit': 18
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
video_id = self._match_id(url)
|
||||||
|
webpage = self._download_webpage(url, video_id)
|
||||||
|
title = self._og_search_title(webpage)
|
||||||
|
duration = parse_duration(self._html_search_meta('video:duration', webpage, 'duration', default=None))
|
||||||
|
description = self._og_search_property('description', webpage, default='').replace(' watch online hight quality video', '')
|
||||||
|
tags = self._html_search_meta('video:tag', webpage, default='').split(', ')
|
||||||
|
view_count = parse_count(self._html_search_meta('ya:ovs:views_total', webpage, default=None))
|
||||||
|
like_count = parse_count(self._html_search_meta('ya:ovs:likes', webpage, default=None))
|
||||||
|
upload_date = unified_strdate(self._html_search_meta('ya:ovs:upload_date', webpage, default=''))
|
||||||
|
|
||||||
|
key = self._html_search_regex(rf'/{video_id}\?(?:.*&)?m=([^&"\'\s,]+)', webpage, 'key')
|
||||||
|
playlist_info = self._download_json(f'https://adult.noodlemagazine.com/playlist/{video_id}?m={key}', video_id)
|
||||||
|
thumbnail = self._og_search_property('image', webpage, default=None) or playlist_info.get('image')
|
||||||
|
|
||||||
|
formats = [{
|
||||||
|
'url': source.get('file'),
|
||||||
|
'quality': source.get('label'),
|
||||||
|
'ext': source.get('type'),
|
||||||
|
} for source in playlist_info.get('sources')]
|
||||||
|
|
||||||
|
self._sort_formats(formats)
|
||||||
|
|
||||||
|
return {
|
||||||
|
'id': video_id,
|
||||||
|
'formats': formats,
|
||||||
|
'title': title,
|
||||||
|
'thumbnail': thumbnail,
|
||||||
|
'duration': duration,
|
||||||
|
'description': description,
|
||||||
|
'tags': tags,
|
||||||
|
'view_count': view_count,
|
||||||
|
'like_count': like_count,
|
||||||
|
'upload_date': upload_date,
|
||||||
|
'age_limit': 18
|
||||||
|
}
|
||||||
@@ -41,9 +41,7 @@ class NovaPlayIE(InfoExtractor):
|
|||||||
def _real_extract(self, url):
|
def _real_extract(self, url):
|
||||||
video_id = self._match_id(url)
|
video_id = self._match_id(url)
|
||||||
webpage = self._download_webpage(url, video_id)
|
webpage = self._download_webpage(url, video_id)
|
||||||
video_props = self._parse_json(self._search_regex(
|
video_props = self._search_nextjs_data(webpage, video_id)['props']['pageProps']['video']
|
||||||
r'<script\s?id=\"__NEXT_DATA__\"\s?type=\"application/json\">({.+})</script>',
|
|
||||||
webpage, 'video_props'), video_id)['props']['pageProps']['video']
|
|
||||||
m3u8_url = self._download_json(
|
m3u8_url = self._download_json(
|
||||||
f'https://nbg-api.fite.tv/api/v2/videos/{video_id}/streams',
|
f'https://nbg-api.fite.tv/api/v2/videos/{video_id}/streams',
|
||||||
video_id, headers={'x-flipps-user-agent': 'Flipps/75/9.7'})[0]['url']
|
video_id, headers={'x-flipps-user-agent': 'Flipps/75/9.7'})[0]['url']
|
||||||
|
|||||||
+66
-38
@@ -4,51 +4,41 @@ from __future__ import unicode_literals
|
|||||||
from .common import InfoExtractor
|
from .common import InfoExtractor
|
||||||
from ..utils import (
|
from ..utils import (
|
||||||
ExtractorError,
|
ExtractorError,
|
||||||
|
int_or_none,
|
||||||
traverse_obj,
|
traverse_obj,
|
||||||
try_get,
|
unified_strdate,
|
||||||
unified_strdate
|
unified_timestamp
|
||||||
)
|
)
|
||||||
from ..compat import compat_str
|
from ..compat import compat_str
|
||||||
|
|
||||||
|
|
||||||
class OpenRecIE(InfoExtractor):
|
class OpenRecBaseIE(InfoExtractor):
|
||||||
IE_NAME = 'openrec'
|
def _extract_pagestore(self, webpage, video_id):
|
||||||
_VALID_URL = r'https?://(?:www\.)?openrec\.tv/live/(?P<id>[^/]+)'
|
return self._parse_json(
|
||||||
_TESTS = [{
|
|
||||||
'url': 'https://www.openrec.tv/live/2p8v31qe4zy',
|
|
||||||
'only_matching': True,
|
|
||||||
}, {
|
|
||||||
'url': 'https://www.openrec.tv/live/wez93eqvjzl',
|
|
||||||
'only_matching': True,
|
|
||||||
}]
|
|
||||||
|
|
||||||
def _real_extract(self, url):
|
|
||||||
video_id = self._match_id(url)
|
|
||||||
webpage = self._download_webpage('https://www.openrec.tv/live/%s' % video_id, video_id)
|
|
||||||
|
|
||||||
window_stores = self._parse_json(
|
|
||||||
self._search_regex(r'(?m)window\.pageStore\s*=\s*(\{.+?\});$', webpage, 'window.pageStore'), video_id)
|
self._search_regex(r'(?m)window\.pageStore\s*=\s*(\{.+?\});$', webpage, 'window.pageStore'), video_id)
|
||||||
|
|
||||||
|
def _extract_movie(self, webpage, video_id, name, is_live):
|
||||||
|
window_stores = self._extract_pagestore(webpage, video_id)
|
||||||
movie_store = traverse_obj(
|
movie_store = traverse_obj(
|
||||||
window_stores,
|
window_stores,
|
||||||
('v8', 'state', 'movie'),
|
('v8', 'state', 'movie'),
|
||||||
('v8', 'movie'),
|
('v8', 'movie'),
|
||||||
expected_type=dict)
|
expected_type=dict)
|
||||||
if not movie_store:
|
if not movie_store:
|
||||||
raise ExtractorError('Failed to extract live info')
|
raise ExtractorError(f'Failed to extract {name} info')
|
||||||
|
|
||||||
title = movie_store.get('title')
|
title = movie_store.get('title')
|
||||||
description = movie_store.get('introduction')
|
description = movie_store.get('introduction')
|
||||||
thumbnail = movie_store.get('thumbnailUrl')
|
thumbnail = movie_store.get('thumbnailUrl')
|
||||||
|
|
||||||
channel_user = movie_store.get('channel', {}).get('user')
|
uploader = traverse_obj(movie_store, ('channel', 'user', 'name'), expected_type=compat_str)
|
||||||
uploader = try_get(channel_user, lambda x: x['name'], compat_str)
|
uploader_id = traverse_obj(movie_store, ('channel', 'user', 'id'), expected_type=compat_str)
|
||||||
uploader_id = try_get(channel_user, lambda x: x['id'], compat_str)
|
|
||||||
|
|
||||||
timestamp = traverse_obj(movie_store, ('startedAt', 'time'), expected_type=int)
|
timestamp = int_or_none(traverse_obj(movie_store, ('publishedAt', 'time')), scale=1000)
|
||||||
|
|
||||||
m3u8_playlists = movie_store.get('media')
|
m3u8_playlists = movie_store.get('media') or {}
|
||||||
formats = []
|
formats = []
|
||||||
for (name, m3u8_url) in m3u8_playlists.items():
|
for name, m3u8_url in m3u8_playlists.items():
|
||||||
if not m3u8_url:
|
if not m3u8_url:
|
||||||
continue
|
continue
|
||||||
formats.extend(self._extract_m3u8_formats(
|
formats.extend(self._extract_m3u8_formats(
|
||||||
@@ -66,11 +56,29 @@ class OpenRecIE(InfoExtractor):
|
|||||||
'uploader': uploader,
|
'uploader': uploader,
|
||||||
'uploader_id': uploader_id,
|
'uploader_id': uploader_id,
|
||||||
'timestamp': timestamp,
|
'timestamp': timestamp,
|
||||||
'is_live': True,
|
'is_live': is_live,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
class OpenRecCaptureIE(InfoExtractor):
|
class OpenRecIE(OpenRecBaseIE):
|
||||||
|
IE_NAME = 'openrec'
|
||||||
|
_VALID_URL = r'https?://(?:www\.)?openrec\.tv/live/(?P<id>[^/]+)'
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://www.openrec.tv/live/2p8v31qe4zy',
|
||||||
|
'only_matching': True,
|
||||||
|
}, {
|
||||||
|
'url': 'https://www.openrec.tv/live/wez93eqvjzl',
|
||||||
|
'only_matching': True,
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
video_id = self._match_id(url)
|
||||||
|
webpage = self._download_webpage('https://www.openrec.tv/live/%s' % video_id, video_id)
|
||||||
|
|
||||||
|
return self._extract_movie(webpage, video_id, 'live', True)
|
||||||
|
|
||||||
|
|
||||||
|
class OpenRecCaptureIE(OpenRecBaseIE):
|
||||||
IE_NAME = 'openrec:capture'
|
IE_NAME = 'openrec:capture'
|
||||||
_VALID_URL = r'https?://(?:www\.)?openrec\.tv/capture/(?P<id>[^/]+)'
|
_VALID_URL = r'https?://(?:www\.)?openrec\.tv/capture/(?P<id>[^/]+)'
|
||||||
_TESTS = [{
|
_TESTS = [{
|
||||||
@@ -91,8 +99,7 @@ class OpenRecCaptureIE(InfoExtractor):
|
|||||||
video_id = self._match_id(url)
|
video_id = self._match_id(url)
|
||||||
webpage = self._download_webpage('https://www.openrec.tv/capture/%s' % video_id, video_id)
|
webpage = self._download_webpage('https://www.openrec.tv/capture/%s' % video_id, video_id)
|
||||||
|
|
||||||
window_stores = self._parse_json(
|
window_stores = self._extract_pagestore(webpage, video_id)
|
||||||
self._search_regex(r'(?m)window\.pageStore\s*=\s*(\{.+?\});$', webpage, 'window.pageStore'), video_id)
|
|
||||||
movie_store = window_stores.get('movie')
|
movie_store = window_stores.get('movie')
|
||||||
|
|
||||||
capture_data = window_stores.get('capture')
|
capture_data = window_stores.get('capture')
|
||||||
@@ -102,17 +109,14 @@ class OpenRecCaptureIE(InfoExtractor):
|
|||||||
thumbnail = capture_data.get('thumbnailUrl')
|
thumbnail = capture_data.get('thumbnailUrl')
|
||||||
upload_date = unified_strdate(capture_data.get('createdAt'))
|
upload_date = unified_strdate(capture_data.get('createdAt'))
|
||||||
|
|
||||||
channel_info = movie_store.get('channel') or {}
|
uploader = traverse_obj(movie_store, ('channel', 'name'), expected_type=compat_str)
|
||||||
uploader = channel_info.get('name')
|
uploader_id = traverse_obj(movie_store, ('channel', 'id'), expected_type=compat_str)
|
||||||
uploader_id = channel_info.get('id')
|
|
||||||
|
timestamp = traverse_obj(movie_store, 'createdAt', expected_type=compat_str)
|
||||||
|
timestamp = unified_timestamp(timestamp)
|
||||||
|
|
||||||
m3u8_url = capture_data.get('source')
|
|
||||||
if not m3u8_url:
|
|
||||||
raise ExtractorError('Cannot extract m3u8 url')
|
|
||||||
formats = self._extract_m3u8_formats(
|
formats = self._extract_m3u8_formats(
|
||||||
m3u8_url, video_id, ext='mp4', entry_protocol='m3u8_native',
|
capture_data.get('source'), video_id, ext='mp4')
|
||||||
m3u8_id='hls')
|
|
||||||
|
|
||||||
self._sort_formats(formats)
|
self._sort_formats(formats)
|
||||||
|
|
||||||
return {
|
return {
|
||||||
@@ -120,7 +124,31 @@ class OpenRecCaptureIE(InfoExtractor):
|
|||||||
'title': title,
|
'title': title,
|
||||||
'thumbnail': thumbnail,
|
'thumbnail': thumbnail,
|
||||||
'formats': formats,
|
'formats': formats,
|
||||||
|
'timestamp': timestamp,
|
||||||
'uploader': uploader,
|
'uploader': uploader,
|
||||||
'uploader_id': uploader_id,
|
'uploader_id': uploader_id,
|
||||||
'upload_date': upload_date,
|
'upload_date': upload_date,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class OpenRecMovieIE(OpenRecBaseIE):
|
||||||
|
IE_NAME = 'openrec:movie'
|
||||||
|
_VALID_URL = r'https?://(?:www\.)?openrec\.tv/movie/(?P<id>[^/]+)'
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://www.openrec.tv/movie/nqz5xl5km8v',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'nqz5xl5km8v',
|
||||||
|
'title': '限定コミュニティ(Discord)参加方法ご説明動画',
|
||||||
|
'description': 'md5:ebd563e5f5b060cda2f02bf26b14d87f',
|
||||||
|
'thumbnail': r're:https://.+',
|
||||||
|
'uploader': 'タイキとカズヒロ',
|
||||||
|
'uploader_id': 'taiki_to_kazuhiro',
|
||||||
|
'timestamp': 1638856800,
|
||||||
|
},
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
video_id = self._match_id(url)
|
||||||
|
webpage = self._download_webpage('https://www.openrec.tv/movie/%s' % video_id, video_id)
|
||||||
|
|
||||||
|
return self._extract_movie(webpage, video_id, 'movie', False)
|
||||||
|
|||||||
@@ -545,7 +545,7 @@ class PBSIE(InfoExtractor):
|
|||||||
for vid_id in video_id]
|
for vid_id in video_id]
|
||||||
return self.playlist_result(entries, display_id)
|
return self.playlist_result(entries, display_id)
|
||||||
|
|
||||||
info = None
|
info = {}
|
||||||
redirects = []
|
redirects = []
|
||||||
redirect_urls = set()
|
redirect_urls = set()
|
||||||
|
|
||||||
@@ -660,6 +660,9 @@ class PBSIE(InfoExtractor):
|
|||||||
'protocol': 'http',
|
'protocol': 'http',
|
||||||
})
|
})
|
||||||
formats.append(f)
|
formats.append(f)
|
||||||
|
for f in formats:
|
||||||
|
if (f.get('format_note') or '').endswith(' AD'): # Audio description
|
||||||
|
f['language_preference'] = -10
|
||||||
self._sort_formats(formats)
|
self._sort_formats(formats)
|
||||||
|
|
||||||
rating_str = info.get('rating')
|
rating_str = info.get('rating')
|
||||||
|
|||||||
@@ -7,6 +7,7 @@ import re
|
|||||||
from .common import InfoExtractor
|
from .common import InfoExtractor
|
||||||
from ..compat import compat_str
|
from ..compat import compat_str
|
||||||
from ..utils import (
|
from ..utils import (
|
||||||
|
format_field,
|
||||||
int_or_none,
|
int_or_none,
|
||||||
parse_resolution,
|
parse_resolution,
|
||||||
str_or_none,
|
str_or_none,
|
||||||
@@ -1386,8 +1387,7 @@ class PeerTubePlaylistIE(InfoExtractor):
|
|||||||
playlist_timestamp = unified_timestamp(info.get('createdAt'))
|
playlist_timestamp = unified_timestamp(info.get('createdAt'))
|
||||||
channel = try_get(info, lambda x: x['ownerAccount']['name']) or info.get('displayName')
|
channel = try_get(info, lambda x: x['ownerAccount']['name']) or info.get('displayName')
|
||||||
channel_id = try_get(info, lambda x: x['ownerAccount']['id']) or info.get('id')
|
channel_id = try_get(info, lambda x: x['ownerAccount']['id']) or info.get('id')
|
||||||
thumbnail = info.get('thumbnailPath')
|
thumbnail = format_field(info, 'thumbnailPath', f'https://{host}%s')
|
||||||
thumbnail = f'https://{host}{thumbnail}' if thumbnail else None
|
|
||||||
|
|
||||||
entries = OnDemandPagedList(functools.partial(
|
entries = OnDemandPagedList(functools.partial(
|
||||||
self.fetch_page, host, id, type), self._PAGE_SIZE)
|
self.fetch_page, host, id, type), self._PAGE_SIZE)
|
||||||
|
|||||||
@@ -0,0 +1,122 @@
|
|||||||
|
# coding: utf-8
|
||||||
|
from __future__ import unicode_literals
|
||||||
|
|
||||||
|
from .common import InfoExtractor
|
||||||
|
from ..utils import (
|
||||||
|
ExtractorError,
|
||||||
|
traverse_obj,
|
||||||
|
unified_timestamp,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class PixivSketchBaseIE(InfoExtractor):
|
||||||
|
def _call_api(self, video_id, path, referer, note='Downloading JSON metadata'):
|
||||||
|
response = self._download_json(f'https://sketch.pixiv.net/api/{path}', video_id, note=note, headers={
|
||||||
|
'Referer': referer,
|
||||||
|
'X-Requested-With': referer,
|
||||||
|
})
|
||||||
|
errors = traverse_obj(response, ('errors', ..., 'message'))
|
||||||
|
if errors:
|
||||||
|
raise ExtractorError(' '.join(f'{e}.' for e in errors))
|
||||||
|
return response.get('data') or {}
|
||||||
|
|
||||||
|
|
||||||
|
class PixivSketchIE(PixivSketchBaseIE):
|
||||||
|
IE_NAME = 'pixiv:sketch'
|
||||||
|
_VALID_URL = r'https?://sketch\.pixiv\.net/@(?P<uploader_id>[a-zA-Z0-9_-]+)/lives/(?P<id>\d+)/?'
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://sketch.pixiv.net/@nuhutya/lives/3654620468641830507',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '7370666691623196569',
|
||||||
|
'title': 'まにあえクリスマス!',
|
||||||
|
'uploader': 'ぬふちゃ',
|
||||||
|
'uploader_id': 'nuhutya',
|
||||||
|
'channel_id': '9844815',
|
||||||
|
'age_limit': 0,
|
||||||
|
'timestamp': 1640351536,
|
||||||
|
},
|
||||||
|
'skip': True,
|
||||||
|
}, {
|
||||||
|
# these two (age_limit > 0) requires you to login on website, but it's actually not required for download
|
||||||
|
'url': 'https://sketch.pixiv.net/@namahyou/lives/4393103321546851377',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '4907995960957946943',
|
||||||
|
'title': 'クリスマスなんて知らん🖕',
|
||||||
|
'uploader': 'すゃもり',
|
||||||
|
'uploader_id': 'suya2mori2',
|
||||||
|
'channel_id': '31169300',
|
||||||
|
'age_limit': 15,
|
||||||
|
'timestamp': 1640347640,
|
||||||
|
},
|
||||||
|
'skip': True,
|
||||||
|
}, {
|
||||||
|
'url': 'https://sketch.pixiv.net/@8aki/lives/3553803162487249670',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '1593420639479156945',
|
||||||
|
'title': 'おまけ本作業(リョナ有)',
|
||||||
|
'uploader': 'おぶい / Obui',
|
||||||
|
'uploader_id': 'oving',
|
||||||
|
'channel_id': '17606',
|
||||||
|
'age_limit': 18,
|
||||||
|
'timestamp': 1640330263,
|
||||||
|
},
|
||||||
|
'skip': True,
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
video_id, uploader_id = self._match_valid_url(url).group('id', 'uploader_id')
|
||||||
|
data = self._call_api(video_id, f'lives/{video_id}.json', url)
|
||||||
|
|
||||||
|
if not traverse_obj(data, 'is_broadcasting'):
|
||||||
|
raise ExtractorError(f'This live is offline. Use https://sketch.pixiv.net/@{uploader_id} for ongoing live.', expected=True)
|
||||||
|
|
||||||
|
m3u8_url = traverse_obj(data, ('owner', 'hls_movie', 'url'))
|
||||||
|
formats = self._extract_m3u8_formats(
|
||||||
|
m3u8_url, video_id, ext='mp4',
|
||||||
|
entry_protocol='m3u8_native', m3u8_id='hls')
|
||||||
|
self._sort_formats(formats)
|
||||||
|
|
||||||
|
return {
|
||||||
|
'id': video_id,
|
||||||
|
'title': data.get('name'),
|
||||||
|
'formats': formats,
|
||||||
|
'uploader': traverse_obj(data, ('user', 'name'), ('owner', 'user', 'name')),
|
||||||
|
'uploader_id': traverse_obj(data, ('user', 'unique_name'), ('owner', 'user', 'unique_name')),
|
||||||
|
'channel_id': str(traverse_obj(data, ('user', 'pixiv_user_id'), ('owner', 'user', 'pixiv_user_id'))),
|
||||||
|
'age_limit': 18 if data.get('is_r18') else 15 if data.get('is_r15') else 0,
|
||||||
|
'timestamp': unified_timestamp(data.get('created_at')),
|
||||||
|
'is_live': True
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class PixivSketchUserIE(PixivSketchBaseIE):
|
||||||
|
IE_NAME = 'pixiv:sketch:user'
|
||||||
|
_VALID_URL = r'https?://sketch\.pixiv\.net/@(?P<id>[a-zA-Z0-9_-]+)/?'
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://sketch.pixiv.net/@nuhutya',
|
||||||
|
'only_matching': True,
|
||||||
|
}, {
|
||||||
|
'url': 'https://sketch.pixiv.net/@namahyou',
|
||||||
|
'only_matching': True,
|
||||||
|
}, {
|
||||||
|
'url': 'https://sketch.pixiv.net/@8aki',
|
||||||
|
'only_matching': True,
|
||||||
|
}]
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def suitable(cls, url):
|
||||||
|
return super(PixivSketchUserIE, cls).suitable(url) and not PixivSketchIE.suitable(url)
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
user_id = self._match_id(url)
|
||||||
|
data = self._call_api(user_id, f'lives/users/@{user_id}.json', url)
|
||||||
|
|
||||||
|
if not traverse_obj(data, 'is_broadcasting'):
|
||||||
|
try:
|
||||||
|
self._call_api(user_id, 'users/current.json', url, 'Investigating reason for request failure')
|
||||||
|
except ExtractorError as ex:
|
||||||
|
if ex.cause and ex.cause.code == 401:
|
||||||
|
self.raise_login_required(f'Please log in, or use direct link like https://sketch.pixiv.net/@{user_id}/1234567890', method='cookies')
|
||||||
|
raise ExtractorError('This user is offline', expected=True)
|
||||||
|
|
||||||
|
return self.url_result(f'https://sketch.pixiv.net/@{user_id}/lives/{data["id"]}')
|
||||||
@@ -0,0 +1,111 @@
|
|||||||
|
# coding: utf-8
|
||||||
|
from __future__ import unicode_literals
|
||||||
|
|
||||||
|
import base64
|
||||||
|
|
||||||
|
from .common import InfoExtractor
|
||||||
|
from ..utils import (
|
||||||
|
ExtractorError,
|
||||||
|
try_get,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class PokerGoBaseIE(InfoExtractor):
|
||||||
|
_NETRC_MACHINE = 'pokergo'
|
||||||
|
_AUTH_TOKEN = None
|
||||||
|
_PROPERTY_ID = '1dfb3940-7d53-4980-b0b0-f28b369a000d'
|
||||||
|
|
||||||
|
def _login(self):
|
||||||
|
username, password = self._get_login_info()
|
||||||
|
if not username:
|
||||||
|
self.raise_login_required(method='password')
|
||||||
|
|
||||||
|
self.report_login()
|
||||||
|
PokerGoBaseIE._AUTH_TOKEN = self._download_json(
|
||||||
|
f'https://subscription.pokergo.com/properties/{self._PROPERTY_ID}/sign-in', None,
|
||||||
|
headers={'authorization': f'Basic {base64.b64encode(f"{username}:{password}".encode()).decode()}'},
|
||||||
|
data=b'')['meta']['token']
|
||||||
|
if not self._AUTH_TOKEN:
|
||||||
|
raise ExtractorError('Unable to get Auth Token.', expected=True)
|
||||||
|
|
||||||
|
def _real_initialize(self):
|
||||||
|
if not self._AUTH_TOKEN:
|
||||||
|
self._login()
|
||||||
|
|
||||||
|
|
||||||
|
class PokerGoIE(PokerGoBaseIE):
|
||||||
|
_VALID_URL = r'https?://(?:www\.)?pokergo\.com/videos/(?P<id>[^&$#/?]+)'
|
||||||
|
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://www.pokergo.com/videos/2a70ec4e-4a80-414b-97ec-725d9b72a7dc',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'aVLOxDzY',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'Poker After Dark | Season 12 (2020) | Cry Me a River | Episode 2',
|
||||||
|
'description': 'md5:c7a8c29556cbfb6eb3c0d5d622251b71',
|
||||||
|
'thumbnail': 'https://cdn.jwplayer.com/v2/media/aVLOxDzY/poster.jpg?width=720',
|
||||||
|
'timestamp': 1608085715,
|
||||||
|
'duration': 2700.12,
|
||||||
|
'season_number': 12,
|
||||||
|
'episode_number': 2,
|
||||||
|
'series': 'poker after dark',
|
||||||
|
'upload_date': '20201216',
|
||||||
|
'season': 'Season 12',
|
||||||
|
'episode': 'Episode 2',
|
||||||
|
'display_id': '2a70ec4e-4a80-414b-97ec-725d9b72a7dc',
|
||||||
|
},
|
||||||
|
'params': {'skip_download': True}
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
id = self._match_id(url)
|
||||||
|
data_json = self._download_json(f'https://api.pokergo.com/v2/properties/{self._PROPERTY_ID}/videos/{id}', id,
|
||||||
|
headers={'authorization': f'Bearer {self._AUTH_TOKEN}'})['data']
|
||||||
|
v_id = data_json['source']
|
||||||
|
|
||||||
|
thumbnails = [{
|
||||||
|
'url': image['url'],
|
||||||
|
'id': image.get('label'),
|
||||||
|
'width': image.get('width'),
|
||||||
|
'height': image.get('height')
|
||||||
|
} for image in data_json.get('images') or [] if image.get('url')]
|
||||||
|
series_json = next(dct for dct in data_json.get('show_tags') or [] if dct.get('video_id') == id) or {}
|
||||||
|
|
||||||
|
return {
|
||||||
|
'_type': 'url_transparent',
|
||||||
|
'display_id': id,
|
||||||
|
'title': data_json.get('title'),
|
||||||
|
'description': data_json.get('description'),
|
||||||
|
'duration': data_json.get('duration'),
|
||||||
|
'thumbnails': thumbnails,
|
||||||
|
'season_number': series_json.get('season'),
|
||||||
|
'episode_number': series_json.get('episode_number'),
|
||||||
|
'series': try_get(series_json, lambda x: x['tag']['name']),
|
||||||
|
'url': f'https://cdn.jwplayer.com/v2/media/{v_id}'
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class PokerGoCollectionIE(PokerGoBaseIE):
|
||||||
|
_VALID_URL = r'https?://(?:www\.)?pokergo\.com/collections/(?P<id>[^&$#/?]+)'
|
||||||
|
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://www.pokergo.com/collections/19ffe481-5dae-481a-8869-75cc0e3c4700',
|
||||||
|
'playlist_mincount': 13,
|
||||||
|
'info_dict': {
|
||||||
|
'id': '19ffe481-5dae-481a-8869-75cc0e3c4700',
|
||||||
|
},
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _entries(self, id):
|
||||||
|
data_json = self._download_json(f'https://api.pokergo.com/v2/properties/{self._PROPERTY_ID}/collections/{id}?include=entities',
|
||||||
|
id, headers={'authorization': f'Bearer {self._AUTH_TOKEN}'})['data']
|
||||||
|
for video in data_json.get('collection_video') or []:
|
||||||
|
video_id = video.get('id')
|
||||||
|
if video_id:
|
||||||
|
yield self.url_result(
|
||||||
|
f'https://www.pokergo.com/videos/{video_id}',
|
||||||
|
ie=PokerGoIE.ie_key(), video_id=video_id)
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
id = self._match_id(url)
|
||||||
|
return self.playlist_result(self._entries(id), playlist_id=id)
|
||||||
@@ -0,0 +1,43 @@
|
|||||||
|
# coding: utf-8
|
||||||
|
from __future__ import unicode_literals
|
||||||
|
from .common import InfoExtractor
|
||||||
|
from ..utils import int_or_none
|
||||||
|
|
||||||
|
|
||||||
|
class PornezIE(InfoExtractor):
|
||||||
|
_VALID_URL = r'https?://(?:www\.)?pornez\.net/video(?P<id>[0-9]+)/'
|
||||||
|
_TEST = {
|
||||||
|
'url': 'https://pornez.net/video344819/mistresst-funny_penis_names-wmv/',
|
||||||
|
'md5': '2e19a0a1cff3a5dbea0ef1b9e80bcbbc',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '344819',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': r'mistresst funny_penis_names wmv',
|
||||||
|
'thumbnail': r're:^https?://.*\.jpg$',
|
||||||
|
'age_limit': 18,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
video_id = self._match_id(url)
|
||||||
|
webpage = self._download_webpage(url, video_id)
|
||||||
|
iframe_src = self._html_search_regex(
|
||||||
|
r'<iframe[^>]+src="(https?://pornez\.net/player/\?[^"]+)"', webpage, 'iframe', fatal=True)
|
||||||
|
title = self._html_search_meta(['name', 'twitter:title', 'og:title'], webpage, 'title', default=None)
|
||||||
|
if title is None:
|
||||||
|
title = self._search_regex(r'<h1>(.*?)</h1>', webpage, 'title', fatal=True)
|
||||||
|
thumbnail = self._html_search_meta(['thumbnailUrl'], webpage, 'title', default=None)
|
||||||
|
webpage = self._download_webpage(iframe_src, video_id)
|
||||||
|
entries = self._parse_html5_media_entries(iframe_src, webpage, video_id)[0]
|
||||||
|
for format in entries['formats']:
|
||||||
|
height = self._search_regex(r'_(\d+)\.m3u8', format['url'], 'height')
|
||||||
|
format['format_id'] = '%sp' % height
|
||||||
|
format['height'] = int_or_none(height)
|
||||||
|
|
||||||
|
entries.update({
|
||||||
|
'id': video_id,
|
||||||
|
'title': title,
|
||||||
|
'thumbnail': thumbnail,
|
||||||
|
'age_limit': 18
|
||||||
|
})
|
||||||
|
return entries
|
||||||
@@ -18,6 +18,7 @@ from ..utils import (
|
|||||||
clean_html,
|
clean_html,
|
||||||
determine_ext,
|
determine_ext,
|
||||||
ExtractorError,
|
ExtractorError,
|
||||||
|
format_field,
|
||||||
int_or_none,
|
int_or_none,
|
||||||
merge_dicts,
|
merge_dicts,
|
||||||
NO_DEFAULT,
|
NO_DEFAULT,
|
||||||
@@ -32,7 +33,7 @@ from ..utils import (
|
|||||||
|
|
||||||
class PornHubBaseIE(InfoExtractor):
|
class PornHubBaseIE(InfoExtractor):
|
||||||
_NETRC_MACHINE = 'pornhub'
|
_NETRC_MACHINE = 'pornhub'
|
||||||
_PORNHUB_HOST_RE = r'(?:(?P<host>pornhub(?:premium)?\.(?:com|net|org))|pornhubthbh7ap3u\.onion)'
|
_PORNHUB_HOST_RE = r'(?:(?P<host>pornhub(?:premium)?\.(?:com|net|org))|pornhubvybmsymdol4iibwgwtkpwmeyd6luq2gxajgjzfjvotyt5zhyd\.onion)'
|
||||||
|
|
||||||
def _download_webpage_handle(self, *args, **kwargs):
|
def _download_webpage_handle(self, *args, **kwargs):
|
||||||
def dl(*args, **kwargs):
|
def dl(*args, **kwargs):
|
||||||
@@ -247,7 +248,7 @@ class PornHubIE(PornHubBaseIE):
|
|||||||
'url': 'https://www.pornhub.com/view_video.php?viewkey=ph5a9813bfa7156',
|
'url': 'https://www.pornhub.com/view_video.php?viewkey=ph5a9813bfa7156',
|
||||||
'only_matching': True,
|
'only_matching': True,
|
||||||
}, {
|
}, {
|
||||||
'url': 'http://pornhubthbh7ap3u.onion/view_video.php?viewkey=ph5a9813bfa7156',
|
'url': 'http://pornhubvybmsymdol4iibwgwtkpwmeyd6luq2gxajgjzfjvotyt5zhyd.onion/view_video.php?viewkey=ph5a9813bfa7156',
|
||||||
'only_matching': True,
|
'only_matching': True,
|
||||||
}]
|
}]
|
||||||
|
|
||||||
@@ -431,7 +432,7 @@ class PornHubIE(PornHubBaseIE):
|
|||||||
default=None))
|
default=None))
|
||||||
formats.append({
|
formats.append({
|
||||||
'url': format_url,
|
'url': format_url,
|
||||||
'format_id': '%dp' % height if height else None,
|
'format_id': format_field(height, template='%dp'),
|
||||||
'height': height,
|
'height': height,
|
||||||
})
|
})
|
||||||
|
|
||||||
@@ -561,7 +562,7 @@ class PornHubUserIE(PornHubPlaylistBaseIE):
|
|||||||
'url': 'https://www.pornhubpremium.com/pornstar/lily-labeau',
|
'url': 'https://www.pornhubpremium.com/pornstar/lily-labeau',
|
||||||
'only_matching': True,
|
'only_matching': True,
|
||||||
}, {
|
}, {
|
||||||
'url': 'https://pornhubthbh7ap3u.onion/model/zoe_ph',
|
'url': 'https://pornhubvybmsymdol4iibwgwtkpwmeyd6luq2gxajgjzfjvotyt5zhyd.onion/model/zoe_ph',
|
||||||
'only_matching': True,
|
'only_matching': True,
|
||||||
}]
|
}]
|
||||||
|
|
||||||
@@ -732,7 +733,7 @@ class PornHubPagedVideoListIE(PornHubPagedPlaylistBaseIE):
|
|||||||
'url': 'https://www.pornhub.com/video/incategories/60fps-1/hd-porn',
|
'url': 'https://www.pornhub.com/video/incategories/60fps-1/hd-porn',
|
||||||
'only_matching': True,
|
'only_matching': True,
|
||||||
}, {
|
}, {
|
||||||
'url': 'https://pornhubthbh7ap3u.onion/model/zoe_ph/videos',
|
'url': 'https://pornhubvybmsymdol4iibwgwtkpwmeyd6luq2gxajgjzfjvotyt5zhyd.onion/model/zoe_ph/videos',
|
||||||
'only_matching': True,
|
'only_matching': True,
|
||||||
}]
|
}]
|
||||||
|
|
||||||
@@ -755,7 +756,7 @@ class PornHubUserVideosUploadIE(PornHubPagedPlaylistBaseIE):
|
|||||||
'url': 'https://www.pornhub.com/model/zoe_ph/videos/upload',
|
'url': 'https://www.pornhub.com/model/zoe_ph/videos/upload',
|
||||||
'only_matching': True,
|
'only_matching': True,
|
||||||
}, {
|
}, {
|
||||||
'url': 'http://pornhubthbh7ap3u.onion/pornstar/jenny-blighe/videos/upload',
|
'url': 'http://pornhubvybmsymdol4iibwgwtkpwmeyd6luq2gxajgjzfjvotyt5zhyd.onion/pornstar/jenny-blighe/videos/upload',
|
||||||
'only_matching': True,
|
'only_matching': True,
|
||||||
}]
|
}]
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,431 @@
|
|||||||
|
# coding: utf-8
|
||||||
|
from __future__ import unicode_literals
|
||||||
|
|
||||||
|
import itertools
|
||||||
|
from .common import InfoExtractor, SearchInfoExtractor
|
||||||
|
from ..utils import (
|
||||||
|
urljoin,
|
||||||
|
traverse_obj,
|
||||||
|
int_or_none,
|
||||||
|
mimetype2ext,
|
||||||
|
clean_html,
|
||||||
|
url_or_none,
|
||||||
|
unified_timestamp,
|
||||||
|
str_or_none,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class PRXBaseIE(InfoExtractor):
|
||||||
|
PRX_BASE_URL_RE = r'https?://(?:(?:beta|listen)\.)?prx.org/%s'
|
||||||
|
|
||||||
|
def _call_api(self, item_id, path, query=None, fatal=True, note='Downloading CMS API JSON'):
|
||||||
|
return self._download_json(
|
||||||
|
urljoin('https://cms.prx.org/api/v1/', path), item_id, query=query, fatal=fatal, note=note)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _get_prx_embed_response(response, section):
|
||||||
|
return traverse_obj(response, ('_embedded', f'prx:{section}'))
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _extract_file_link(response):
|
||||||
|
return url_or_none(traverse_obj(
|
||||||
|
response, ('_links', 'enclosure', 'href'), expected_type=str))
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def _extract_image(cls, image_response):
|
||||||
|
if not isinstance(image_response, dict):
|
||||||
|
return
|
||||||
|
return {
|
||||||
|
'id': str_or_none(image_response.get('id')),
|
||||||
|
'filesize': image_response.get('size'),
|
||||||
|
'width': image_response.get('width'),
|
||||||
|
'height': image_response.get('height'),
|
||||||
|
'url': cls._extract_file_link(image_response)
|
||||||
|
}
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def _extract_base_info(cls, response):
|
||||||
|
if not isinstance(response, dict):
|
||||||
|
return
|
||||||
|
item_id = str_or_none(response.get('id'))
|
||||||
|
if not item_id:
|
||||||
|
return
|
||||||
|
thumbnail_dict = cls._extract_image(cls._get_prx_embed_response(response, 'image'))
|
||||||
|
description = (
|
||||||
|
clean_html(response.get('description'))
|
||||||
|
or response.get('shortDescription'))
|
||||||
|
return {
|
||||||
|
'id': item_id,
|
||||||
|
'title': response.get('title') or item_id,
|
||||||
|
'thumbnails': [thumbnail_dict] if thumbnail_dict else None,
|
||||||
|
'description': description,
|
||||||
|
'release_timestamp': unified_timestamp(response.get('releasedAt')),
|
||||||
|
'timestamp': unified_timestamp(response.get('createdAt')),
|
||||||
|
'modified_timestamp': unified_timestamp(response.get('updatedAt')),
|
||||||
|
'duration': int_or_none(response.get('duration')),
|
||||||
|
'tags': response.get('tags'),
|
||||||
|
'episode_number': int_or_none(response.get('episodeIdentifier')),
|
||||||
|
'season_number': int_or_none(response.get('seasonIdentifier'))
|
||||||
|
}
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def _extract_series_info(cls, series_response):
|
||||||
|
base_info = cls._extract_base_info(series_response)
|
||||||
|
if not base_info:
|
||||||
|
return
|
||||||
|
account_info = cls._extract_account_info(
|
||||||
|
cls._get_prx_embed_response(series_response, 'account')) or {}
|
||||||
|
return {
|
||||||
|
**base_info,
|
||||||
|
'channel_id': account_info.get('channel_id'),
|
||||||
|
'channel_url': account_info.get('channel_url'),
|
||||||
|
'channel': account_info.get('channel'),
|
||||||
|
'series': base_info.get('title'),
|
||||||
|
'series_id': base_info.get('id'),
|
||||||
|
}
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def _extract_account_info(cls, account_response):
|
||||||
|
base_info = cls._extract_base_info(account_response)
|
||||||
|
if not base_info:
|
||||||
|
return
|
||||||
|
name = account_response.get('name')
|
||||||
|
return {
|
||||||
|
**base_info,
|
||||||
|
'title': name,
|
||||||
|
'channel_id': base_info.get('id'),
|
||||||
|
'channel_url': 'https://beta.prx.org/accounts/%s' % base_info.get('id'),
|
||||||
|
'channel': name,
|
||||||
|
}
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def _extract_story_info(cls, story_response):
|
||||||
|
base_info = cls._extract_base_info(story_response)
|
||||||
|
if not base_info:
|
||||||
|
return
|
||||||
|
series = cls._extract_series_info(
|
||||||
|
cls._get_prx_embed_response(story_response, 'series')) or {}
|
||||||
|
account = cls._extract_account_info(
|
||||||
|
cls._get_prx_embed_response(story_response, 'account')) or {}
|
||||||
|
return {
|
||||||
|
**base_info,
|
||||||
|
'series': series.get('series'),
|
||||||
|
'series_id': series.get('series_id'),
|
||||||
|
'channel_id': account.get('channel_id'),
|
||||||
|
'channel_url': account.get('channel_url'),
|
||||||
|
'channel': account.get('channel')
|
||||||
|
}
|
||||||
|
|
||||||
|
def _entries(self, item_id, endpoint, entry_func, query=None):
|
||||||
|
"""
|
||||||
|
Extract entries from paginated list API
|
||||||
|
@param entry_func: Function to generate entry from response item
|
||||||
|
"""
|
||||||
|
total = 0
|
||||||
|
for page in itertools.count(1):
|
||||||
|
response = self._call_api(f'{item_id}: page {page}', endpoint, query={
|
||||||
|
**(query or {}),
|
||||||
|
'page': page,
|
||||||
|
'per': 100
|
||||||
|
})
|
||||||
|
items = self._get_prx_embed_response(response, 'items')
|
||||||
|
if not response or not items:
|
||||||
|
break
|
||||||
|
|
||||||
|
yield from filter(None, map(entry_func, items))
|
||||||
|
|
||||||
|
total += response['count']
|
||||||
|
if total >= response['total']:
|
||||||
|
break
|
||||||
|
|
||||||
|
def _story_playlist_entry(self, response):
|
||||||
|
story = self._extract_story_info(response)
|
||||||
|
if not story:
|
||||||
|
return
|
||||||
|
story.update({
|
||||||
|
'_type': 'url',
|
||||||
|
'url': 'https://beta.prx.org/stories/%s' % story['id'],
|
||||||
|
'ie_key': PRXStoryIE.ie_key()
|
||||||
|
})
|
||||||
|
return story
|
||||||
|
|
||||||
|
def _series_playlist_entry(self, response):
|
||||||
|
series = self._extract_series_info(response)
|
||||||
|
if not series:
|
||||||
|
return
|
||||||
|
series.update({
|
||||||
|
'_type': 'url',
|
||||||
|
'url': 'https://beta.prx.org/series/%s' % series['id'],
|
||||||
|
'ie_key': PRXSeriesIE.ie_key()
|
||||||
|
})
|
||||||
|
return series
|
||||||
|
|
||||||
|
|
||||||
|
class PRXStoryIE(PRXBaseIE):
|
||||||
|
_VALID_URL = PRXBaseIE.PRX_BASE_URL_RE % r'stories/(?P<id>\d+)'
|
||||||
|
|
||||||
|
_TESTS = [
|
||||||
|
{
|
||||||
|
# Story with season and episode details
|
||||||
|
'url': 'https://beta.prx.org/stories/399200',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '399200',
|
||||||
|
'title': 'Fly Me To The Moon',
|
||||||
|
'description': 'md5:43230168390b95d3322048d8a56bf2bb',
|
||||||
|
'release_timestamp': 1640250000,
|
||||||
|
'timestamp': 1640208972,
|
||||||
|
'modified_timestamp': 1641318202,
|
||||||
|
'duration': 1004,
|
||||||
|
'tags': 'count:7',
|
||||||
|
'episode_number': 8,
|
||||||
|
'season_number': 5,
|
||||||
|
'series': 'AirSpace',
|
||||||
|
'series_id': '38057',
|
||||||
|
'channel_id': '220986',
|
||||||
|
'channel_url': 'https://beta.prx.org/accounts/220986',
|
||||||
|
'channel': 'Air and Space Museum',
|
||||||
|
},
|
||||||
|
'playlist': [{
|
||||||
|
'info_dict': {
|
||||||
|
'id': '399200_part1',
|
||||||
|
'title': 'Fly Me To The Moon',
|
||||||
|
'description': 'md5:43230168390b95d3322048d8a56bf2bb',
|
||||||
|
'release_timestamp': 1640250000,
|
||||||
|
'timestamp': 1640208972,
|
||||||
|
'modified_timestamp': 1641318202,
|
||||||
|
'duration': 530,
|
||||||
|
'tags': 'count:7',
|
||||||
|
'episode_number': 8,
|
||||||
|
'season_number': 5,
|
||||||
|
'series': 'AirSpace',
|
||||||
|
'series_id': '38057',
|
||||||
|
'channel_id': '220986',
|
||||||
|
'channel_url': 'https://beta.prx.org/accounts/220986',
|
||||||
|
'channel': 'Air and Space Museum',
|
||||||
|
'ext': 'mp3',
|
||||||
|
'upload_date': '20211222',
|
||||||
|
'episode': 'Episode 8',
|
||||||
|
'release_date': '20211223',
|
||||||
|
'season': 'Season 5',
|
||||||
|
'modified_date': '20220104'
|
||||||
|
}
|
||||||
|
}, {
|
||||||
|
'info_dict': {
|
||||||
|
'id': '399200_part2',
|
||||||
|
'title': 'Fly Me To The Moon',
|
||||||
|
'description': 'md5:43230168390b95d3322048d8a56bf2bb',
|
||||||
|
'release_timestamp': 1640250000,
|
||||||
|
'timestamp': 1640208972,
|
||||||
|
'modified_timestamp': 1641318202,
|
||||||
|
'duration': 474,
|
||||||
|
'tags': 'count:7',
|
||||||
|
'episode_number': 8,
|
||||||
|
'season_number': 5,
|
||||||
|
'series': 'AirSpace',
|
||||||
|
'series_id': '38057',
|
||||||
|
'channel_id': '220986',
|
||||||
|
'channel_url': 'https://beta.prx.org/accounts/220986',
|
||||||
|
'channel': 'Air and Space Museum',
|
||||||
|
'ext': 'mp3',
|
||||||
|
'upload_date': '20211222',
|
||||||
|
'episode': 'Episode 8',
|
||||||
|
'release_date': '20211223',
|
||||||
|
'season': 'Season 5',
|
||||||
|
'modified_date': '20220104'
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
]
|
||||||
|
}, {
|
||||||
|
# Story with only split audio
|
||||||
|
'url': 'https://beta.prx.org/stories/326414',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '326414',
|
||||||
|
'title': 'Massachusetts v EPA',
|
||||||
|
'description': 'md5:744fffba08f19f4deab69fa8d49d5816',
|
||||||
|
'timestamp': 1592509124,
|
||||||
|
'modified_timestamp': 1592510457,
|
||||||
|
'duration': 3088,
|
||||||
|
'tags': 'count:0',
|
||||||
|
'series': 'Outside/In',
|
||||||
|
'series_id': '36252',
|
||||||
|
'channel_id': '206',
|
||||||
|
'channel_url': 'https://beta.prx.org/accounts/206',
|
||||||
|
'channel': 'New Hampshire Public Radio',
|
||||||
|
},
|
||||||
|
'playlist_count': 4
|
||||||
|
}, {
|
||||||
|
# Story with single combined audio
|
||||||
|
'url': 'https://beta.prx.org/stories/400404',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '400404',
|
||||||
|
'title': 'Cafe Chill (Episode 2022-01)',
|
||||||
|
'thumbnails': 'count:1',
|
||||||
|
'description': 'md5:9f1b5a3cbd64fb159d08c3baa31f1539',
|
||||||
|
'timestamp': 1641233952,
|
||||||
|
'modified_timestamp': 1641234248,
|
||||||
|
'duration': 3540,
|
||||||
|
'series': 'Café Chill',
|
||||||
|
'series_id': '37762',
|
||||||
|
'channel_id': '5767',
|
||||||
|
'channel_url': 'https://beta.prx.org/accounts/5767',
|
||||||
|
'channel': 'C89.5 - KNHC Seattle',
|
||||||
|
'ext': 'mp3',
|
||||||
|
'tags': 'count:0',
|
||||||
|
'thumbnail': r're:https?://cms\.prx\.org/pub/\w+/0/web/story_image/767965/medium/Aurora_Over_Trees\.jpg',
|
||||||
|
'upload_date': '20220103',
|
||||||
|
'modified_date': '20220103'
|
||||||
|
}
|
||||||
|
}, {
|
||||||
|
'url': 'https://listen.prx.org/stories/399200',
|
||||||
|
'only_matching': True
|
||||||
|
}
|
||||||
|
]
|
||||||
|
|
||||||
|
def _extract_audio_pieces(self, audio_response):
|
||||||
|
return [{
|
||||||
|
'format_id': str_or_none(piece_response.get('id')),
|
||||||
|
'format_note': str_or_none(piece_response.get('label')),
|
||||||
|
'filesize': int_or_none(piece_response.get('size')),
|
||||||
|
'duration': int_or_none(piece_response.get('duration')),
|
||||||
|
'ext': mimetype2ext(piece_response.get('contentType')),
|
||||||
|
'asr': int_or_none(piece_response.get('frequency'), scale=1000),
|
||||||
|
'abr': int_or_none(piece_response.get('bitRate')),
|
||||||
|
'url': self._extract_file_link(piece_response),
|
||||||
|
'vcodec': 'none'
|
||||||
|
} for piece_response in sorted(
|
||||||
|
self._get_prx_embed_response(audio_response, 'items') or [],
|
||||||
|
key=lambda p: int_or_none(p.get('position')))]
|
||||||
|
|
||||||
|
def _extract_story(self, story_response):
|
||||||
|
info = self._extract_story_info(story_response)
|
||||||
|
if not info:
|
||||||
|
return
|
||||||
|
audio_pieces = self._extract_audio_pieces(
|
||||||
|
self._get_prx_embed_response(story_response, 'audio'))
|
||||||
|
if len(audio_pieces) == 1:
|
||||||
|
return {
|
||||||
|
'formats': audio_pieces,
|
||||||
|
**info
|
||||||
|
}
|
||||||
|
|
||||||
|
entries = [{
|
||||||
|
**info,
|
||||||
|
'id': '%s_part%d' % (info['id'], (idx + 1)),
|
||||||
|
'formats': [fmt],
|
||||||
|
} for idx, fmt in enumerate(audio_pieces)]
|
||||||
|
return {
|
||||||
|
'_type': 'multi_video',
|
||||||
|
'entries': entries,
|
||||||
|
**info
|
||||||
|
}
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
story_id = self._match_id(url)
|
||||||
|
response = self._call_api(story_id, f'stories/{story_id}')
|
||||||
|
return self._extract_story(response)
|
||||||
|
|
||||||
|
|
||||||
|
class PRXSeriesIE(PRXBaseIE):
|
||||||
|
_VALID_URL = PRXBaseIE.PRX_BASE_URL_RE % r'series/(?P<id>\d+)'
|
||||||
|
_TESTS = [
|
||||||
|
{
|
||||||
|
'url': 'https://beta.prx.org/series/36252',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '36252',
|
||||||
|
'title': 'Outside/In',
|
||||||
|
'thumbnails': 'count:1',
|
||||||
|
'description': 'md5:a6bedc5f810777bcb09ab30ff9059114',
|
||||||
|
'timestamp': 1470684964,
|
||||||
|
'modified_timestamp': 1582308830,
|
||||||
|
'channel_id': '206',
|
||||||
|
'channel_url': 'https://beta.prx.org/accounts/206',
|
||||||
|
'channel': 'New Hampshire Public Radio',
|
||||||
|
'series': 'Outside/In',
|
||||||
|
'series_id': '36252'
|
||||||
|
},
|
||||||
|
'playlist_mincount': 39
|
||||||
|
}, {
|
||||||
|
# Blank series
|
||||||
|
'url': 'https://beta.prx.org/series/25038',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '25038',
|
||||||
|
'title': '25038',
|
||||||
|
'timestamp': 1207612800,
|
||||||
|
'modified_timestamp': 1207612800,
|
||||||
|
'channel_id': '206',
|
||||||
|
'channel_url': 'https://beta.prx.org/accounts/206',
|
||||||
|
'channel': 'New Hampshire Public Radio',
|
||||||
|
'series': '25038',
|
||||||
|
'series_id': '25038'
|
||||||
|
},
|
||||||
|
'playlist_count': 0
|
||||||
|
}
|
||||||
|
]
|
||||||
|
|
||||||
|
def _extract_series(self, series_response):
|
||||||
|
info = self._extract_series_info(series_response)
|
||||||
|
return {
|
||||||
|
'_type': 'playlist',
|
||||||
|
'entries': self._entries(info['id'], 'series/%s/stories' % info['id'], self._story_playlist_entry),
|
||||||
|
**info
|
||||||
|
}
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
series_id = self._match_id(url)
|
||||||
|
response = self._call_api(series_id, f'series/{series_id}')
|
||||||
|
return self._extract_series(response)
|
||||||
|
|
||||||
|
|
||||||
|
class PRXAccountIE(PRXBaseIE):
|
||||||
|
_VALID_URL = PRXBaseIE.PRX_BASE_URL_RE % r'accounts/(?P<id>\d+)'
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://beta.prx.org/accounts/206',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '206',
|
||||||
|
'title': 'New Hampshire Public Radio',
|
||||||
|
'description': 'md5:277f2395301d0aca563c80c70a18ee0a',
|
||||||
|
'channel_id': '206',
|
||||||
|
'channel_url': 'https://beta.prx.org/accounts/206',
|
||||||
|
'channel': 'New Hampshire Public Radio',
|
||||||
|
'thumbnails': 'count:1'
|
||||||
|
},
|
||||||
|
'playlist_mincount': 380
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _extract_account(self, account_response):
|
||||||
|
info = self._extract_account_info(account_response)
|
||||||
|
series = self._entries(
|
||||||
|
info['id'], f'accounts/{info["id"]}/series', self._series_playlist_entry)
|
||||||
|
stories = self._entries(
|
||||||
|
info['id'], f'accounts/{info["id"]}/stories', self._story_playlist_entry)
|
||||||
|
return {
|
||||||
|
'_type': 'playlist',
|
||||||
|
'entries': itertools.chain(series, stories),
|
||||||
|
**info
|
||||||
|
}
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
account_id = self._match_id(url)
|
||||||
|
response = self._call_api(account_id, f'accounts/{account_id}')
|
||||||
|
return self._extract_account(response)
|
||||||
|
|
||||||
|
|
||||||
|
class PRXStoriesSearchIE(PRXBaseIE, SearchInfoExtractor):
|
||||||
|
IE_DESC = 'PRX Stories Search'
|
||||||
|
IE_NAME = 'prxstories:search'
|
||||||
|
_SEARCH_KEY = 'prxstories'
|
||||||
|
|
||||||
|
def _search_results(self, query):
|
||||||
|
yield from self._entries(
|
||||||
|
f'query {query}', 'stories/search', self._story_playlist_entry, query={'q': query})
|
||||||
|
|
||||||
|
|
||||||
|
class PRXSeriesSearchIE(PRXBaseIE, SearchInfoExtractor):
|
||||||
|
IE_DESC = 'PRX Series Search'
|
||||||
|
IE_NAME = 'prxseries:search'
|
||||||
|
_SEARCH_KEY = 'prxseries'
|
||||||
|
|
||||||
|
def _search_results(self, query):
|
||||||
|
yield from self._entries(
|
||||||
|
f'query {query}', 'series/search', self._series_playlist_entry, query={'q': query})
|
||||||
@@ -1,6 +1,12 @@
|
|||||||
import json
|
import json
|
||||||
|
|
||||||
from ..utils import ExtractorError, traverse_obj, try_get, unified_timestamp
|
from ..utils import (
|
||||||
|
ExtractorError,
|
||||||
|
format_field,
|
||||||
|
traverse_obj,
|
||||||
|
try_get,
|
||||||
|
unified_timestamp
|
||||||
|
)
|
||||||
from .common import InfoExtractor
|
from .common import InfoExtractor
|
||||||
|
|
||||||
|
|
||||||
@@ -74,7 +80,7 @@ class RadLiveIE(InfoExtractor):
|
|||||||
'release_timestamp': release_date,
|
'release_timestamp': release_date,
|
||||||
'channel': channel.get('name'),
|
'channel': channel.get('name'),
|
||||||
'channel_id': channel_id,
|
'channel_id': channel_id,
|
||||||
'channel_url': f'https://rad.live/content/channel/{channel_id}' if channel_id else None,
|
'channel_url': format_field(channel_id, template='https://rad.live/content/channel/%s'),
|
||||||
|
|
||||||
}
|
}
|
||||||
if content_type == 'episode':
|
if content_type == 'episode':
|
||||||
|
|||||||
+153
-91
@@ -14,16 +14,14 @@ from ..utils import (
|
|||||||
find_xpath_attr,
|
find_xpath_attr,
|
||||||
fix_xml_ampersands,
|
fix_xml_ampersands,
|
||||||
GeoRestrictedError,
|
GeoRestrictedError,
|
||||||
get_element_by_class,
|
|
||||||
HEADRequest,
|
HEADRequest,
|
||||||
int_or_none,
|
int_or_none,
|
||||||
join_nonempty,
|
join_nonempty,
|
||||||
parse_duration,
|
parse_duration,
|
||||||
parse_list,
|
|
||||||
remove_start,
|
remove_start,
|
||||||
strip_or_none,
|
strip_or_none,
|
||||||
|
traverse_obj,
|
||||||
try_get,
|
try_get,
|
||||||
unescapeHTML,
|
|
||||||
unified_strdate,
|
unified_strdate,
|
||||||
unified_timestamp,
|
unified_timestamp,
|
||||||
update_url_query,
|
update_url_query,
|
||||||
@@ -37,7 +35,7 @@ class RaiBaseIE(InfoExtractor):
|
|||||||
_GEO_COUNTRIES = ['IT']
|
_GEO_COUNTRIES = ['IT']
|
||||||
_GEO_BYPASS = False
|
_GEO_BYPASS = False
|
||||||
|
|
||||||
def _extract_relinker_info(self, relinker_url, video_id):
|
def _extract_relinker_info(self, relinker_url, video_id, audio_only=False):
|
||||||
if not re.match(r'https?://', relinker_url):
|
if not re.match(r'https?://', relinker_url):
|
||||||
return {'formats': [{'url': relinker_url}]}
|
return {'formats': [{'url': relinker_url}]}
|
||||||
|
|
||||||
@@ -80,7 +78,15 @@ class RaiBaseIE(InfoExtractor):
|
|||||||
if (ext == 'm3u8' and platform != 'mon') or (ext == 'f4m' and platform != 'flash'):
|
if (ext == 'm3u8' and platform != 'mon') or (ext == 'f4m' and platform != 'flash'):
|
||||||
continue
|
continue
|
||||||
|
|
||||||
if ext == 'm3u8' or 'format=m3u8' in media_url or platform == 'mon':
|
if ext == 'mp3':
|
||||||
|
formats.append({
|
||||||
|
'url': media_url,
|
||||||
|
'vcodec': 'none',
|
||||||
|
'acodec': 'mp3',
|
||||||
|
'format_id': 'http-mp3',
|
||||||
|
})
|
||||||
|
break
|
||||||
|
elif ext == 'm3u8' or 'format=m3u8' in media_url or platform == 'mon':
|
||||||
formats.extend(self._extract_m3u8_formats(
|
formats.extend(self._extract_m3u8_formats(
|
||||||
media_url, video_id, 'mp4', 'm3u8_native',
|
media_url, video_id, 'mp4', 'm3u8_native',
|
||||||
m3u8_id='hls', fatal=False))
|
m3u8_id='hls', fatal=False))
|
||||||
@@ -101,7 +107,8 @@ class RaiBaseIE(InfoExtractor):
|
|||||||
if not formats and geoprotection is True:
|
if not formats and geoprotection is True:
|
||||||
self.raise_geo_restricted(countries=self._GEO_COUNTRIES, metadata_available=True)
|
self.raise_geo_restricted(countries=self._GEO_COUNTRIES, metadata_available=True)
|
||||||
|
|
||||||
formats.extend(self._create_http_urls(relinker_url, formats))
|
if not audio_only:
|
||||||
|
formats.extend(self._create_http_urls(relinker_url, formats))
|
||||||
|
|
||||||
return dict((k, v) for k, v in {
|
return dict((k, v) for k, v in {
|
||||||
'is_live': is_live,
|
'is_live': is_live,
|
||||||
@@ -359,26 +366,44 @@ class RaiPlayLiveIE(RaiPlayIE):
|
|||||||
|
|
||||||
|
|
||||||
class RaiPlayPlaylistIE(InfoExtractor):
|
class RaiPlayPlaylistIE(InfoExtractor):
|
||||||
_VALID_URL = r'(?P<base>https?://(?:www\.)?raiplay\.it/programmi/(?P<id>[^/?#&]+))'
|
_VALID_URL = r'(?P<base>https?://(?:www\.)?raiplay\.it/programmi/(?P<id>[^/?#&]+))(?:/(?P<extra_id>[^?#&]+))?'
|
||||||
_TESTS = [{
|
_TESTS = [{
|
||||||
'url': 'http://www.raiplay.it/programmi/nondirloalmiocapo/',
|
'url': 'https://www.raiplay.it/programmi/nondirloalmiocapo/',
|
||||||
'info_dict': {
|
'info_dict': {
|
||||||
'id': 'nondirloalmiocapo',
|
'id': 'nondirloalmiocapo',
|
||||||
'title': 'Non dirlo al mio capo',
|
'title': 'Non dirlo al mio capo',
|
||||||
'description': 'md5:98ab6b98f7f44c2843fd7d6f045f153b',
|
'description': 'md5:98ab6b98f7f44c2843fd7d6f045f153b',
|
||||||
},
|
},
|
||||||
'playlist_mincount': 12,
|
'playlist_mincount': 12,
|
||||||
|
}, {
|
||||||
|
'url': 'https://www.raiplay.it/programmi/nondirloalmiocapo/episodi/stagione-2/',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'nondirloalmiocapo',
|
||||||
|
'title': 'Non dirlo al mio capo - Stagione 2',
|
||||||
|
'description': 'md5:98ab6b98f7f44c2843fd7d6f045f153b',
|
||||||
|
},
|
||||||
|
'playlist_mincount': 12,
|
||||||
}]
|
}]
|
||||||
|
|
||||||
def _real_extract(self, url):
|
def _real_extract(self, url):
|
||||||
base, playlist_id = self._match_valid_url(url).groups()
|
base, playlist_id, extra_id = self._match_valid_url(url).groups()
|
||||||
|
|
||||||
program = self._download_json(
|
program = self._download_json(
|
||||||
base + '.json', playlist_id, 'Downloading program JSON')
|
base + '.json', playlist_id, 'Downloading program JSON')
|
||||||
|
|
||||||
|
if extra_id:
|
||||||
|
extra_id = extra_id.upper().rstrip('/')
|
||||||
|
|
||||||
|
playlist_title = program.get('name')
|
||||||
entries = []
|
entries = []
|
||||||
for b in (program.get('blocks') or []):
|
for b in (program.get('blocks') or []):
|
||||||
for s in (b.get('sets') or []):
|
for s in (b.get('sets') or []):
|
||||||
|
if extra_id:
|
||||||
|
if extra_id != join_nonempty(
|
||||||
|
b.get('name'), s.get('name'), delim='/').replace(' ', '-').upper():
|
||||||
|
continue
|
||||||
|
playlist_title = join_nonempty(playlist_title, s.get('name'), delim=' - ')
|
||||||
|
|
||||||
s_id = s.get('id')
|
s_id = s.get('id')
|
||||||
if not s_id:
|
if not s_id:
|
||||||
continue
|
continue
|
||||||
@@ -397,10 +422,128 @@ class RaiPlayPlaylistIE(InfoExtractor):
|
|||||||
video_id=RaiPlayIE._match_id(video_url)))
|
video_id=RaiPlayIE._match_id(video_url)))
|
||||||
|
|
||||||
return self.playlist_result(
|
return self.playlist_result(
|
||||||
entries, playlist_id, program.get('name'),
|
entries, playlist_id, playlist_title,
|
||||||
try_get(program, lambda x: x['program_info']['description']))
|
try_get(program, lambda x: x['program_info']['description']))
|
||||||
|
|
||||||
|
|
||||||
|
class RaiPlaySoundIE(RaiBaseIE):
|
||||||
|
_VALID_URL = r'(?P<base>https?://(?:www\.)?raiplaysound\.it/.+?-(?P<id>%s))\.(?:html|json)' % RaiBaseIE._UUID_RE
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://www.raiplaysound.it/audio/2021/12/IL-RUGGITO-DEL-CONIGLIO-1ebae2a7-7cdb-42bb-842e-fe0d193e9707.html',
|
||||||
|
'md5': '8970abf8caf8aef4696e7b1f2adfc696',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '1ebae2a7-7cdb-42bb-842e-fe0d193e9707',
|
||||||
|
'ext': 'mp3',
|
||||||
|
'title': 'Il Ruggito del Coniglio del 10/12/2021',
|
||||||
|
'description': 'md5:2a17d2107e59a4a8faa0e18334139ee2',
|
||||||
|
'thumbnail': r're:^https?://.*\.jpg$',
|
||||||
|
'uploader': 'rai radio 2',
|
||||||
|
'duration': 5685,
|
||||||
|
'series': 'Il Ruggito del Coniglio',
|
||||||
|
},
|
||||||
|
'params': {
|
||||||
|
'skip_download': True,
|
||||||
|
},
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
base, audio_id = self._match_valid_url(url).group('base', 'id')
|
||||||
|
media = self._download_json(f'{base}.json', audio_id, 'Downloading audio JSON')
|
||||||
|
uid = try_get(media, lambda x: remove_start(remove_start(x['uniquename'], 'ContentItem-'), 'Page-'))
|
||||||
|
|
||||||
|
info = {}
|
||||||
|
formats = []
|
||||||
|
relinkers = set(traverse_obj(media, (('downloadable_audio', 'audio', ('live', 'cards', 0, 'audio')), 'url')))
|
||||||
|
for r in relinkers:
|
||||||
|
info = self._extract_relinker_info(r, audio_id, True)
|
||||||
|
formats.extend(info.get('formats'))
|
||||||
|
|
||||||
|
date_published = try_get(media, (lambda x: f'{x["create_date"]} {x.get("create_time") or ""}',
|
||||||
|
lambda x: x['live']['create_date']))
|
||||||
|
|
||||||
|
podcast_info = traverse_obj(media, 'podcast_info', ('live', 'cards', 0)) or {}
|
||||||
|
thumbnails = [{
|
||||||
|
'url': urljoin(url, thumb_url),
|
||||||
|
} for thumb_url in (podcast_info.get('images') or {}).values() if thumb_url]
|
||||||
|
|
||||||
|
return {
|
||||||
|
**info,
|
||||||
|
'id': uid or audio_id,
|
||||||
|
'display_id': audio_id,
|
||||||
|
'title': traverse_obj(media, 'title', 'episode_title'),
|
||||||
|
'alt_title': traverse_obj(media, ('track_info', 'media_name')),
|
||||||
|
'description': media.get('description'),
|
||||||
|
'uploader': traverse_obj(media, ('track_info', 'channel'), expected_type=strip_or_none),
|
||||||
|
'creator': traverse_obj(media, ('track_info', 'editor'), expected_type=strip_or_none),
|
||||||
|
'timestamp': unified_timestamp(date_published),
|
||||||
|
'thumbnails': thumbnails,
|
||||||
|
'series': podcast_info.get('title'),
|
||||||
|
'season_number': int_or_none(media.get('season')),
|
||||||
|
'episode': media.get('episode_title'),
|
||||||
|
'episode_number': int_or_none(media.get('episode')),
|
||||||
|
'formats': formats,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class RaiPlaySoundLiveIE(RaiPlaySoundIE):
|
||||||
|
_VALID_URL = r'(?P<base>https?://(?:www\.)?raiplaysound\.it/(?P<id>[^/?#&]+)$)'
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://www.raiplaysound.it/radio2',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'b00a50e6-f404-4af6-8f8c-ff3b9af73a44',
|
||||||
|
'display_id': 'radio2',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'Rai Radio 2',
|
||||||
|
'uploader': 'rai radio 2',
|
||||||
|
'creator': 'raiplaysound',
|
||||||
|
'is_live': True,
|
||||||
|
},
|
||||||
|
'params': {
|
||||||
|
'skip_download': 'live',
|
||||||
|
},
|
||||||
|
}]
|
||||||
|
|
||||||
|
|
||||||
|
class RaiPlaySoundPlaylistIE(InfoExtractor):
|
||||||
|
_VALID_URL = r'(?P<base>https?://(?:www\.)?raiplaysound\.it/(?:programmi|playlist|audiolibri)/(?P<id>[^/?#&]+))(?:/(?P<extra_id>[^?#&]+))?'
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://www.raiplaysound.it/programmi/ilruggitodelconiglio',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'ilruggitodelconiglio',
|
||||||
|
'title': 'Il Ruggito del Coniglio',
|
||||||
|
'description': 'md5:1bbaf631245a7ab1ec4d9fbb3c7aa8f3',
|
||||||
|
},
|
||||||
|
'playlist_mincount': 65,
|
||||||
|
}, {
|
||||||
|
'url': 'https://www.raiplaysound.it/programmi/ilruggitodelconiglio/puntate/prima-stagione-1995',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'ilruggitodelconiglio_puntate_prima-stagione-1995',
|
||||||
|
'title': 'Prima Stagione 1995',
|
||||||
|
},
|
||||||
|
'playlist_count': 1,
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
base, playlist_id, extra_id = self._match_valid_url(url).group('base', 'id', 'extra_id')
|
||||||
|
url = f'{base}.json'
|
||||||
|
program = self._download_json(url, playlist_id, 'Downloading program JSON')
|
||||||
|
|
||||||
|
if extra_id:
|
||||||
|
extra_id = extra_id.rstrip('/')
|
||||||
|
playlist_id += '_' + extra_id.replace('/', '_')
|
||||||
|
path = next(c['path_id'] for c in program.get('filters') or [] if extra_id in c.get('weblink'))
|
||||||
|
program = self._download_json(
|
||||||
|
urljoin('https://www.raiplaysound.it', path), playlist_id, 'Downloading program secondary JSON')
|
||||||
|
|
||||||
|
entries = [
|
||||||
|
self.url_result(urljoin(base, c['path_id']), ie=RaiPlaySoundIE.ie_key())
|
||||||
|
for c in traverse_obj(program, 'cards', ('block', 'cards')) or []
|
||||||
|
if c.get('path_id')]
|
||||||
|
|
||||||
|
return self.playlist_result(entries, playlist_id, program.get('title'),
|
||||||
|
traverse_obj(program, ('podcast_info', 'description')))
|
||||||
|
|
||||||
|
|
||||||
class RaiIE(RaiBaseIE):
|
class RaiIE(RaiBaseIE):
|
||||||
_VALID_URL = r'https?://[^/]+\.(?:rai\.(?:it|tv)|rainews\.it)/.+?-(?P<id>%s)(?:-.+?)?\.html' % RaiBaseIE._UUID_RE
|
_VALID_URL = r'https?://[^/]+\.(?:rai\.(?:it|tv)|rainews\.it)/.+?-(?P<id>%s)(?:-.+?)?\.html' % RaiBaseIE._UUID_RE
|
||||||
_TESTS = [{
|
_TESTS = [{
|
||||||
@@ -593,84 +736,3 @@ class RaiIE(RaiBaseIE):
|
|||||||
info.update(relinker_info)
|
info.update(relinker_info)
|
||||||
|
|
||||||
return info
|
return info
|
||||||
|
|
||||||
|
|
||||||
class RaiPlayRadioBaseIE(InfoExtractor):
|
|
||||||
_BASE = 'https://www.raiplayradio.it'
|
|
||||||
|
|
||||||
def get_playlist_iter(self, url, uid):
|
|
||||||
webpage = self._download_webpage(url, uid)
|
|
||||||
for attrs in parse_list(webpage):
|
|
||||||
title = attrs['data-title'].strip()
|
|
||||||
audio_url = urljoin(url, attrs['data-mediapolis'])
|
|
||||||
entry = {
|
|
||||||
'url': audio_url,
|
|
||||||
'id': attrs['data-uniquename'].lstrip('ContentItem-'),
|
|
||||||
'title': title,
|
|
||||||
'ext': 'mp3',
|
|
||||||
'language': 'it',
|
|
||||||
}
|
|
||||||
if 'data-image' in attrs:
|
|
||||||
entry['thumbnail'] = urljoin(url, attrs['data-image'])
|
|
||||||
yield entry
|
|
||||||
|
|
||||||
|
|
||||||
class RaiPlayRadioIE(RaiPlayRadioBaseIE):
|
|
||||||
_VALID_URL = r'%s/audio/.+?-(?P<id>%s)\.html' % (
|
|
||||||
RaiPlayRadioBaseIE._BASE, RaiBaseIE._UUID_RE)
|
|
||||||
_TEST = {
|
|
||||||
'url': 'https://www.raiplayradio.it/audio/2019/07/RADIO3---LEZIONI-DI-MUSICA-36b099ff-4123-4443-9bf9-38e43ef5e025.html',
|
|
||||||
'info_dict': {
|
|
||||||
'id': '36b099ff-4123-4443-9bf9-38e43ef5e025',
|
|
||||||
'ext': 'mp3',
|
|
||||||
'title': 'Dal "Chiaro di luna" al "Clair de lune", prima parte con Giovanni Bietti',
|
|
||||||
'thumbnail': r're:^https?://.*\.jpg$',
|
|
||||||
'language': 'it',
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
def _real_extract(self, url):
|
|
||||||
audio_id = self._match_id(url)
|
|
||||||
list_url = url.replace('.html', '-list.html')
|
|
||||||
return next(entry for entry in self.get_playlist_iter(list_url, audio_id) if entry['id'] == audio_id)
|
|
||||||
|
|
||||||
|
|
||||||
class RaiPlayRadioPlaylistIE(RaiPlayRadioBaseIE):
|
|
||||||
_VALID_URL = r'%s/playlist/.+?-(?P<id>%s)\.html' % (
|
|
||||||
RaiPlayRadioBaseIE._BASE, RaiBaseIE._UUID_RE)
|
|
||||||
_TEST = {
|
|
||||||
'url': 'https://www.raiplayradio.it/playlist/2017/12/Alice-nel-paese-delle-meraviglie-72371d3c-d998-49f3-8860-d168cfdf4966.html',
|
|
||||||
'info_dict': {
|
|
||||||
'id': '72371d3c-d998-49f3-8860-d168cfdf4966',
|
|
||||||
'title': "Alice nel paese delle meraviglie",
|
|
||||||
'description': "di Lewis Carrol letto da Aldo Busi",
|
|
||||||
},
|
|
||||||
'playlist_count': 11,
|
|
||||||
}
|
|
||||||
|
|
||||||
def _real_extract(self, url):
|
|
||||||
playlist_id = self._match_id(url)
|
|
||||||
playlist_webpage = self._download_webpage(url, playlist_id)
|
|
||||||
playlist_title = unescapeHTML(self._html_search_regex(
|
|
||||||
r'data-playlist-title="(.+?)"', playlist_webpage, 'title'))
|
|
||||||
playlist_creator = self._html_search_meta(
|
|
||||||
'nomeProgramma', playlist_webpage)
|
|
||||||
playlist_description = get_element_by_class(
|
|
||||||
'textDescriptionProgramma', playlist_webpage)
|
|
||||||
|
|
||||||
player_href = self._html_search_regex(
|
|
||||||
r'data-player-href="(.+?)"', playlist_webpage, 'href')
|
|
||||||
list_url = urljoin(url, player_href)
|
|
||||||
|
|
||||||
entries = list(self.get_playlist_iter(list_url, playlist_id))
|
|
||||||
for index, entry in enumerate(entries, start=1):
|
|
||||||
entry.update({
|
|
||||||
'track': entry['title'],
|
|
||||||
'track_number': index,
|
|
||||||
'artist': playlist_creator,
|
|
||||||
'album': playlist_title
|
|
||||||
})
|
|
||||||
|
|
||||||
return self.playlist_result(
|
|
||||||
entries, playlist_id, playlist_title, playlist_description,
|
|
||||||
creator=playlist_creator)
|
|
||||||
|
|||||||
@@ -81,12 +81,11 @@ class RedBullTVIE(InfoExtractor):
|
|||||||
|
|
||||||
title = video['title'].strip()
|
title = video['title'].strip()
|
||||||
|
|
||||||
formats = self._extract_m3u8_formats(
|
formats, subtitles = self._extract_m3u8_formats_and_subtitles(
|
||||||
'https://dms.redbull.tv/v3/%s/%s/playlist.m3u8' % (video_id, token),
|
'https://dms.redbull.tv/v3/%s/%s/playlist.m3u8' % (video_id, token),
|
||||||
video_id, 'mp4', entry_protocol='m3u8_native', m3u8_id='hls')
|
video_id, 'mp4', entry_protocol='m3u8_native', m3u8_id='hls')
|
||||||
self._sort_formats(formats)
|
self._sort_formats(formats)
|
||||||
|
|
||||||
subtitles = {}
|
|
||||||
for resource in video.get('resources', []):
|
for resource in video.get('resources', []):
|
||||||
if resource.startswith('closed_caption_'):
|
if resource.startswith('closed_caption_'):
|
||||||
splitted_resource = resource.split('_')
|
splitted_resource = resource.split('_')
|
||||||
|
|||||||
@@ -0,0 +1,199 @@
|
|||||||
|
# coding: utf-8
|
||||||
|
from __future__ import unicode_literals
|
||||||
|
|
||||||
|
import re
|
||||||
|
|
||||||
|
from .common import InfoExtractor
|
||||||
|
from ..utils import js_to_json
|
||||||
|
|
||||||
|
|
||||||
|
class RTNewsIE(InfoExtractor):
|
||||||
|
_VALID_URL = r'https?://(?:www\.)?rt\.com/[^/]+/(?:[^/]+/)?(?P<id>\d+)'
|
||||||
|
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://www.rt.com/sport/546301-djokovic-arrives-belgrade-crowds/',
|
||||||
|
'playlist_mincount': 2,
|
||||||
|
'info_dict': {
|
||||||
|
'id': '546301',
|
||||||
|
'title': 'Crowds gather to greet deported Djokovic as he returns to Serbia (VIDEO)',
|
||||||
|
'description': 'md5:1d5bfe1a988d81fd74227cfdf93d314d',
|
||||||
|
'thumbnail': 'https://cdni.rt.com/files/2022.01/article/61e587a085f540102c3386c1.png'
|
||||||
|
},
|
||||||
|
}, {
|
||||||
|
'url': 'https://www.rt.com/shows/in-question/535980-plot-to-assassinate-julian-assange/',
|
||||||
|
'playlist_mincount': 1,
|
||||||
|
'info_dict': {
|
||||||
|
'id': '535980',
|
||||||
|
'title': 'The plot to assassinate Julian Assange',
|
||||||
|
'description': 'md5:55279ce5e4441dc1d16e2e4a730152cd',
|
||||||
|
'thumbnail': 'https://cdni.rt.com/files/2021.09/article/615226f42030274e8879b53d.png'
|
||||||
|
},
|
||||||
|
'playlist': [{
|
||||||
|
'info_dict': {
|
||||||
|
'id': '6152271d85f5400464496162',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': '6152271d85f5400464496162',
|
||||||
|
},
|
||||||
|
}]
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _entries(self, webpage):
|
||||||
|
video_urls = set(re.findall(r'https://cdnv\.rt\.com/.*[a-f0-9]+\.mp4', webpage))
|
||||||
|
for v_url in video_urls:
|
||||||
|
v_id = re.search(r'([a-f0-9]+)\.mp4', v_url).group(1)
|
||||||
|
if v_id:
|
||||||
|
yield {
|
||||||
|
'id': v_id,
|
||||||
|
'title': v_id,
|
||||||
|
'url': v_url,
|
||||||
|
}
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
id = self._match_id(url)
|
||||||
|
webpage = self._download_webpage(url, id)
|
||||||
|
|
||||||
|
return {
|
||||||
|
'_type': 'playlist',
|
||||||
|
'id': id,
|
||||||
|
'entries': self._entries(webpage),
|
||||||
|
'title': self._og_search_title(webpage),
|
||||||
|
'description': self._og_search_description(webpage),
|
||||||
|
'thumbnail': self._og_search_thumbnail(webpage),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class RTDocumentryIE(InfoExtractor):
|
||||||
|
_VALID_URL = r'https?://rtd\.rt\.com/(?:(?:series|shows)/[^/]+|films)/(?P<id>[^/?$&#]+)'
|
||||||
|
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://rtd.rt.com/films/escobars-hitman/',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'escobars-hitman',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': "Escobar's Hitman. Former drug-gang killer, now loved and loathed in Colombia",
|
||||||
|
'description': 'md5:647c76984b7cb9a8b52a567e87448d88',
|
||||||
|
'thumbnail': 'https://cdni.rt.com/rtd-files/films/escobars-hitman/escobars-hitman_11.jpg',
|
||||||
|
'average_rating': 8.53,
|
||||||
|
'duration': 3134.0
|
||||||
|
},
|
||||||
|
'params': {'skip_download': True}
|
||||||
|
}, {
|
||||||
|
'url': 'https://rtd.rt.com/shows/the-kalashnikova-show-military-secrets-anna-knishenko/iskander-tactical-system-natos-headache/',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'iskander-tactical-system-natos-headache',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': "Iskander tactical system. NATO's headache | The Kalashnikova Show. Episode 10",
|
||||||
|
'description': 'md5:da7c24a0aa67bc2bb88c86658508ca87',
|
||||||
|
'thumbnail': 'md5:89de8ce38c710b7c501ff02d47e2aa89',
|
||||||
|
'average_rating': 9.27,
|
||||||
|
'duration': 274.0,
|
||||||
|
'timestamp': 1605726000,
|
||||||
|
'view_count': int,
|
||||||
|
'upload_date': '20201118'
|
||||||
|
},
|
||||||
|
'params': {'skip_download': True}
|
||||||
|
}, {
|
||||||
|
'url': 'https://rtd.rt.com/series/i-am-hacked-trailer/introduction-to-safe-digital-life-ep2/',
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'introduction-to-safe-digital-life-ep2',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'How to Keep your Money away from Hackers | I am Hacked. Episode 2',
|
||||||
|
'description': 'md5:c46fa9a5af86c0008c45a3940a8cce87',
|
||||||
|
'thumbnail': 'md5:a5e81b9bf5aed8f5e23d9c053601b825',
|
||||||
|
'average_rating': 10.0,
|
||||||
|
'duration': 1524.0,
|
||||||
|
'timestamp': 1636977600,
|
||||||
|
'view_count': int,
|
||||||
|
'upload_date': '20211115'
|
||||||
|
},
|
||||||
|
'params': {'skip_download': True}
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
id = self._match_id(url)
|
||||||
|
webpage = self._download_webpage(url, id)
|
||||||
|
ld_json = self._search_json_ld(webpage, None, fatal=False)
|
||||||
|
if not ld_json:
|
||||||
|
self.raise_no_formats('No video/audio found at the provided url.', expected=True)
|
||||||
|
media_json = self._parse_json(
|
||||||
|
self._search_regex(r'(?s)\'Med\'\s*:\s*\[\s*({.+})\s*\]\s*};', webpage, 'media info'),
|
||||||
|
id, transform_source=js_to_json)
|
||||||
|
if 'title' not in ld_json and 'title' in media_json:
|
||||||
|
ld_json['title'] = media_json['title']
|
||||||
|
formats = [{'url': src['file']} for src in media_json.get('sources') or [] if src.get('file')]
|
||||||
|
|
||||||
|
return {
|
||||||
|
'id': id,
|
||||||
|
'thumbnail': media_json.get('image'),
|
||||||
|
'formats': formats,
|
||||||
|
**ld_json
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class RTDocumentryPlaylistIE(InfoExtractor):
|
||||||
|
_VALID_URL = r'https?://rtd\.rt\.com/(?:series|shows)/(?P<id>[^/]+)/$'
|
||||||
|
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://rtd.rt.com/series/i-am-hacked-trailer/',
|
||||||
|
'playlist_mincount': 6,
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'i-am-hacked-trailer',
|
||||||
|
},
|
||||||
|
}, {
|
||||||
|
'url': 'https://rtd.rt.com/shows/the-kalashnikova-show-military-secrets-anna-knishenko/',
|
||||||
|
'playlist_mincount': 34,
|
||||||
|
'info_dict': {
|
||||||
|
'id': 'the-kalashnikova-show-military-secrets-anna-knishenko',
|
||||||
|
},
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _entries(self, webpage, id):
|
||||||
|
video_urls = set(re.findall(r'list-2__link\s*"\s*href="([^"]+)"', webpage))
|
||||||
|
for v_url in video_urls:
|
||||||
|
if id not in v_url:
|
||||||
|
continue
|
||||||
|
yield self.url_result(
|
||||||
|
'https://rtd.rt.com%s' % v_url,
|
||||||
|
ie=RTDocumentryIE.ie_key())
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
id = self._match_id(url)
|
||||||
|
webpage = self._download_webpage(url, id)
|
||||||
|
|
||||||
|
return {
|
||||||
|
'_type': 'playlist',
|
||||||
|
'id': id,
|
||||||
|
'entries': self._entries(webpage, id),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class RuptlyIE(InfoExtractor):
|
||||||
|
_VALID_URL = r'https?://(?:www\.)?ruptly\.tv/[a-z]{2}/videos/(?P<id>\d+-\d+)'
|
||||||
|
|
||||||
|
_TESTS = [{
|
||||||
|
'url': 'https://www.ruptly.tv/en/videos/20220112-020-Japan-Double-trouble-Tokyo-zoo-presents-adorable-panda-twins',
|
||||||
|
'info_dict': {
|
||||||
|
'id': '20220112-020',
|
||||||
|
'ext': 'mp4',
|
||||||
|
'title': 'Japan: Double trouble! Tokyo zoo presents adorable panda twins | Video Ruptly',
|
||||||
|
'description': 'md5:85a8da5fdb31486f0562daf4360ce75a',
|
||||||
|
'thumbnail': 'https://storage.ruptly.tv/thumbnails/20220112-020/i6JQKnTNpYuqaXsR/i6JQKnTNpYuqaXsR.jpg'
|
||||||
|
},
|
||||||
|
'params': {'skip_download': True}
|
||||||
|
}]
|
||||||
|
|
||||||
|
def _real_extract(self, url):
|
||||||
|
id = self._match_id(url)
|
||||||
|
webpage = self._download_webpage(url, id)
|
||||||
|
m3u8_url = self._search_regex(r'preview_url"\s?:\s?"(https?://storage\.ruptly\.tv/video_projects/.+\.m3u8)"', webpage, 'm3u8 url', fatal=False)
|
||||||
|
if not m3u8_url:
|
||||||
|
self.raise_no_formats('No video/audio found at the provided url.', expected=True)
|
||||||
|
formats, subs = self._extract_m3u8_formats_and_subtitles(m3u8_url, id, ext='mp4')
|
||||||
|
return {
|
||||||
|
'id': id,
|
||||||
|
'formats': formats,
|
||||||
|
'subtitles': subs,
|
||||||
|
'title': self._og_search_title(webpage),
|
||||||
|
'description': self._og_search_description(webpage),
|
||||||
|
'thumbnail': self._og_search_thumbnail(webpage),
|
||||||
|
}
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user