mirror of
https://github.com/rembo10/headphones.git
synced 2026-09-10 08:41:41 +01:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6720575bd9 | ||
|
|
b1ec750d1d | ||
|
|
d9c78da997 | ||
|
|
1fa26e763d | ||
|
|
bb71b3d3d9 | ||
|
|
65e8e56e4a | ||
|
|
ee415d0e50 | ||
|
|
9db88a6c4f | ||
|
|
e5151a0a19 | ||
|
|
570234717a | ||
|
|
3f60d1f8de | ||
|
|
65d100c814 | ||
|
|
61bdab8621 | ||
|
|
8c94c9ecbf | ||
|
|
7fd895ecb5 | ||
|
|
213bc07646 | ||
|
|
1f081c181c | ||
|
|
db1d31900e | ||
|
|
2219c79ac3 | ||
|
|
d2782179aa | ||
|
|
d90a31afc7 | ||
|
|
1c236fe0f3 | ||
|
|
24165fb5f3 | ||
|
|
71da960386 | ||
|
|
a16c2d0a26 | ||
|
|
53de44a225 | ||
|
|
e32cff60e5 | ||
|
|
0ebba8d17c | ||
|
|
464db79c04 | ||
|
|
65019d8323 | ||
|
|
f131d316ca | ||
|
|
e6663c96b5 | ||
|
|
5f06d11fe9 | ||
|
|
6db2836662 | ||
|
|
cb209f8d98 | ||
|
|
5634e17a2e | ||
|
|
7c990edeaf | ||
|
|
9149f62760 | ||
|
|
872ec6b11c | ||
|
|
064a21a07c | ||
|
|
57c3c7b2e0 | ||
|
|
6dc75d7a44 | ||
|
|
5f3b7e05c5 | ||
|
|
61d3d8ad00 | ||
|
|
e36440f834 | ||
|
|
4029543d51 | ||
|
|
8aee9ac91f | ||
|
|
6e5e15f06f | ||
|
|
a72bd4cb7c | ||
|
|
d062ab553c | ||
|
|
a41edde12a | ||
|
|
7608927b4f | ||
|
|
24d7647c08 | ||
|
|
a14369cb7d | ||
|
|
3e81c4f3b6 | ||
|
|
84a6c19677 | ||
|
|
624a1d5509 | ||
|
|
8b6d35cfe2 | ||
|
|
867e502448 | ||
|
|
74275b969e | ||
|
|
275dde0630 | ||
|
|
46b470f30b | ||
|
|
fbfec95a8a | ||
|
|
1e75a18f55 | ||
|
|
a98849bf30 | ||
|
|
1b998e9d01 | ||
|
|
ebd6d3f3f9 | ||
|
|
1204afadbb | ||
|
|
cd08b72df7 |
@@ -1,5 +1,51 @@
|
|||||||
# Changelog
|
# Changelog
|
||||||
|
|
||||||
|
## v0.5.9
|
||||||
|
Released 05 September 2015
|
||||||
|
|
||||||
|
Highlights:
|
||||||
|
* Added: Providers Strike, Jackett, custom Torznabs
|
||||||
|
* Added: Option to stop post-processing if no good match found (#2343)
|
||||||
|
* Fixed: Blackhole -> Magnet, limit to torcache
|
||||||
|
* Fixed: Kat 403 flac error
|
||||||
|
* Fixed: Last.fm errors
|
||||||
|
* Fixed: Pushover notifications
|
||||||
|
* Improved: Rutracker logging, switched to requests lib
|
||||||
|
|
||||||
|
The full list of commits can be found [here](https://github.com/rembo10/headphones/compare/v0.5.6...v0.5.7).
|
||||||
|
|
||||||
|
## v0.5.8
|
||||||
|
Released 13 July 2015
|
||||||
|
|
||||||
|
Highlights:
|
||||||
|
* Added: Option to only include official extras
|
||||||
|
* Added: Option to wait until album release date before searching
|
||||||
|
* Fixed: NotifyMyAndroid notifications
|
||||||
|
* Fixed: Plex Notifications
|
||||||
|
* Fixed: Metacritic parsing
|
||||||
|
* Fixed: Pushbullet notifications
|
||||||
|
* Fixed: What.cd not honoring custom search term (#2279)
|
||||||
|
* Improved: XSS Search bug
|
||||||
|
* Improved: Config page layout
|
||||||
|
* Improved: Set localhost as default
|
||||||
|
* Improved: Better single artist scanning
|
||||||
|
|
||||||
|
The full list of commits can be found [here](https://github.com/rembo10/headphones/compare/v0.5.6...v0.5.7).
|
||||||
|
|
||||||
|
## v0.5.7
|
||||||
|
Released 01 July 2015
|
||||||
|
|
||||||
|
Highlights:
|
||||||
|
* Improved: Moved pushover to use requests lib
|
||||||
|
* Improved: Plex tokens with Plex Home
|
||||||
|
* Improved: Added getLogs & clearLogs to api
|
||||||
|
* Improved: Cache MetaCritic scores. Added user-agent header to fix 403 errors
|
||||||
|
* Improved: Specify whether to delete folders when force post-processing
|
||||||
|
* Improved: Convert target bitrate to vbr preset for what.cd searching
|
||||||
|
* Improved: Switched Pushover to requests lib
|
||||||
|
|
||||||
|
The full list of commits can be found [here](https://github.com/rembo10/headphones/compare/v0.5.6...v0.5.7).
|
||||||
|
|
||||||
## v0.5.6
|
## v0.5.6
|
||||||
Released 08 June 2015
|
Released 08 June 2015
|
||||||
|
|
||||||
|
|||||||
@@ -35,6 +35,7 @@
|
|||||||
%for alternate_album in alternate_albums:
|
%for alternate_album in alternate_albums:
|
||||||
<%
|
<%
|
||||||
track_count = len(myDB.select("SELECT * from alltracks WHERE ReleaseID=?", [alternate_album['ReleaseID']]))
|
track_count = len(myDB.select("SELECT * from alltracks WHERE ReleaseID=?", [alternate_album['ReleaseID']]))
|
||||||
|
mb_link = "http://musicbrainz.org/release/" + alternate_album['ReleaseID']
|
||||||
have_track_count = len(myDB.select("SELECT * from alltracks WHERE ReleaseID=? AND Location IS NOT NULL", [alternate_album['ReleaseID']]))
|
have_track_count = len(myDB.select("SELECT * from alltracks WHERE ReleaseID=? AND Location IS NOT NULL", [alternate_album['ReleaseID']]))
|
||||||
if alternate_album['AlbumID'] == alternate_album['ReleaseID']:
|
if alternate_album['AlbumID'] == alternate_album['ReleaseID']:
|
||||||
alternate_album_name = "Headphones Default Release (" + str(alternate_album['ReleaseDate']) + ") [" + str(have_track_count) + "/" + str(track_count) + " tracks]"
|
alternate_album_name = "Headphones Default Release (" + str(alternate_album['ReleaseDate']) + ") [" + str(have_track_count) + "/" + str(track_count) + " tracks]"
|
||||||
@@ -42,7 +43,7 @@
|
|||||||
alternate_album_name = alternate_album['AlbumTitle'] + " (" + alternate_album['ReleaseCountry'] + ", " + str(alternate_album['ReleaseDate']) + ", " + alternate_album['ReleaseFormat'] + ") [" + str(have_track_count) + "/" + str(track_count) + " tracks]"
|
alternate_album_name = alternate_album['AlbumTitle'] + " (" + alternate_album['ReleaseCountry'] + ", " + str(alternate_album['ReleaseDate']) + ", " + alternate_album['ReleaseFormat'] + ") [" + str(have_track_count) + "/" + str(track_count) + " tracks]"
|
||||||
|
|
||||||
%>
|
%>
|
||||||
<a href="#" onclick="doAjaxCall('switchAlbum?AlbumID=${album['AlbumID']}&ReleaseID=${alternate_album['ReleaseID']}', $(this), 'table');" data-success="Switched release to: ${alternate_album_name}">${alternate_album_name}</a><br>
|
<a href="#" onclick="doAjaxCall('switchAlbum?AlbumID=${album['AlbumID']}&ReleaseID=${alternate_album['ReleaseID']}', $(this), 'table');" data-success="Switched release to: ${alternate_album_name}">${alternate_album_name}</a><a href="${mb_link}" target="_blank">MB</a><br>
|
||||||
%endfor
|
%endfor
|
||||||
%endif
|
%endif
|
||||||
</div>
|
</div>
|
||||||
|
|||||||
+572
-437
File diff suppressed because it is too large
Load Diff
@@ -325,7 +325,7 @@ form fieldset small.heading {
|
|||||||
margin-top: -15px;
|
margin-top: -15px;
|
||||||
}
|
}
|
||||||
form .row {
|
form .row {
|
||||||
font-family: Helvetica, Arial;
|
font-family: Helvetica, Arial, sans-serif;
|
||||||
margin-bottom: 10px;
|
margin-bottom: 10px;
|
||||||
}
|
}
|
||||||
form .row label {
|
form .row label {
|
||||||
@@ -408,6 +408,18 @@ form .checkbox small {
|
|||||||
form .indent input {
|
form .indent input {
|
||||||
margin-left: 15px;
|
margin-left: 15px;
|
||||||
}
|
}
|
||||||
|
form .suboptions {
|
||||||
|
margin-left: 15px;
|
||||||
|
}
|
||||||
|
.option{
|
||||||
|
font-size: 15px;
|
||||||
|
font-weight: 600;
|
||||||
|
vertical-align: middle;
|
||||||
|
}
|
||||||
|
input.bigcheck[type="checkbox"] {
|
||||||
|
width: 16px;
|
||||||
|
height: 16px;
|
||||||
|
}
|
||||||
ul,
|
ul,
|
||||||
ol {
|
ol {
|
||||||
margin-left: 2em;
|
margin-left: 2em;
|
||||||
@@ -1420,6 +1432,13 @@ div#artistheader h2 a {
|
|||||||
line-height: 10px;
|
line-height: 10px;
|
||||||
height: 10px;
|
height: 10px;
|
||||||
}
|
}
|
||||||
|
.nopad {
|
||||||
|
margin-bottom: 0px !important;
|
||||||
|
margin-top: 0px !important;
|
||||||
|
padding-top: 0px !important;
|
||||||
|
padding-bottom: 0px !important;
|
||||||
|
padding: 0px 0px !important;
|
||||||
|
}
|
||||||
#album_table th#albumname,
|
#album_table th#albumname,
|
||||||
#upcoming_table th#artistname,
|
#upcoming_table th#artistname,
|
||||||
#wanted_table th#artistname {
|
#wanted_table th#artistname {
|
||||||
|
|||||||
@@ -52,6 +52,8 @@
|
|||||||
fileid = 'torrent'
|
fileid = 'torrent'
|
||||||
if item['URL'].find('rutracker') != -1:
|
if item['URL'].find('rutracker') != -1:
|
||||||
fileid = 'torrent'
|
fileid = 'torrent'
|
||||||
|
if item['URL'].find('codeshy') != -1:
|
||||||
|
fileid = 'nzb'
|
||||||
|
|
||||||
folder = 'Folder: ' + item['FolderName']
|
folder = 'Folder: ' + item['FolderName']
|
||||||
|
|
||||||
|
|||||||
@@ -137,27 +137,26 @@
|
|||||||
No empty artists found.
|
No empty artists found.
|
||||||
%endif
|
%endif
|
||||||
</div>
|
</div>
|
||||||
<a href="#" onclick="doAjaxCall('forcePostProcess',$(this))" data-success="Post-Processor is being loaded" data-error="Error during Post-Processing"><i class="fa fa-wrench fa-fw"></i> Force Post-Process Albums in Download Folder</a>
|
<div id="post_process">
|
||||||
|
<a href="#" class="btnOpenDialog"><i class="fa fa-wrench fa-fw"></i> Force Post-Process Albums in Download Folder</a>
|
||||||
|
</div>
|
||||||
</div>
|
</div>
|
||||||
</fieldset>
|
</fieldset>
|
||||||
<form action="forcePostProcess" method="GET">
|
<fieldset>
|
||||||
<fieldset>
|
<div class="row" id="post_process_alternate">
|
||||||
<div class="row">
|
<label>Force Post-Process Albums in Alternate Folder</label>
|
||||||
<label>Force Post-Process Albums in Alternate Folder</label>
|
<input type="text" value="" name="dir" id="dir" size="50" />
|
||||||
<input type="text" value="" name="dir" id="dir" size="50" /><input type="submit" />
|
<input type="button" class="btnOpenDialog" value="Submit" />
|
||||||
</div>
|
</div>
|
||||||
</fieldset>
|
</fieldset>
|
||||||
|
|
||||||
</form>
|
<fieldset>
|
||||||
<form action="forcePostProcess" method="GET">
|
<div class="row" id="post_process_single">
|
||||||
<fieldset>
|
<label>Post-Process Single Folder</label>
|
||||||
<div class="row">
|
<input type="text" value="" name="album_dir" id="album_dir" size="50" />
|
||||||
<label>Post-Process Single Folder</label>
|
<input type="button" class="btnOpenDialog" value="Submit" />
|
||||||
<input type="text" value="" name="album_dir" id="album_dir" size="50" /><input type="submit" />
|
</div>
|
||||||
</div>
|
</fieldset>
|
||||||
</fieldset>
|
|
||||||
|
|
||||||
</form>
|
|
||||||
|
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
@@ -177,8 +176,7 @@
|
|||||||
</fieldset>
|
</fieldset>
|
||||||
|
|
||||||
</div>
|
</div>
|
||||||
|
<div id="dialog-confirm"></div>
|
||||||
|
|
||||||
|
|
||||||
</div>
|
</div>
|
||||||
</%def>
|
</%def>
|
||||||
@@ -188,6 +186,55 @@
|
|||||||
$('#autoadd').append('<input type="hidden" name="scan" value=1 />');
|
$('#autoadd').append('<input type="hidden" name="scan" value=1 />');
|
||||||
};
|
};
|
||||||
|
|
||||||
|
function fnOpenNormalDialog(id) {
|
||||||
|
if (id === "post_process"){
|
||||||
|
var url = "forcePostProcess?"
|
||||||
|
}
|
||||||
|
if (id === "post_process_alternate"){
|
||||||
|
var dir = $('#dir').val();
|
||||||
|
var url = "forcePostProcess?dir=" + dir + "&"
|
||||||
|
}
|
||||||
|
if (id === "post_process_single"){
|
||||||
|
var dir = $('#album_dir').val();
|
||||||
|
var url = "forcePostProcess?album_dir=" + dir + "&"
|
||||||
|
}
|
||||||
|
var t = $('<a data-success="Post-Processor is being loaded" data-error="Error during Post-Processing">');
|
||||||
|
$("#dialog-confirm").html("Do you want to keep the original folder(s) after post-processing? If you click no, the folders still might be kept depending on your global settings");
|
||||||
|
|
||||||
|
// Define the Dialog and its properties.
|
||||||
|
$("#dialog-confirm").dialog({
|
||||||
|
resizable: false,
|
||||||
|
modal: true,
|
||||||
|
title: "Keep original folder(s)?",
|
||||||
|
height: 170,
|
||||||
|
width: 400,
|
||||||
|
buttons: {
|
||||||
|
"Yes": function () {
|
||||||
|
$(this).dialog('close');
|
||||||
|
doAjaxCall(url + "keep_original_folder=True", t);
|
||||||
|
},
|
||||||
|
"No": function () {
|
||||||
|
$(this).dialog('close');
|
||||||
|
doAjaxCall(url + "keep_original_folder=False", t);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
$('.btnOpenDialog').click(function(e){
|
||||||
|
e.preventDefault();
|
||||||
|
var parentId = $(this).closest('div').prop('id');
|
||||||
|
fnOpenNormalDialog(parentId);
|
||||||
|
});
|
||||||
|
|
||||||
|
function callback(value) {
|
||||||
|
if (value) {
|
||||||
|
alert("Confirmed");
|
||||||
|
} else {
|
||||||
|
alert("Rejected");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
function initThisPage() {
|
function initThisPage() {
|
||||||
$('#manage_albums').click(function() {
|
$('#manage_albums').click(function() {
|
||||||
$('#dialog').dialog();
|
$('#dialog').dialog();
|
||||||
|
|||||||
@@ -353,7 +353,7 @@ def dbcheck():
|
|||||||
conn = sqlite3.connect(DB_FILE)
|
conn = sqlite3.connect(DB_FILE)
|
||||||
c = conn.cursor()
|
c = conn.cursor()
|
||||||
c.execute(
|
c.execute(
|
||||||
'CREATE TABLE IF NOT EXISTS artists (ArtistID TEXT UNIQUE, ArtistName TEXT, ArtistSortName TEXT, DateAdded TEXT, Status TEXT, IncludeExtras INTEGER, LatestAlbum TEXT, ReleaseDate TEXT, AlbumID TEXT, HaveTracks INTEGER, TotalTracks INTEGER, LastUpdated TEXT, ArtworkURL TEXT, ThumbURL TEXT, Extras TEXT, Type TEXT)')
|
'CREATE TABLE IF NOT EXISTS artists (ArtistID TEXT UNIQUE, ArtistName TEXT, ArtistSortName TEXT, DateAdded TEXT, Status TEXT, IncludeExtras INTEGER, LatestAlbum TEXT, ReleaseDate TEXT, AlbumID TEXT, HaveTracks INTEGER, TotalTracks INTEGER, LastUpdated TEXT, ArtworkURL TEXT, ThumbURL TEXT, Extras TEXT, Type TEXT, MetaCritic TEXT)')
|
||||||
# ReleaseFormat here means CD,Digital,Vinyl, etc. If using the default
|
# ReleaseFormat here means CD,Digital,Vinyl, etc. If using the default
|
||||||
# Headphones hybrid release, ReleaseID will equal AlbumID (AlbumID is
|
# Headphones hybrid release, ReleaseID will equal AlbumID (AlbumID is
|
||||||
# releasegroup id)
|
# releasegroup id)
|
||||||
@@ -599,6 +599,11 @@ def dbcheck():
|
|||||||
except sqlite3.OperationalError:
|
except sqlite3.OperationalError:
|
||||||
c.execute('ALTER TABLE artists ADD COLUMN Type TEXT DEFAULT NULL')
|
c.execute('ALTER TABLE artists ADD COLUMN Type TEXT DEFAULT NULL')
|
||||||
|
|
||||||
|
try:
|
||||||
|
c.execute('SELECT MetaCritic from artists')
|
||||||
|
except sqlite3.OperationalError:
|
||||||
|
c.execute('ALTER TABLE artists ADD COLUMN MetaCritic TEXT DEFAULT NULL')
|
||||||
|
|
||||||
conn.commit()
|
conn.commit()
|
||||||
c.close()
|
c.close()
|
||||||
|
|
||||||
|
|||||||
+9
-3
@@ -21,8 +21,8 @@ import json
|
|||||||
cmd_list = ['getIndex', 'getArtist', 'getAlbum', 'getUpcoming', 'getWanted', 'getSnatched', 'getSimilar', 'getHistory', 'getLogs',
|
cmd_list = ['getIndex', 'getArtist', 'getAlbum', 'getUpcoming', 'getWanted', 'getSnatched', 'getSimilar', 'getHistory', 'getLogs',
|
||||||
'findArtist', 'findAlbum', 'addArtist', 'delArtist', 'pauseArtist', 'resumeArtist', 'refreshArtist',
|
'findArtist', 'findAlbum', 'addArtist', 'delArtist', 'pauseArtist', 'resumeArtist', 'refreshArtist',
|
||||||
'addAlbum', 'queueAlbum', 'unqueueAlbum', 'forceSearch', 'forceProcess', 'forceActiveArtistsUpdate',
|
'addAlbum', 'queueAlbum', 'unqueueAlbum', 'forceSearch', 'forceProcess', 'forceActiveArtistsUpdate',
|
||||||
'getVersion', 'checkGithub','shutdown', 'restart', 'update', 'getArtistArt', 'getAlbumArt',
|
'getVersion', 'checkGithub', 'shutdown', 'restart', 'update', 'getArtistArt', 'getAlbumArt',
|
||||||
'getArtistInfo', 'getAlbumInfo', 'getArtistThumb', 'getAlbumThumb',
|
'getArtistInfo', 'getAlbumInfo', 'getArtistThumb', 'getAlbumThumb', 'clearLogs',
|
||||||
'choose_specific_download', 'download_specific_release']
|
'choose_specific_download', 'download_specific_release']
|
||||||
|
|
||||||
|
|
||||||
@@ -176,7 +176,13 @@ class Api(object):
|
|||||||
return
|
return
|
||||||
|
|
||||||
def _getLogs(self, **kwargs):
|
def _getLogs(self, **kwargs):
|
||||||
pass
|
self.data = headphones.LOG_LIST
|
||||||
|
return
|
||||||
|
|
||||||
|
def _clearLogs(self, **kwargs):
|
||||||
|
headphones.LOG_LIST = []
|
||||||
|
self.data = 'Cleared log'
|
||||||
|
return
|
||||||
|
|
||||||
def _findArtist(self, **kwargs):
|
def _findArtist(self, **kwargs):
|
||||||
if 'name' not in kwargs:
|
if 'name' not in kwargs:
|
||||||
|
|||||||
+32
-1
@@ -53,6 +53,7 @@ _CONFIG_DEFINITIONS = {
|
|||||||
'DELETE_LOSSLESS_FILES': (int, 'General', 1),
|
'DELETE_LOSSLESS_FILES': (int, 'General', 1),
|
||||||
'DESTINATION_DIR': (str, 'General', ''),
|
'DESTINATION_DIR': (str, 'General', ''),
|
||||||
'DETECT_BITRATE': (int, 'General', 0),
|
'DETECT_BITRATE': (int, 'General', 0),
|
||||||
|
'DO_NOT_PROCESS_UNMATCHED': (int, 'General', 0),
|
||||||
'DOWNLOAD_DIR': (str, 'General', ''),
|
'DOWNLOAD_DIR': (str, 'General', ''),
|
||||||
'DOWNLOAD_SCAN_INTERVAL': (int, 'General', 5),
|
'DOWNLOAD_SCAN_INTERVAL': (int, 'General', 5),
|
||||||
'DOWNLOAD_TORRENT_DIR': (str, 'General', ''),
|
'DOWNLOAD_TORRENT_DIR': (str, 'General', ''),
|
||||||
@@ -81,6 +82,7 @@ _CONFIG_DEFINITIONS = {
|
|||||||
'ENCODER_PATH': (str, 'General', ''),
|
'ENCODER_PATH': (str, 'General', ''),
|
||||||
'EXTRAS': (str, 'General', ''),
|
'EXTRAS': (str, 'General', ''),
|
||||||
'EXTRA_NEWZNABS': (list, 'Newznab', ''),
|
'EXTRA_NEWZNABS': (list, 'Newznab', ''),
|
||||||
|
'EXTRA_TORZNABS': (list, 'Torznab', ''),
|
||||||
'FILE_FORMAT': (str, 'General', 'Track Artist - Album [Year] - Title'),
|
'FILE_FORMAT': (str, 'General', 'Track Artist - Album [Year] - Title'),
|
||||||
'FILE_PERMISSIONS': (str, 'General', '0644'),
|
'FILE_PERMISSIONS': (str, 'General', '0644'),
|
||||||
'FILE_UNDERSCORES': (int, 'General', 0),
|
'FILE_UNDERSCORES': (int, 'General', 0),
|
||||||
@@ -99,7 +101,7 @@ _CONFIG_DEFINITIONS = {
|
|||||||
'HPUSER': (str, 'General', ''),
|
'HPUSER': (str, 'General', ''),
|
||||||
'HTTPS_CERT': (str, 'General', ''),
|
'HTTPS_CERT': (str, 'General', ''),
|
||||||
'HTTPS_KEY': (str, 'General', ''),
|
'HTTPS_KEY': (str, 'General', ''),
|
||||||
'HTTP_HOST': (str, 'General', '0.0.0.0'),
|
'HTTP_HOST': (str, 'General', 'localhost'),
|
||||||
'HTTP_PASSWORD': (str, 'General', ''),
|
'HTTP_PASSWORD': (str, 'General', ''),
|
||||||
'HTTP_PORT': (int, 'General', 8181),
|
'HTTP_PORT': (int, 'General', 8181),
|
||||||
'HTTP_PROXY': (int, 'General', 0),
|
'HTTP_PROXY': (int, 'General', 0),
|
||||||
@@ -154,6 +156,7 @@ _CONFIG_DEFINITIONS = {
|
|||||||
'NZBSORG_HASH': (str, 'NZBsorg', ''),
|
'NZBSORG_HASH': (str, 'NZBsorg', ''),
|
||||||
'NZBSORG_UID': (str, 'NZBsorg', ''),
|
'NZBSORG_UID': (str, 'NZBsorg', ''),
|
||||||
'NZB_DOWNLOADER': (int, 'General', 0),
|
'NZB_DOWNLOADER': (int, 'General', 0),
|
||||||
|
'OFFICIAL_RELEASES_ONLY': (int, 'General', 0),
|
||||||
'OMGWTFNZBS': (int, 'omgwtfnzbs', 0),
|
'OMGWTFNZBS': (int, 'omgwtfnzbs', 0),
|
||||||
'OMGWTFNZBS_APIKEY': (str, 'omgwtfnzbs', ''),
|
'OMGWTFNZBS_APIKEY': (str, 'omgwtfnzbs', ''),
|
||||||
'OMGWTFNZBS_UID': (str, 'omgwtfnzbs', ''),
|
'OMGWTFNZBS_UID': (str, 'omgwtfnzbs', ''),
|
||||||
@@ -175,6 +178,7 @@ _CONFIG_DEFINITIONS = {
|
|||||||
'PLEX_SERVER_HOST': (str, 'Plex', ''),
|
'PLEX_SERVER_HOST': (str, 'Plex', ''),
|
||||||
'PLEX_UPDATE': (int, 'Plex', 0),
|
'PLEX_UPDATE': (int, 'Plex', 0),
|
||||||
'PLEX_USERNAME': (str, 'Plex', ''),
|
'PLEX_USERNAME': (str, 'Plex', ''),
|
||||||
|
'PLEX_TOKEN': (str, 'Plex', ''),
|
||||||
'PREFERRED_BITRATE': (str, 'General', ''),
|
'PREFERRED_BITRATE': (str, 'General', ''),
|
||||||
'PREFERRED_BITRATE_ALLOW_LOSSLESS': (int, 'General', 0),
|
'PREFERRED_BITRATE_ALLOW_LOSSLESS': (int, 'General', 0),
|
||||||
'PREFERRED_BITRATE_HIGH_BUFFER': (int, 'General', 0),
|
'PREFERRED_BITRATE_HIGH_BUFFER': (int, 'General', 0),
|
||||||
@@ -200,6 +204,7 @@ _CONFIG_DEFINITIONS = {
|
|||||||
'PUSHOVER_PRIORITY': (int, 'Pushover', 0),
|
'PUSHOVER_PRIORITY': (int, 'Pushover', 0),
|
||||||
'RENAME_FILES': (int, 'General', 0),
|
'RENAME_FILES': (int, 'General', 0),
|
||||||
'REPLACE_EXISTING_FOLDERS': (int, 'General', 0),
|
'REPLACE_EXISTING_FOLDERS': (int, 'General', 0),
|
||||||
|
'KEEP_ORIGINAL_FOLDER': (int, 'General', 0),
|
||||||
'REQUIRED_WORDS': (str, 'General', ''),
|
'REQUIRED_WORDS': (str, 'General', ''),
|
||||||
'RUTRACKER': (int, 'Rutracker', 0),
|
'RUTRACKER': (int, 'Rutracker', 0),
|
||||||
'RUTRACKER_PASSWORD': (str, 'Rutracker', ''),
|
'RUTRACKER_PASSWORD': (str, 'Rutracker', ''),
|
||||||
@@ -216,6 +221,8 @@ _CONFIG_DEFINITIONS = {
|
|||||||
'SONGKICK_ENABLED': (int, 'Songkick', 1),
|
'SONGKICK_ENABLED': (int, 'Songkick', 1),
|
||||||
'SONGKICK_FILTER_ENABLED': (int, 'Songkick', 0),
|
'SONGKICK_FILTER_ENABLED': (int, 'Songkick', 0),
|
||||||
'SONGKICK_LOCATION': (str, 'Songkick', ''),
|
'SONGKICK_LOCATION': (str, 'Songkick', ''),
|
||||||
|
'STRIKE': (int, 'Strike', 0),
|
||||||
|
'STRIKE_RATIO': (str, 'Strike', ''),
|
||||||
'SUBSONIC_ENABLED': (int, 'Subsonic', 0),
|
'SUBSONIC_ENABLED': (int, 'Subsonic', 0),
|
||||||
'SUBSONIC_HOST': (str, 'Subsonic', ''),
|
'SUBSONIC_HOST': (str, 'Subsonic', ''),
|
||||||
'SUBSONIC_PASSWORD': (str, 'Subsonic', ''),
|
'SUBSONIC_PASSWORD': (str, 'Subsonic', ''),
|
||||||
@@ -224,6 +231,10 @@ _CONFIG_DEFINITIONS = {
|
|||||||
'TORRENTBLACKHOLE_DIR': (str, 'General', ''),
|
'TORRENTBLACKHOLE_DIR': (str, 'General', ''),
|
||||||
'TORRENT_DOWNLOADER': (int, 'General', 0),
|
'TORRENT_DOWNLOADER': (int, 'General', 0),
|
||||||
'TORRENT_REMOVAL_INTERVAL': (int, 'General', 720),
|
'TORRENT_REMOVAL_INTERVAL': (int, 'General', 720),
|
||||||
|
'TORZNAB': (int, 'Torznab', 0),
|
||||||
|
'TORZNAB_APIKEY': (str, 'Torznab', ''),
|
||||||
|
'TORZNAB_ENABLED': (int, 'Torznab', 1),
|
||||||
|
'TORZNAB_HOST': (str, 'Torznab', ''),
|
||||||
'TRANSMISSION_HOST': (str, 'Transmission', ''),
|
'TRANSMISSION_HOST': (str, 'Transmission', ''),
|
||||||
'TRANSMISSION_PASSWORD': (str, 'Transmission', ''),
|
'TRANSMISSION_PASSWORD': (str, 'Transmission', ''),
|
||||||
'TRANSMISSION_USERNAME': (str, 'Transmission', ''),
|
'TRANSMISSION_USERNAME': (str, 'Transmission', ''),
|
||||||
@@ -239,6 +250,7 @@ _CONFIG_DEFINITIONS = {
|
|||||||
'UTORRENT_PASSWORD': (str, 'uTorrent', ''),
|
'UTORRENT_PASSWORD': (str, 'uTorrent', ''),
|
||||||
'UTORRENT_USERNAME': (str, 'uTorrent', ''),
|
'UTORRENT_USERNAME': (str, 'uTorrent', ''),
|
||||||
'VERIFY_SSL_CERT': (bool_int, 'Advanced', 1),
|
'VERIFY_SSL_CERT': (bool_int, 'Advanced', 1),
|
||||||
|
'WAIT_UNTIL_RELEASE_DATE' : (int, 'General', 0),
|
||||||
'WAFFLES': (int, 'Waffles', 0),
|
'WAFFLES': (int, 'Waffles', 0),
|
||||||
'WAFFLES_PASSKEY': (str, 'Waffles', ''),
|
'WAFFLES_PASSKEY': (str, 'Waffles', ''),
|
||||||
'WAFFLES_RATIO': (str, 'Waffles', ''),
|
'WAFFLES_RATIO': (str, 'Waffles', ''),
|
||||||
@@ -347,6 +359,25 @@ class Config(object):
|
|||||||
extra_newznabs.append(item)
|
extra_newznabs.append(item)
|
||||||
self.EXTRA_NEWZNABS = extra_newznabs
|
self.EXTRA_NEWZNABS = extra_newznabs
|
||||||
|
|
||||||
|
def get_extra_torznabs(self):
|
||||||
|
""" Return the extra torznab tuples """
|
||||||
|
extra_torznabs = list(
|
||||||
|
itertools.izip(*[itertools.islice(self.EXTRA_TORZNABS, i, None, 3)
|
||||||
|
for i in range(3)])
|
||||||
|
)
|
||||||
|
return extra_torznabs
|
||||||
|
|
||||||
|
def clear_extra_torznabs(self):
|
||||||
|
""" Forget about the configured extra torznabs """
|
||||||
|
self.EXTRA_TORZNABS = []
|
||||||
|
|
||||||
|
def add_extra_torznab(self, torznab):
|
||||||
|
""" Add a new extra torznab """
|
||||||
|
extra_torznabs = self.EXTRA_TORZNABS
|
||||||
|
for item in torznab:
|
||||||
|
extra_torznabs.append(item)
|
||||||
|
self.EXTRA_TORZNABS = extra_torznabs
|
||||||
|
|
||||||
def __getattr__(self, name):
|
def __getattr__(self, name):
|
||||||
"""
|
"""
|
||||||
Returns something from the ini unless it is a real property
|
Returns something from the ini unless it is a real property
|
||||||
|
|||||||
@@ -149,7 +149,7 @@ def get_age(date):
|
|||||||
|
|
||||||
try:
|
try:
|
||||||
days_old = int(split_date[0]) * 365 + int(split_date[1]) * 30 + int(split_date[2])
|
days_old = int(split_date[0]) * 365 + int(split_date[1]) * 30 + int(split_date[2])
|
||||||
except IndexError:
|
except (IndexError,ValueError):
|
||||||
days_old = False
|
days_old = False
|
||||||
|
|
||||||
return days_old
|
return days_old
|
||||||
|
|||||||
@@ -496,7 +496,7 @@ def addArtisttoDB(artistid, extrasonly=False, forcefull=False, type="artist"):
|
|||||||
cache.getThumb(ArtistID=artistid)
|
cache.getThumb(ArtistID=artistid)
|
||||||
|
|
||||||
logger.info(u"Fetching Metacritic reviews for: %s" % artist['artist_name'])
|
logger.info(u"Fetching Metacritic reviews for: %s" % artist['artist_name'])
|
||||||
metacritic.update(artist['artist_name'], artist['releasegroups'])
|
metacritic.update(artistid, artist['artist_name'], artist['releasegroups'])
|
||||||
|
|
||||||
if errors:
|
if errors:
|
||||||
logger.info("[%s] Finished updating artist: %s but with errors, so not marking it as updated in the database" % (artist['artist_name'], artist['artist_name']))
|
logger.info("[%s] Finished updating artist: %s but with errors, so not marking it as updated in the database" % (artist['artist_name'], artist['artist_name']))
|
||||||
|
|||||||
@@ -79,7 +79,7 @@ def getSimilar():
|
|||||||
try:
|
try:
|
||||||
artist_mbid = artist["mbid"]
|
artist_mbid = artist["mbid"]
|
||||||
artist_name = artist["name"]
|
artist_name = artist["name"]
|
||||||
except TypeError:
|
except KeyError:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
if not any(artist_mbid in x for x in results):
|
if not any(artist_mbid in x for x in results):
|
||||||
@@ -116,7 +116,7 @@ def getArtists():
|
|||||||
return
|
return
|
||||||
|
|
||||||
logger.info("Fetching artists from Last.FM for username: %s", headphones.CONFIG.LASTFM_USERNAME)
|
logger.info("Fetching artists from Last.FM for username: %s", headphones.CONFIG.LASTFM_USERNAME)
|
||||||
data = request_lastfm("library.getartists", limit=10000, user=headphones.CONFIG.LASTFM_USERNAME)
|
data = request_lastfm("library.getartists", limit=1000, user=headphones.CONFIG.LASTFM_USERNAME)
|
||||||
|
|
||||||
if data and "artists" in data:
|
if data and "artists" in data:
|
||||||
artistlist = []
|
artistlist = []
|
||||||
|
|||||||
@@ -23,7 +23,7 @@ from headphones import db, logger, helpers, importer, lastfm
|
|||||||
# You can scan a single directory and append it to the current library by
|
# You can scan a single directory and append it to the current library by
|
||||||
# specifying append=True, ArtistID and ArtistName.
|
# specifying append=True, ArtistID and ArtistName.
|
||||||
def libraryScan(dir=None, append=False, ArtistID=None, ArtistName=None,
|
def libraryScan(dir=None, append=False, ArtistID=None, ArtistName=None,
|
||||||
cron=False):
|
cron=False, artistScan=False):
|
||||||
|
|
||||||
if cron and not headphones.CONFIG.LIBRARYSCAN:
|
if cron and not headphones.CONFIG.LIBRARYSCAN:
|
||||||
return
|
return
|
||||||
@@ -36,7 +36,7 @@ def libraryScan(dir=None, append=False, ArtistID=None, ArtistName=None,
|
|||||||
|
|
||||||
# If we're appending a dir, it's coming from the post processor which is
|
# If we're appending a dir, it's coming from the post processor which is
|
||||||
# already bytestring
|
# already bytestring
|
||||||
if not append:
|
if not append or artistScan:
|
||||||
dir = dir.encode(headphones.SYS_ENCODING)
|
dir = dir.encode(headphones.SYS_ENCODING)
|
||||||
|
|
||||||
if not os.path.isdir(dir):
|
if not os.path.isdir(dir):
|
||||||
@@ -53,7 +53,7 @@ def libraryScan(dir=None, append=False, ArtistID=None, ArtistName=None,
|
|||||||
tracks = myDB.select('SELECT Location from alltracks WHERE Location IS NOT NULL UNION SELECT Location from tracks WHERE Location IS NOT NULL')
|
tracks = myDB.select('SELECT Location from alltracks WHERE Location IS NOT NULL UNION SELECT Location from tracks WHERE Location IS NOT NULL')
|
||||||
|
|
||||||
for track in tracks:
|
for track in tracks:
|
||||||
encoded_track_string = track['Location'].encode(headphones.SYS_ENCODING)
|
encoded_track_string = track['Location'].encode(headphones.SYS_ENCODING, 'replace')
|
||||||
if not os.path.isfile(encoded_track_string):
|
if not os.path.isfile(encoded_track_string):
|
||||||
myDB.action('UPDATE tracks SET Location=?, BitRate=?, Format=? WHERE Location=?', [None, None, None, track['Location']])
|
myDB.action('UPDATE tracks SET Location=?, BitRate=?, Format=? WHERE Location=?', [None, None, None, track['Location']])
|
||||||
myDB.action('UPDATE alltracks SET Location=?, BitRate=?, Format=? WHERE Location=?', [None, None, None, track['Location']])
|
myDB.action('UPDATE alltracks SET Location=?, BitRate=?, Format=? WHERE Location=?', [None, None, None, track['Location']])
|
||||||
@@ -287,7 +287,7 @@ def libraryScan(dir=None, append=False, ArtistID=None, ArtistName=None,
|
|||||||
|
|
||||||
logger.info('Completed matching tracks from directory: %s' % dir.decode(headphones.SYS_ENCODING, 'replace'))
|
logger.info('Completed matching tracks from directory: %s' % dir.decode(headphones.SYS_ENCODING, 'replace'))
|
||||||
|
|
||||||
if not append:
|
if not append or artistScan:
|
||||||
logger.info('Updating scanned artist track counts')
|
logger.info('Updating scanned artist track counts')
|
||||||
|
|
||||||
# Clean up the new artist list
|
# Clean up the new artist list
|
||||||
@@ -334,7 +334,7 @@ def libraryScan(dir=None, append=False, ArtistID=None, ArtistName=None,
|
|||||||
for artist in artist_list:
|
for artist in artist_list:
|
||||||
myDB.action('INSERT OR IGNORE INTO newartists VALUES (?)', [artist])
|
myDB.action('INSERT OR IGNORE INTO newartists VALUES (?)', [artist])
|
||||||
|
|
||||||
if headphones.CONFIG.DETECT_BITRATE:
|
if headphones.CONFIG.DETECT_BITRATE and bitrates:
|
||||||
headphones.CONFIG.PREFERRED_BITRATE = sum(bitrates) / len(bitrates) / 1000
|
headphones.CONFIG.PREFERRED_BITRATE = sum(bitrates) / len(bitrates) / 1000
|
||||||
|
|
||||||
else:
|
else:
|
||||||
@@ -346,6 +346,8 @@ def libraryScan(dir=None, append=False, ArtistID=None, ArtistName=None,
|
|||||||
|
|
||||||
if not append:
|
if not append:
|
||||||
update_album_status()
|
update_album_status()
|
||||||
|
|
||||||
|
if not append and not artistScan:
|
||||||
lastfm.getSimilar()
|
lastfm.getSimilar()
|
||||||
|
|
||||||
logger.info('Library scan complete')
|
logger.info('Library scan complete')
|
||||||
@@ -374,7 +376,7 @@ def update_album_status(AlbumID=None):
|
|||||||
album_completion = 0
|
album_completion = 0
|
||||||
logger.info('Album %s does not have any tracks in database' % album['AlbumTitle'])
|
logger.info('Album %s does not have any tracks in database' % album['AlbumTitle'])
|
||||||
|
|
||||||
if album_completion >= headphones.CONFIG.ALBUM_COMPLETION_PCT and album['Status'] == 'Skipped':
|
if album_completion >= headphones.CONFIG.ALBUM_COMPLETION_PCT:
|
||||||
new_album_status = "Downloaded"
|
new_album_status = "Downloaded"
|
||||||
|
|
||||||
# I don't think we want to change Downloaded->Skipped.....
|
# I don't think we want to change Downloaded->Skipped.....
|
||||||
|
|||||||
+7
-5
@@ -51,7 +51,7 @@ def startmb():
|
|||||||
mbpass = headphones.CONFIG.CUSTOMPASS
|
mbpass = headphones.CONFIG.CUSTOMPASS
|
||||||
sleepytime = int(headphones.CONFIG.CUSTOMSLEEP)
|
sleepytime = int(headphones.CONFIG.CUSTOMSLEEP)
|
||||||
elif headphones.CONFIG.MIRROR == "headphones":
|
elif headphones.CONFIG.MIRROR == "headphones":
|
||||||
mbhost = "144.76.94.239"
|
mbhost = "codeshy.com"
|
||||||
mbport = 8181
|
mbport = 8181
|
||||||
mbuser = headphones.CONFIG.HPUSER
|
mbuser = headphones.CONFIG.HPUSER
|
||||||
mbpass = headphones.CONFIG.HPPASS
|
mbpass = headphones.CONFIG.HPPASS
|
||||||
@@ -466,6 +466,11 @@ def get_new_releases(rgid, includeExtras=False, forcefull=False):
|
|||||||
|
|
||||||
myDB = db.DBConnection()
|
myDB = db.DBConnection()
|
||||||
results = []
|
results = []
|
||||||
|
|
||||||
|
release_status = "official"
|
||||||
|
if includeExtras and not headphones.CONFIG.OFFICIAL_RELEASES_ONLY:
|
||||||
|
release_status = []
|
||||||
|
|
||||||
try:
|
try:
|
||||||
limit = 100
|
limit = 100
|
||||||
newResults = None
|
newResults = None
|
||||||
@@ -474,6 +479,7 @@ def get_new_releases(rgid, includeExtras=False, forcefull=False):
|
|||||||
newResults = musicbrainzngs.browse_releases(
|
newResults = musicbrainzngs.browse_releases(
|
||||||
release_group=rgid,
|
release_group=rgid,
|
||||||
includes=['artist-credits', 'labels', 'recordings', 'release-groups', 'media'],
|
includes=['artist-credits', 'labels', 'recordings', 'release-groups', 'media'],
|
||||||
|
release_status = release_status,
|
||||||
limit=limit,
|
limit=limit,
|
||||||
offset=len(results))
|
offset=len(results))
|
||||||
if 'release-list' not in newResults:
|
if 'release-list' not in newResults:
|
||||||
@@ -513,10 +519,6 @@ def get_new_releases(rgid, includeExtras=False, forcefull=False):
|
|||||||
num_new_releases = 0
|
num_new_releases = 0
|
||||||
|
|
||||||
for releasedata in results:
|
for releasedata in results:
|
||||||
#releasedata.get will return None if it doesn't have a status
|
|
||||||
#all official releases should have the Official status included
|
|
||||||
if not includeExtras and releasedata.get('status') != 'Official':
|
|
||||||
continue
|
|
||||||
|
|
||||||
release = {}
|
release = {}
|
||||||
rel_id_check = releasedata['id']
|
rel_id_check = releasedata['id']
|
||||||
|
|||||||
+42
-12
@@ -14,11 +14,13 @@
|
|||||||
# along with Headphones. If not, see <http://www.gnu.org/licenses/>.
|
# along with Headphones. If not, see <http://www.gnu.org/licenses/>.
|
||||||
|
|
||||||
import re
|
import re
|
||||||
|
import json
|
||||||
import headphones
|
import headphones
|
||||||
|
|
||||||
from headphones import db, helpers, logger, request
|
from headphones import db, helpers, logger, request
|
||||||
|
from headphones.common import USER_AGENT
|
||||||
|
|
||||||
def update(artist_name,release_groups):
|
def update(artistid, artist_name,release_groups):
|
||||||
""" Pretty simple and crude function to find the artist page on metacritic,
|
""" Pretty simple and crude function to find the artist page on metacritic,
|
||||||
then parse that page to get critic & user scores for albums"""
|
then parse that page to get critic & user scores for albums"""
|
||||||
|
|
||||||
@@ -31,28 +33,56 @@ def update(artist_name,release_groups):
|
|||||||
|
|
||||||
mc_artist_name = mc_artist_name.replace(" ","-")
|
mc_artist_name = mc_artist_name.replace(" ","-")
|
||||||
|
|
||||||
|
headers = {'User-Agent': 'Mozilla/5.0 (Windows NT 6.3; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/41.0.2243.2 Safari/537.36'}
|
||||||
|
|
||||||
url = "http://www.metacritic.com/person/" + mc_artist_name + "?filter-options=music&sort_options=date&num_items=100"
|
url = "http://www.metacritic.com/person/" + mc_artist_name + "?filter-options=music&sort_options=date&num_items=100"
|
||||||
|
|
||||||
res = request.request_soup(url, parser='html.parser')
|
res = request.request_soup(url, headers=headers)
|
||||||
|
|
||||||
|
rows = None
|
||||||
|
|
||||||
try:
|
try:
|
||||||
rows = res.tbody.find_all('tr')
|
table = res.find("table", class_="credits person_credits")
|
||||||
|
rows = table.tbody.find_all('tr')
|
||||||
except:
|
except:
|
||||||
logger.info("Unable to get metacritic scores for: %s" % artist_name)
|
logger.info("Unable to get metacritic scores for: %s" % artist_name)
|
||||||
return
|
|
||||||
|
|
||||||
myDB = db.DBConnection()
|
myDB = db.DBConnection()
|
||||||
|
artist = myDB.action('SELECT * FROM artists WHERE ArtistID=?', [artistid]).fetchone()
|
||||||
|
|
||||||
for row in rows:
|
score_list = []
|
||||||
title = row.a.string
|
|
||||||
|
# If we couldn't get anything from MetaCritic for whatever reason,
|
||||||
|
# let's try to load scores from the db
|
||||||
|
if not rows:
|
||||||
|
if artist['MetaCritic']:
|
||||||
|
score_list = json.loads(artist['MetaCritic'])
|
||||||
|
else:
|
||||||
|
return
|
||||||
|
|
||||||
|
# If we did get scores, let's update the db with them
|
||||||
|
else:
|
||||||
|
for row in rows:
|
||||||
|
title = row.a.string
|
||||||
|
scores = row.find_all("span")
|
||||||
|
critic_score = scores[0].string
|
||||||
|
user_score = scores[1].string
|
||||||
|
score_dict = {'title':title,'critic_score':critic_score,'user_score':user_score}
|
||||||
|
score_list.append(score_dict)
|
||||||
|
|
||||||
|
# Save scores to the database
|
||||||
|
controlValueDict = {"ArtistID": artistid}
|
||||||
|
newValueDict = {'MetaCritic':json.dumps(score_list)}
|
||||||
|
myDB.upsert("artists", newValueDict, controlValueDict)
|
||||||
|
|
||||||
|
for score in score_list:
|
||||||
|
title = score['title']
|
||||||
|
# Iterate through the release groups we got passed to see if we can find
|
||||||
|
# a match
|
||||||
for rg in release_groups:
|
for rg in release_groups:
|
||||||
if rg['title'].lower() == title.lower():
|
if rg['title'].lower() == title.lower():
|
||||||
scores = row.find_all("span")
|
critic_score = score['critic_score']
|
||||||
critic_score = scores[0].string
|
user_score = score['user_score']
|
||||||
user_score = scores[1].string
|
|
||||||
controlValueDict = {"AlbumID": rg['id']}
|
controlValueDict = {"AlbumID": rg['id']}
|
||||||
newValueDict = {'CriticScore':critic_score,'UserScore':user_score}
|
newValueDict = {'CriticScore':critic_score,'UserScore':user_score}
|
||||||
myDB.upsert("albums", newValueDict, controlValueDict)
|
myDB.upsert("albums", newValueDict, controlValueDict)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
+65
-93
@@ -316,34 +316,32 @@ class Plex(object):
|
|||||||
self.client_hosts = headphones.CONFIG.PLEX_CLIENT_HOST
|
self.client_hosts = headphones.CONFIG.PLEX_CLIENT_HOST
|
||||||
self.username = headphones.CONFIG.PLEX_USERNAME
|
self.username = headphones.CONFIG.PLEX_USERNAME
|
||||||
self.password = headphones.CONFIG.PLEX_PASSWORD
|
self.password = headphones.CONFIG.PLEX_PASSWORD
|
||||||
|
self.token = headphones.CONFIG.PLEX_TOKEN
|
||||||
|
|
||||||
def _sendhttp(self, host, command):
|
def _sendhttp(self, host, command):
|
||||||
|
|
||||||
username = self.username
|
url = host + '/xbmcCmds/xbmcHttp/?' + command
|
||||||
password = self.password
|
|
||||||
|
|
||||||
url_command = urllib.urlencode(command)
|
if self.password:
|
||||||
|
response = request.request_response(url, auth=(self.username, self.password))
|
||||||
url = host + '/xbmcCmds/xbmcHttp/?' + url_command
|
else:
|
||||||
|
response = request.request_response(url)
|
||||||
req = urllib2.Request(url)
|
|
||||||
|
|
||||||
if password:
|
|
||||||
base64string = base64.encodestring('%s:%s' % (username, password)).replace('\n', '')
|
|
||||||
req.add_header("Authorization", "Basic %s" % base64string)
|
|
||||||
|
|
||||||
logger.info('Plex url: %s' % url)
|
|
||||||
|
|
||||||
try:
|
|
||||||
handle = urllib2.urlopen(req)
|
|
||||||
except Exception as e:
|
|
||||||
logger.warn('Error opening Plex url: %s' % e)
|
|
||||||
return
|
|
||||||
|
|
||||||
response = handle.read().decode(headphones.SYS_ENCODING)
|
|
||||||
|
|
||||||
return response
|
return response
|
||||||
|
|
||||||
|
def _sendjson(self, host, method, params={}):
|
||||||
|
data = [{'id': 0, 'jsonrpc': '2.0', 'method': method, 'params': params}]
|
||||||
|
headers = {'Content-Type': 'application/json'}
|
||||||
|
url = host + '/jsonrpc'
|
||||||
|
|
||||||
|
if self.password:
|
||||||
|
response = request.request_json(url, method="post", data=json.dumps(data), headers=headers, auth=(self.username, self.password))
|
||||||
|
else:
|
||||||
|
response = request.request_json(url, method="post", data=json.dumps(data), headers=headers)
|
||||||
|
|
||||||
|
if response:
|
||||||
|
return response[0]['result']
|
||||||
|
|
||||||
def update(self):
|
def update(self):
|
||||||
|
|
||||||
# From what I read you can't update the music library on a per directory or per path basis
|
# From what I read you can't update the music library on a per directory or per path basis
|
||||||
@@ -354,13 +352,15 @@ class Plex(object):
|
|||||||
for host in hosts:
|
for host in hosts:
|
||||||
logger.info('Sending library update command to Plex Media Server@ ' + host)
|
logger.info('Sending library update command to Plex Media Server@ ' + host)
|
||||||
url = "%s/library/sections" % host
|
url = "%s/library/sections" % host
|
||||||
try:
|
if self.token:
|
||||||
xml_sections = minidom.parse(urllib.urlopen(url))
|
params = {'X-Plex-Token': self.token}
|
||||||
except IOError, e:
|
else:
|
||||||
logger.warn("Error while trying to contact Plex Media Server: %s" % e)
|
params = False
|
||||||
return False
|
|
||||||
|
r = request.request_minidom(url, params=params)
|
||||||
|
|
||||||
|
sections = r.getElementsByTagName('Directory')
|
||||||
|
|
||||||
sections = xml_sections.getElementsByTagName('Directory')
|
|
||||||
if not sections:
|
if not sections:
|
||||||
logger.info(u"Plex Media Server not running on: " + host)
|
logger.info(u"Plex Media Server not running on: " + host)
|
||||||
return False
|
return False
|
||||||
@@ -368,11 +368,7 @@ class Plex(object):
|
|||||||
for s in sections:
|
for s in sections:
|
||||||
if s.getAttribute('type') == "artist":
|
if s.getAttribute('type') == "artist":
|
||||||
url = "%s/library/sections/%s/refresh" % (host, s.getAttribute('key'))
|
url = "%s/library/sections/%s/refresh" % (host, s.getAttribute('key'))
|
||||||
try:
|
request.request_response(url, params=params)
|
||||||
urllib.urlopen(url)
|
|
||||||
except Exception as e:
|
|
||||||
logger.warn("Error updating library section for Plex Media Server: %s" % e)
|
|
||||||
return False
|
|
||||||
|
|
||||||
def notify(self, artist, album, albumartpath):
|
def notify(self, artist, album, albumartpath):
|
||||||
|
|
||||||
@@ -383,17 +379,24 @@ class Plex(object):
|
|||||||
time = "3000" # in ms
|
time = "3000" # in ms
|
||||||
|
|
||||||
for host in hosts:
|
for host in hosts:
|
||||||
logger.info('Sending notification command to Plex Media Server @ ' + host)
|
logger.info('Sending notification command to Plex client @ ' + host)
|
||||||
try:
|
try:
|
||||||
notification = header + "," + message + "," + time + "," + albumartpath
|
version = self._sendjson(host, 'Application.GetProperties', {'properties': ['version']})['version']['major']
|
||||||
notifycommand = {'command': 'ExecBuiltIn', 'parameter': 'Notification(' + notification + ')'}
|
|
||||||
request = self._sendhttp(host, notifycommand)
|
if version < 12: #Eden
|
||||||
|
notification = header + "," + message + "," + time + "," + albumartpath
|
||||||
|
notifycommand = {'command': 'ExecBuiltIn', 'parameter': 'Notification(' + notification + ')'}
|
||||||
|
request = self._sendhttp(host, notifycommand)
|
||||||
|
|
||||||
|
else: #Frodo
|
||||||
|
params = {'title': header, 'message': message, 'displaytime': int(time), 'image': albumartpath}
|
||||||
|
request = self._sendjson(host, 'GUI.ShowNotification', params)
|
||||||
|
|
||||||
if not request:
|
if not request:
|
||||||
raise Exception
|
raise Exception
|
||||||
|
|
||||||
except:
|
except Exception:
|
||||||
logger.warn('Error sending notification request to Plex Media Server')
|
logger.error('Error sending notification request to Plex client @ ' + host)
|
||||||
|
|
||||||
|
|
||||||
class NMA(object):
|
class NMA(object):
|
||||||
@@ -440,52 +443,30 @@ class PUSHBULLET(object):
|
|||||||
self.apikey = headphones.CONFIG.PUSHBULLET_APIKEY
|
self.apikey = headphones.CONFIG.PUSHBULLET_APIKEY
|
||||||
self.deviceid = headphones.CONFIG.PUSHBULLET_DEVICEID
|
self.deviceid = headphones.CONFIG.PUSHBULLET_DEVICEID
|
||||||
|
|
||||||
def conf(self, options):
|
def notify(self, message, status):
|
||||||
return cherrypy.config['config'].get('PUSHBULLET', options)
|
|
||||||
|
|
||||||
def notify(self, message, event):
|
|
||||||
if not headphones.CONFIG.PUSHBULLET_ENABLED:
|
if not headphones.CONFIG.PUSHBULLET_ENABLED:
|
||||||
return
|
return
|
||||||
|
|
||||||
http_handler = HTTPSConnection("api.pushbullet.com")
|
url = "https://api.pushbullet.com/v2/pushes"
|
||||||
|
|
||||||
data = {'type': "note",
|
data = {'type': "note",
|
||||||
'title': "Headphones",
|
'title': "Headphones",
|
||||||
'body': message.encode("utf-8")}
|
'body': message + ': ' + status}
|
||||||
|
|
||||||
http_handler.request("POST",
|
if self.deviceid:
|
||||||
"/v2/pushes",
|
data['device_iden'] = self.deviceid
|
||||||
headers={'Content-type': "application/json",
|
|
||||||
'Authorization': 'Basic %s' % base64.b64encode(headphones.CONFIG.PUSHBULLET_APIKEY + ":")},
|
|
||||||
body=json.dumps(data))
|
|
||||||
response = http_handler.getresponse()
|
|
||||||
request_status = response.status
|
|
||||||
logger.debug(u"PushBullet response status: %r" % request_status)
|
|
||||||
logger.debug(u"PushBullet response headers: %r" % response.getheaders())
|
|
||||||
logger.debug(u"PushBullet response body: %r" % response.read())
|
|
||||||
|
|
||||||
if request_status == 200:
|
headers={'Content-type': "application/json",
|
||||||
logger.info(u"PushBullet notifications sent.")
|
'Authorization': 'Bearer ' + headphones.CONFIG.PUSHBULLET_APIKEY}
|
||||||
return True
|
|
||||||
elif request_status >= 400 and request_status < 500:
|
response = request.request_json(url, method="post", headers=headers, data=json.dumps(data))
|
||||||
logger.info(u"PushBullet request failed: %s" % response.reason)
|
|
||||||
return False
|
if response:
|
||||||
|
logger.info(u"PushBullet notifications sent.")
|
||||||
|
return True
|
||||||
else:
|
else:
|
||||||
logger.info(u"PushBullet notification failed serverside.")
|
logger.info(u"PushBullet notification failed.")
|
||||||
return False
|
return False
|
||||||
|
|
||||||
def updateLibrary(self):
|
|
||||||
#For uniformity reasons not removed
|
|
||||||
return
|
|
||||||
|
|
||||||
def test(self, apikey, deviceid):
|
|
||||||
|
|
||||||
self.enabled = True
|
|
||||||
self.apikey = apikey
|
|
||||||
self.deviceid = deviceid
|
|
||||||
|
|
||||||
self.notify('Main Screen Activate', 'Test Message')
|
|
||||||
|
|
||||||
|
|
||||||
class PUSHALOT(object):
|
class PUSHALOT(object):
|
||||||
|
|
||||||
@@ -583,7 +564,7 @@ class PUSHOVER(object):
|
|||||||
if not headphones.CONFIG.PUSHOVER_ENABLED:
|
if not headphones.CONFIG.PUSHOVER_ENABLED:
|
||||||
return
|
return
|
||||||
|
|
||||||
http_handler = HTTPSConnection("api.pushover.net")
|
url = "https://api.pushover.net/1/messages.json"
|
||||||
|
|
||||||
data = {'token': self.application_token,
|
data = {'token': self.application_token,
|
||||||
'user': headphones.CONFIG.PUSHOVER_KEYS,
|
'user': headphones.CONFIG.PUSHOVER_KEYS,
|
||||||
@@ -591,25 +572,16 @@ class PUSHOVER(object):
|
|||||||
'message': message.encode("utf-8"),
|
'message': message.encode("utf-8"),
|
||||||
'priority': headphones.CONFIG.PUSHOVER_PRIORITY}
|
'priority': headphones.CONFIG.PUSHOVER_PRIORITY}
|
||||||
|
|
||||||
http_handler.request("POST",
|
headers = {'Content-type': "application/x-www-form-urlencoded"}
|
||||||
"/1/messages.json",
|
|
||||||
headers={'Content-type': "application/x-www-form-urlencoded"},
|
|
||||||
body=urlencode(data))
|
|
||||||
response = http_handler.getresponse()
|
|
||||||
request_status = response.status
|
|
||||||
logger.debug(u"Pushover response status: %r" % request_status)
|
|
||||||
logger.debug(u"Pushover response headers: %r" % response.getheaders())
|
|
||||||
logger.debug(u"Pushover response body: %r" % response.read())
|
|
||||||
|
|
||||||
if request_status == 200:
|
response = request.request_response(url, method="POST", headers=headers, data=data)
|
||||||
logger.info(u"Pushover notifications sent.")
|
|
||||||
return True
|
if response:
|
||||||
elif request_status >= 400 and request_status < 500:
|
logger.info(u"Pushover notifications sent.")
|
||||||
logger.info(u"Pushover request failed: %s" % response.reason)
|
return True
|
||||||
return False
|
|
||||||
else:
|
else:
|
||||||
logger.info(u"Pushover notification failed.")
|
logger.error(u"Pushover notification failed.")
|
||||||
return False
|
return False
|
||||||
|
|
||||||
def updateLibrary(self):
|
def updateLibrary(self):
|
||||||
#For uniformity reasons not removed
|
#For uniformity reasons not removed
|
||||||
|
|||||||
+25
-22
@@ -59,7 +59,7 @@ def checkFolder():
|
|||||||
|
|
||||||
logger.debug("Checking download folder finished.")
|
logger.debug("Checking download folder finished.")
|
||||||
|
|
||||||
def verify(albumid, albumpath, Kind=None, forced=False):
|
def verify(albumid, albumpath, Kind=None, forced=False, keep_original_folder=False):
|
||||||
|
|
||||||
myDB = db.DBConnection()
|
myDB = db.DBConnection()
|
||||||
release = myDB.action('SELECT * from albums WHERE AlbumID=?', [albumid]).fetchone()
|
release = myDB.action('SELECT * from albums WHERE AlbumID=?', [albumid]).fetchone()
|
||||||
@@ -216,7 +216,7 @@ def verify(albumid, albumpath, Kind=None, forced=False):
|
|||||||
logger.debug('Matching metadata album: %s with album name: %s' % (metaalbum, dbalbum))
|
logger.debug('Matching metadata album: %s with album name: %s' % (metaalbum, dbalbum))
|
||||||
|
|
||||||
if metaartist == dbartist and metaalbum == dbalbum:
|
if metaartist == dbartist and metaalbum == dbalbum:
|
||||||
doPostProcessing(albumid, albumpath, release, tracks, downloaded_track_list, Kind)
|
doPostProcessing(albumid, albumpath, release, tracks, downloaded_track_list, Kind, keep_original_folder)
|
||||||
return
|
return
|
||||||
|
|
||||||
# test #2: filenames
|
# test #2: filenames
|
||||||
@@ -234,7 +234,7 @@ def verify(albumid, albumpath, Kind=None, forced=False):
|
|||||||
logger.debug('Checking if track title: %s is in file name: %s' % (dbtrack, filetrack))
|
logger.debug('Checking if track title: %s is in file name: %s' % (dbtrack, filetrack))
|
||||||
|
|
||||||
if dbtrack in filetrack:
|
if dbtrack in filetrack:
|
||||||
doPostProcessing(albumid, albumpath, release, tracks, downloaded_track_list, Kind)
|
doPostProcessing(albumid, albumpath, release, tracks, downloaded_track_list, Kind, keep_original_folder)
|
||||||
return
|
return
|
||||||
|
|
||||||
# test #3: number of songs and duration
|
# test #3: number of songs and duration
|
||||||
@@ -266,7 +266,7 @@ def verify(albumid, albumpath, Kind=None, forced=False):
|
|||||||
logger.debug('Database track duration: %i' % db_track_duration)
|
logger.debug('Database track duration: %i' % db_track_duration)
|
||||||
delta = abs(downloaded_track_duration - db_track_duration)
|
delta = abs(downloaded_track_duration - db_track_duration)
|
||||||
if delta < 240:
|
if delta < 240:
|
||||||
doPostProcessing(albumid, albumpath, release, tracks, downloaded_track_list, Kind)
|
doPostProcessing(albumid, albumpath, release, tracks, downloaded_track_list, Kind, keep_original_folder)
|
||||||
return
|
return
|
||||||
|
|
||||||
logger.warn(u'Could not identify album: %s. It may not be the intended album.' % albumpath.decode(headphones.SYS_ENCODING, 'replace'))
|
logger.warn(u'Could not identify album: %s. It may not be the intended album.' % albumpath.decode(headphones.SYS_ENCODING, 'replace'))
|
||||||
@@ -276,11 +276,11 @@ def verify(albumid, albumpath, Kind=None, forced=False):
|
|||||||
renameUnprocessedFolder(albumpath, tag="Unprocessed")
|
renameUnprocessedFolder(albumpath, tag="Unprocessed")
|
||||||
|
|
||||||
|
|
||||||
def doPostProcessing(albumid, albumpath, release, tracks, downloaded_track_list, Kind=None):
|
def doPostProcessing(albumid, albumpath, release, tracks, downloaded_track_list, Kind=None, keep_original_folder=False):
|
||||||
|
|
||||||
logger.info('Starting post-processing for: %s - %s' % (release['ArtistName'], release['AlbumTitle']))
|
logger.info('Starting post-processing for: %s - %s' % (release['ArtistName'], release['AlbumTitle']))
|
||||||
# Check to see if we're preserving the torrent dir
|
# Check to see if we're preserving the torrent dir
|
||||||
if headphones.CONFIG.KEEP_TORRENT_FILES and Kind == "torrent" and 'headphones-modified' not in albumpath:
|
if (headphones.CONFIG.KEEP_TORRENT_FILES and Kind == "torrent" and 'headphones-modified' not in albumpath) or headphones.CONFIG.KEEP_ORIGINAL_FOLDER or keep_original_folder:
|
||||||
new_folder = os.path.join(albumpath, 'headphones-modified'.encode(headphones.SYS_ENCODING, 'replace'))
|
new_folder = os.path.join(albumpath, 'headphones-modified'.encode(headphones.SYS_ENCODING, 'replace'))
|
||||||
logger.info("Copying files to 'headphones-modified' subfolder to preserve downloaded files for seeding")
|
logger.info("Copying files to 'headphones-modified' subfolder to preserve downloaded files for seeding")
|
||||||
try:
|
try:
|
||||||
@@ -372,7 +372,9 @@ def doPostProcessing(albumid, albumpath, release, tracks, downloaded_track_list,
|
|||||||
addAlbumArt(artwork, albumpath, release)
|
addAlbumArt(artwork, albumpath, release)
|
||||||
|
|
||||||
if headphones.CONFIG.CORRECT_METADATA:
|
if headphones.CONFIG.CORRECT_METADATA:
|
||||||
correctMetadata(albumid, release, downloaded_track_list)
|
correctedMetadata = correctMetadata(albumid, release, downloaded_track_list)
|
||||||
|
if not correctedMetadata and headphones.CONFIG.DO_NOT_PROCESS_UNMATCHED:
|
||||||
|
return
|
||||||
|
|
||||||
if headphones.CONFIG.EMBED_LYRICS:
|
if headphones.CONFIG.EMBED_LYRICS:
|
||||||
embedLyrics(downloaded_track_list)
|
embedLyrics(downloaded_track_list)
|
||||||
@@ -475,7 +477,7 @@ def doPostProcessing(albumid, albumpath, release, tracks, downloaded_track_list,
|
|||||||
if headphones.CONFIG.PUSHBULLET_ENABLED:
|
if headphones.CONFIG.PUSHBULLET_ENABLED:
|
||||||
logger.info(u"PushBullet request")
|
logger.info(u"PushBullet request")
|
||||||
pushbullet = notifiers.PUSHBULLET()
|
pushbullet = notifiers.PUSHBULLET()
|
||||||
pushbullet.notify(pushmessage, "Download and Postprocessing completed")
|
pushbullet.notify(pushmessage, statusmessage)
|
||||||
|
|
||||||
if headphones.CONFIG.TWITTER_ENABLED:
|
if headphones.CONFIG.TWITTER_ENABLED:
|
||||||
logger.info(u"Sending Twitter notification")
|
logger.info(u"Sending Twitter notification")
|
||||||
@@ -595,7 +597,6 @@ def renameNFO(albumpath):
|
|||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.error(u'Could not rename file: %s. Error: %s' % (os.path.join(r, file).decode(headphones.SYS_ENCODING, 'replace'), e))
|
logger.error(u'Could not rename file: %s. Error: %s' % (os.path.join(r, file).decode(headphones.SYS_ENCODING, 'replace'), e))
|
||||||
|
|
||||||
|
|
||||||
def moveFiles(albumpath, release, tracks):
|
def moveFiles(albumpath, release, tracks):
|
||||||
logger.info("Moving files: %s" % albumpath)
|
logger.info("Moving files: %s" % albumpath)
|
||||||
try:
|
try:
|
||||||
@@ -858,16 +859,16 @@ def correctMetadata(albumid, release, downloaded_track_list):
|
|||||||
cur_artist, cur_album, candidates, rec = autotag.tag_album(items, search_artist=helpers.latinToAscii(release['ArtistName']), search_album=helpers.latinToAscii(release['AlbumTitle']))
|
cur_artist, cur_album, candidates, rec = autotag.tag_album(items, search_artist=helpers.latinToAscii(release['ArtistName']), search_album=helpers.latinToAscii(release['AlbumTitle']))
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.error('Error getting recommendation: %s. Not writing metadata', e)
|
logger.error('Error getting recommendation: %s. Not writing metadata', e)
|
||||||
return
|
return False
|
||||||
if str(rec) == 'Recommendation.none':
|
if str(rec) == 'Recommendation.none':
|
||||||
logger.warn('No accurate album match found for %s, %s - not writing metadata', release['ArtistName'], release['AlbumTitle'])
|
logger.warn('No accurate album match found for %s, %s - not writing metadata', release['ArtistName'], release['AlbumTitle'])
|
||||||
return
|
return False
|
||||||
|
|
||||||
if candidates:
|
if candidates:
|
||||||
dist, info, mapping, extra_items, extra_tracks = candidates[0]
|
dist, info, mapping, extra_items, extra_tracks = candidates[0]
|
||||||
else:
|
else:
|
||||||
logger.warn('No accurate album match found for %s, %s - not writing metadata', release['ArtistName'], release['AlbumTitle'])
|
logger.warn('No accurate album match found for %s, %s - not writing metadata', release['ArtistName'], release['AlbumTitle'])
|
||||||
return
|
return False
|
||||||
|
|
||||||
logger.info('Beets recommendation for tagging items: %s' % rec)
|
logger.info('Beets recommendation for tagging items: %s' % rec)
|
||||||
|
|
||||||
@@ -889,7 +890,9 @@ def correctMetadata(albumid, release, downloaded_track_list):
|
|||||||
logger.info("Successfully applied metadata to: %s", item.path.decode(headphones.SYS_ENCODING, 'replace'))
|
logger.info("Successfully applied metadata to: %s", item.path.decode(headphones.SYS_ENCODING, 'replace'))
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.warn("Error writing metadata to '%s': %s", item.path.decode(headphones.SYS_ENCODING, 'replace'), str(e))
|
logger.warn("Error writing metadata to '%s': %s", item.path.decode(headphones.SYS_ENCODING, 'replace'), str(e))
|
||||||
|
return False
|
||||||
|
|
||||||
|
return True
|
||||||
|
|
||||||
def embedLyrics(downloaded_track_list):
|
def embedLyrics(downloaded_track_list):
|
||||||
logger.info('Adding lyrics')
|
logger.info('Adding lyrics')
|
||||||
@@ -1060,7 +1063,7 @@ def renameUnprocessedFolder(path, tag):
|
|||||||
return
|
return
|
||||||
|
|
||||||
|
|
||||||
def forcePostProcess(dir=None, expand_subfolders=True, album_dir=None):
|
def forcePostProcess(dir=None, expand_subfolders=True, album_dir=None, keep_original_folder=False):
|
||||||
|
|
||||||
logger.info('Force checking download folder for completed downloads')
|
logger.info('Force checking download folder for completed downloads')
|
||||||
|
|
||||||
@@ -1136,7 +1139,7 @@ def forcePostProcess(dir=None, expand_subfolders=True, album_dir=None):
|
|||||||
continue
|
continue
|
||||||
else:
|
else:
|
||||||
logger.info('Found a match in the database: %s. Verifying to make sure it is the correct album', snatched['Title'])
|
logger.info('Found a match in the database: %s. Verifying to make sure it is the correct album', snatched['Title'])
|
||||||
verify(snatched['AlbumID'], folder, snatched['Kind'])
|
verify(snatched['AlbumID'], folder, snatched['Kind'], keep_original_folder=keep_original_folder)
|
||||||
continue
|
continue
|
||||||
|
|
||||||
# Attempt 2: strip release group id from filename
|
# Attempt 2: strip release group id from filename
|
||||||
@@ -1153,10 +1156,10 @@ def forcePostProcess(dir=None, expand_subfolders=True, album_dir=None):
|
|||||||
release = myDB.action('SELECT ArtistName, AlbumTitle, AlbumID from albums WHERE AlbumID=?', [rgid]).fetchone()
|
release = myDB.action('SELECT ArtistName, AlbumTitle, AlbumID from albums WHERE AlbumID=?', [rgid]).fetchone()
|
||||||
if release:
|
if release:
|
||||||
logger.info('Found a match in the database: %s - %s. Verifying to make sure it is the correct album', release['ArtistName'], release['AlbumTitle'])
|
logger.info('Found a match in the database: %s - %s. Verifying to make sure it is the correct album', release['ArtistName'], release['AlbumTitle'])
|
||||||
verify(release['AlbumID'], folder, forced=True)
|
verify(release['AlbumID'], folder, forced=True, keep_original_folder=keep_original_folder)
|
||||||
continue
|
continue
|
||||||
else:
|
else:
|
||||||
logger.info('Found a (possibly) valid Musicbrainz realse group id in album folder name.')
|
logger.info('Found a (possibly) valid Musicbrainz release group id in album folder name.')
|
||||||
verify(rgid, folder, forced=True)
|
verify(rgid, folder, forced=True)
|
||||||
continue
|
continue
|
||||||
|
|
||||||
@@ -1172,7 +1175,7 @@ def forcePostProcess(dir=None, expand_subfolders=True, album_dir=None):
|
|||||||
release = myDB.action('SELECT AlbumID, ArtistName, AlbumTitle from albums WHERE ArtistName LIKE ? and AlbumTitle LIKE ?', [name, album]).fetchone()
|
release = myDB.action('SELECT AlbumID, ArtistName, AlbumTitle from albums WHERE ArtistName LIKE ? and AlbumTitle LIKE ?', [name, album]).fetchone()
|
||||||
if release:
|
if release:
|
||||||
logger.info('Found a match in the database: %s - %s. Verifying to make sure it is the correct album', release['ArtistName'], release['AlbumTitle'])
|
logger.info('Found a match in the database: %s - %s. Verifying to make sure it is the correct album', release['ArtistName'], release['AlbumTitle'])
|
||||||
verify(release['AlbumID'], folder)
|
verify(release['AlbumID'], folder, keep_original_folder=keep_original_folder)
|
||||||
continue
|
continue
|
||||||
else:
|
else:
|
||||||
logger.info('Querying MusicBrainz for the release group id for: %s - %s', name, album)
|
logger.info('Querying MusicBrainz for the release group id for: %s - %s', name, album)
|
||||||
@@ -1183,7 +1186,7 @@ def forcePostProcess(dir=None, expand_subfolders=True, album_dir=None):
|
|||||||
rgid = None
|
rgid = None
|
||||||
|
|
||||||
if rgid:
|
if rgid:
|
||||||
verify(rgid, folder)
|
verify(rgid, folder, keep_original_folder=keep_original_folder)
|
||||||
continue
|
continue
|
||||||
else:
|
else:
|
||||||
logger.info('No match found on MusicBrainz for: %s - %s', name, album)
|
logger.info('No match found on MusicBrainz for: %s - %s', name, album)
|
||||||
@@ -1207,7 +1210,7 @@ def forcePostProcess(dir=None, expand_subfolders=True, album_dir=None):
|
|||||||
release = myDB.action('SELECT AlbumID, ArtistName, AlbumTitle from albums WHERE ArtistName LIKE ? and AlbumTitle LIKE ?', [name, album]).fetchone()
|
release = myDB.action('SELECT AlbumID, ArtistName, AlbumTitle from albums WHERE ArtistName LIKE ? and AlbumTitle LIKE ?', [name, album]).fetchone()
|
||||||
if release:
|
if release:
|
||||||
logger.info('Found a match in the database: %s - %s. Verifying to make sure it is the correct album', release['ArtistName'], release['AlbumTitle'])
|
logger.info('Found a match in the database: %s - %s. Verifying to make sure it is the correct album', release['ArtistName'], release['AlbumTitle'])
|
||||||
verify(release['AlbumID'], folder)
|
verify(release['AlbumID'], folder, keep_original_folder=keep_original_folder)
|
||||||
continue
|
continue
|
||||||
else:
|
else:
|
||||||
logger.info('Querying MusicBrainz for the release group id for: %s - %s', name, album)
|
logger.info('Querying MusicBrainz for the release group id for: %s - %s', name, album)
|
||||||
@@ -1218,7 +1221,7 @@ def forcePostProcess(dir=None, expand_subfolders=True, album_dir=None):
|
|||||||
rgid = None
|
rgid = None
|
||||||
|
|
||||||
if rgid:
|
if rgid:
|
||||||
verify(rgid, folder)
|
verify(rgid, folder, keep_original_folder=keep_original_folder)
|
||||||
continue
|
continue
|
||||||
else:
|
else:
|
||||||
logger.info('No match found on MusicBrainz for: %s - %s', name, album)
|
logger.info('No match found on MusicBrainz for: %s - %s', name, album)
|
||||||
@@ -1231,7 +1234,7 @@ def forcePostProcess(dir=None, expand_subfolders=True, album_dir=None):
|
|||||||
release = myDB.action('SELECT AlbumID, ArtistName, AlbumTitle from albums WHERE AlbumTitle LIKE ?', [folder_basename]).fetchone()
|
release = myDB.action('SELECT AlbumID, ArtistName, AlbumTitle from albums WHERE AlbumTitle LIKE ?', [folder_basename]).fetchone()
|
||||||
if release:
|
if release:
|
||||||
logger.info('Found a match in the database: %s - %s. Verifying to make sure it is the correct album', release['ArtistName'], release['AlbumTitle'])
|
logger.info('Found a match in the database: %s - %s. Verifying to make sure it is the correct album', release['ArtistName'], release['AlbumTitle'])
|
||||||
verify(release['AlbumID'], folder)
|
verify(release['AlbumID'], folder, keep_original_folder=keep_original_folder)
|
||||||
continue
|
continue
|
||||||
else:
|
else:
|
||||||
logger.info('Querying MusicBrainz for the release group id for: %s', folder_basename)
|
logger.info('Querying MusicBrainz for the release group id for: %s', folder_basename)
|
||||||
@@ -1242,7 +1245,7 @@ def forcePostProcess(dir=None, expand_subfolders=True, album_dir=None):
|
|||||||
rgid = None
|
rgid = None
|
||||||
|
|
||||||
if rgid:
|
if rgid:
|
||||||
verify(rgid, folder)
|
verify(rgid, folder, keep_original_folder=keep_original_folder)
|
||||||
continue
|
continue
|
||||||
else:
|
else:
|
||||||
logger.info('No match found on MusicBrainz for: %s - %s', name, album)
|
logger.info('No match found on MusicBrainz for: %s - %s', name, album)
|
||||||
|
|||||||
@@ -206,7 +206,7 @@ def server_message(response):
|
|||||||
message = None
|
message = None
|
||||||
|
|
||||||
# First attempt is to 'read' the response as HTML
|
# First attempt is to 'read' the response as HTML
|
||||||
if "text/html" in response.headers.get("content-type"):
|
if response.headers.get("content-type") and "text/html" in response.headers.get("content-type"):
|
||||||
try:
|
try:
|
||||||
soup = BeautifulSoup(response.content, "html5lib")
|
soup = BeautifulSoup(response.content, "html5lib")
|
||||||
except Exception:
|
except Exception:
|
||||||
|
|||||||
@@ -0,0 +1,224 @@
|
|||||||
|
#!/usr/bin/env python
|
||||||
|
|
||||||
|
import urllib
|
||||||
|
import requests as requests
|
||||||
|
from urlparse import urlparse
|
||||||
|
from bs4 import BeautifulSoup
|
||||||
|
|
||||||
|
import os
|
||||||
|
import time
|
||||||
|
import re
|
||||||
|
|
||||||
|
import headphones
|
||||||
|
from headphones import logger
|
||||||
|
|
||||||
|
class Rutracker(object):
|
||||||
|
|
||||||
|
def __init__(self):
|
||||||
|
self.session = requests.session()
|
||||||
|
self.timeout = 60
|
||||||
|
self.loggedin = False
|
||||||
|
self.maxsize = 0
|
||||||
|
self.search_referer = 'http://rutracker.org/forum/tracker.php'
|
||||||
|
|
||||||
|
def logged_in(self):
|
||||||
|
return self.loggedin
|
||||||
|
|
||||||
|
def still_logged_in(self, html):
|
||||||
|
if not html or "action=\"http://login.rutracker.org/forum/login.php\">" in html:
|
||||||
|
return False
|
||||||
|
else:
|
||||||
|
return True
|
||||||
|
|
||||||
|
def login(self):
|
||||||
|
"""
|
||||||
|
Logs in user
|
||||||
|
"""
|
||||||
|
|
||||||
|
loginpage = 'http://login.rutracker.org/forum/login.php'
|
||||||
|
post_params = {
|
||||||
|
'login_username': headphones.CONFIG.RUTRACKER_USER,
|
||||||
|
'login_password': headphones.CONFIG.RUTRACKER_PASSWORD,
|
||||||
|
'login': b'\xc2\xf5\xee\xe4' # '%C2%F5%EE%E4'
|
||||||
|
}
|
||||||
|
|
||||||
|
logger.info("Attempting to log in to rutracker...")
|
||||||
|
|
||||||
|
try:
|
||||||
|
r = self.session.post(loginpage, data=post_params, timeout=self.timeout)
|
||||||
|
# try again
|
||||||
|
if 'bb_data' not in r.cookies.keys():
|
||||||
|
time.sleep(10)
|
||||||
|
r = self.session.post(loginpage, data=post_params, timeout=self.timeout)
|
||||||
|
if r.status_code != 200:
|
||||||
|
logger.error("rutracker login returned status code %s" % r.status_code)
|
||||||
|
self.loggedin = False
|
||||||
|
else:
|
||||||
|
if 'bb_data' in r.cookies.keys():
|
||||||
|
self.loggedin = True
|
||||||
|
logger.info("Successfully logged in to rutracker")
|
||||||
|
else:
|
||||||
|
logger.error("Could not login to rutracker, credentials maybe incorrect, site is down or too many attempts. Try again later")
|
||||||
|
self.loggedin = False
|
||||||
|
return self.loggedin
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("Unknown error logging in to rutracker: %s" % e)
|
||||||
|
self.loggedin = False
|
||||||
|
return self.loggedin
|
||||||
|
|
||||||
|
def searchurl(self, artist, album, year, format):
|
||||||
|
"""
|
||||||
|
Return the search url
|
||||||
|
"""
|
||||||
|
|
||||||
|
# Build search url
|
||||||
|
searchterm = ''
|
||||||
|
if artist != 'Various Artists':
|
||||||
|
searchterm = artist
|
||||||
|
searchterm = searchterm + ' '
|
||||||
|
searchterm = searchterm + album
|
||||||
|
searchterm = searchterm + ' '
|
||||||
|
searchterm = searchterm + year
|
||||||
|
|
||||||
|
if format == 'lossless':
|
||||||
|
format = '+lossless'
|
||||||
|
self.maxsize = 10000000000
|
||||||
|
elif format == 'lossless+mp3':
|
||||||
|
format = '+lossless||mp3||aac'
|
||||||
|
self.maxsize = 10000000000
|
||||||
|
else:
|
||||||
|
format = '+mp3||aac'
|
||||||
|
self.maxsize = 300000000
|
||||||
|
|
||||||
|
# sort by size, descending.
|
||||||
|
sort = '&o=7&s=2'
|
||||||
|
|
||||||
|
searchurl = "%s?nm=%s%s%s" % (self.search_referer, urllib.quote(searchterm), format, sort)
|
||||||
|
|
||||||
|
logger.info("Searching rutracker using term: %s", searchterm)
|
||||||
|
|
||||||
|
return searchurl
|
||||||
|
|
||||||
|
def search(self, searchurl):
|
||||||
|
"""
|
||||||
|
Parse the search results and return valid torrent list
|
||||||
|
"""
|
||||||
|
|
||||||
|
try:
|
||||||
|
headers = {'Referer': self.search_referer}
|
||||||
|
r = self.session.get(url=searchurl, headers=headers, timeout=self.timeout)
|
||||||
|
|
||||||
|
soup = BeautifulSoup(r.content, 'html5lib')
|
||||||
|
|
||||||
|
# Debug
|
||||||
|
#logger.debug (soup.prettify())
|
||||||
|
|
||||||
|
# Check if still logged in
|
||||||
|
if not self.still_logged_in(soup):
|
||||||
|
self.login()
|
||||||
|
r = self.session.get(url=searchurl, timeout=self.timeout)
|
||||||
|
soup = BeautifulSoup(r.content, 'html5lib')
|
||||||
|
if not self.still_logged_in(soup):
|
||||||
|
logger.error("Error getting rutracker data")
|
||||||
|
return None
|
||||||
|
|
||||||
|
# Process
|
||||||
|
rulist = []
|
||||||
|
i = soup.find('table', id='tor-tbl')
|
||||||
|
if not i:
|
||||||
|
logger.info("No valid results found from rutracker")
|
||||||
|
return None
|
||||||
|
minimumseeders = int(headphones.CONFIG.NUMBEROFSEEDERS) - 1
|
||||||
|
|
||||||
|
for item in zip(i.find_all(class_='hl-tags'),i.find_all(class_='dl-stub'),i.find_all(class_='seedmed')):
|
||||||
|
title = item[0].get_text()
|
||||||
|
url = item[1].get('href')
|
||||||
|
size_formatted = item[1].get_text()[:-2]
|
||||||
|
seeds = item[2].get_text()
|
||||||
|
size_parts = size_formatted.split()
|
||||||
|
size = float(size_parts[0])
|
||||||
|
|
||||||
|
if size_parts[1] == 'KB':
|
||||||
|
size *= 1024
|
||||||
|
if size_parts[1] == 'MB':
|
||||||
|
size *= 1024 ** 2
|
||||||
|
if size_parts[1] == 'GB':
|
||||||
|
size *= 1024 ** 3
|
||||||
|
if size_parts[1] == 'TB':
|
||||||
|
size *= 1024 ** 4
|
||||||
|
|
||||||
|
if size < self.maxsize and minimumseeders < int(seeds):
|
||||||
|
logger.info('Found %s. Size: %s' % (title, size_formatted))
|
||||||
|
#Torrent topic page
|
||||||
|
torrent_id = dict([part.split('=') for part in urlparse(url)[4].split('&')])['t']
|
||||||
|
topicurl = 'http://rutracker.org/forum/viewtopic.php?t=' + torrent_id
|
||||||
|
rulist.append((title, size, topicurl, 'rutracker.org', 'torrent', True))
|
||||||
|
else:
|
||||||
|
logger.info("%s is larger than the maxsize or has too little seeders for this category, skipping. (Size: %i bytes, Seeders: %i)" % (title, size, int(seeds)))
|
||||||
|
|
||||||
|
if not rulist:
|
||||||
|
logger.info("No valid results found from rutracker")
|
||||||
|
|
||||||
|
return rulist
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("An unknown error occurred in the rutracker parser: %s" % e)
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def get_torrent_data(self, url):
|
||||||
|
"""
|
||||||
|
return the .torrent data
|
||||||
|
"""
|
||||||
|
|
||||||
|
torrent_id = dict([part.split('=') for part in urlparse(url)[4].split('&')])['t']
|
||||||
|
downloadurl = 'http://dl.rutracker.org/forum/dl.php?t=' + torrent_id
|
||||||
|
cookie = {'bb_dl': torrent_id}
|
||||||
|
try:
|
||||||
|
headers = {'Referer': url}
|
||||||
|
r = self.session.get(url=downloadurl, cookies=cookie, headers=headers, timeout=self.timeout)
|
||||||
|
return r.content
|
||||||
|
except Exception as e:
|
||||||
|
logger.error('Error getting torrent: %s', e)
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
#TODO get this working in utorrent.py
|
||||||
|
def utorrent_add_file(self, data):
|
||||||
|
|
||||||
|
host = headphones.CONFIG.UTORRENT_HOST
|
||||||
|
if not host.startswith('http'):
|
||||||
|
host = 'http://' + host
|
||||||
|
if host.endswith('/'):
|
||||||
|
host = host[:-1]
|
||||||
|
if host.endswith('/gui'):
|
||||||
|
host = host[:-4]
|
||||||
|
|
||||||
|
base_url = host
|
||||||
|
|
||||||
|
url = base_url + '/gui/'
|
||||||
|
self.session.auth = (headphones.CONFIG.UTORRENT_USERNAME, headphones.CONFIG.UTORRENT_PASSWORD)
|
||||||
|
|
||||||
|
try:
|
||||||
|
r = self.session.get(url + 'token.html')
|
||||||
|
except Exception as e:
|
||||||
|
logger.error('Error getting token: %s', e)
|
||||||
|
return
|
||||||
|
|
||||||
|
if r.status_code == 401:
|
||||||
|
logger.debug('Error reaching utorrent')
|
||||||
|
return
|
||||||
|
|
||||||
|
regex = re.search(r'.+>([^<]+)</div></html>', r.text)
|
||||||
|
if regex is None:
|
||||||
|
logger.debug('Error reading token')
|
||||||
|
return
|
||||||
|
|
||||||
|
self.session.params = {'token': regex.group(1)}
|
||||||
|
files = {'torrent_file': ("", data)}
|
||||||
|
|
||||||
|
try:
|
||||||
|
self.session.post(url, params={'action': 'add-file'}, files=files)
|
||||||
|
except Exception as e:
|
||||||
|
logger.exception('Error adding file to utorrent %s', e)
|
||||||
|
|
||||||
+193
-82
@@ -29,30 +29,27 @@ import string
|
|||||||
import shutil
|
import shutil
|
||||||
import random
|
import random
|
||||||
import urllib
|
import urllib
|
||||||
|
import datetime
|
||||||
import headphones
|
import headphones
|
||||||
import subprocess
|
import subprocess
|
||||||
import unicodedata
|
import unicodedata
|
||||||
|
|
||||||
from headphones.common import USER_AGENT
|
from headphones.common import USER_AGENT
|
||||||
from headphones import logger, db, helpers, classes, sab, nzbget, request
|
from headphones import logger, db, helpers, classes, sab, nzbget, request
|
||||||
from headphones import utorrent, transmission, notifiers
|
from headphones import utorrent, transmission, notifiers, rutracker
|
||||||
|
|
||||||
from bencode import bencode, bdecode
|
from bencode import bencode, bdecode
|
||||||
|
|
||||||
import headphones.searcher_rutracker as rutrackersearch
|
|
||||||
|
|
||||||
# Magnet to torrent services, for Black hole. Stolen from CouchPotato.
|
# Magnet to torrent services, for Black hole. Stolen from CouchPotato.
|
||||||
TORRENT_TO_MAGNET_SERVICES = [
|
TORRENT_TO_MAGNET_SERVICES = [
|
||||||
'https://zoink.it/torrent/%s.torrent',
|
#'https://zoink.it/torrent/%s.torrent',
|
||||||
'http://torrage.com/torrent/%s.torrent',
|
#'http://torrage.com/torrent/%s.torrent',
|
||||||
'https://torcache.net/torrent/%s.torrent',
|
'https://torcache.net/torrent/%s.torrent',
|
||||||
]
|
]
|
||||||
|
|
||||||
# Persistent What.cd API object
|
# Persistent What.cd API object
|
||||||
gazelle = None
|
gazelle = None
|
||||||
|
ruobj = None
|
||||||
# RUtracker search object
|
|
||||||
rutracker = rutrackersearch.Rutracker()
|
|
||||||
|
|
||||||
|
|
||||||
def fix_url(s, charset="utf-8"):
|
def fix_url(s, charset="utf-8"):
|
||||||
@@ -167,6 +164,8 @@ def get_seed_ratio(provider):
|
|||||||
seed_ratio = headphones.CONFIG.WAFFLES_RATIO
|
seed_ratio = headphones.CONFIG.WAFFLES_RATIO
|
||||||
elif provider == 'Mininova':
|
elif provider == 'Mininova':
|
||||||
seed_ratio = headphones.CONFIG.MININOVA_RATIO
|
seed_ratio = headphones.CONFIG.MININOVA_RATIO
|
||||||
|
elif provider == 'Strike':
|
||||||
|
seed_ratio = headphones.CONFIG.STRIKE_RATIO
|
||||||
else:
|
else:
|
||||||
seed_ratio = None
|
seed_ratio = None
|
||||||
|
|
||||||
@@ -189,10 +188,22 @@ def searchforalbum(albumid=None, new=False, losslessOnly=False,
|
|||||||
results = myDB.select('SELECT * from albums WHERE Status="Wanted" OR Status="Wanted Lossless"')
|
results = myDB.select('SELECT * from albums WHERE Status="Wanted" OR Status="Wanted Lossless"')
|
||||||
|
|
||||||
for album in results:
|
for album in results:
|
||||||
|
|
||||||
if not album['AlbumTitle'] or not album['ArtistName']:
|
if not album['AlbumTitle'] or not album['ArtistName']:
|
||||||
logger.warn('Skipping release %s. No title available', album['AlbumID'])
|
logger.warn('Skipping release %s. No title available', album['AlbumID'])
|
||||||
continue
|
continue
|
||||||
|
|
||||||
|
if headphones.CONFIG.WAIT_UNTIL_RELEASE_DATE and album['ReleaseDate']:
|
||||||
|
try:
|
||||||
|
release_date = datetime.datetime.strptime(album['ReleaseDate'], "%Y-%m-%d")
|
||||||
|
except:
|
||||||
|
logger.warn("No valid date for: %s. Skipping automatic search" % album['AlbumTitle'])
|
||||||
|
continue
|
||||||
|
|
||||||
|
if release_date > datetime.datetime.today():
|
||||||
|
logger.info("Skipping: %s. Waiting for release date of: %s" % (album['AlbumTitle'], album['ReleaseDate']))
|
||||||
|
continue
|
||||||
|
|
||||||
new = True
|
new = True
|
||||||
|
|
||||||
if album['Status'] == "Wanted Lossless":
|
if album['Status'] == "Wanted Lossless":
|
||||||
@@ -219,7 +230,7 @@ def do_sorted_search(album, new, losslessOnly, choose_specific_download=False):
|
|||||||
|
|
||||||
NZB_PROVIDERS = (headphones.CONFIG.HEADPHONES_INDEXER or headphones.CONFIG.NEWZNAB or headphones.CONFIG.NZBSORG or headphones.CONFIG.OMGWTFNZBS)
|
NZB_PROVIDERS = (headphones.CONFIG.HEADPHONES_INDEXER or headphones.CONFIG.NEWZNAB or headphones.CONFIG.NZBSORG or headphones.CONFIG.OMGWTFNZBS)
|
||||||
NZB_DOWNLOADERS = (headphones.CONFIG.SAB_HOST or headphones.CONFIG.BLACKHOLE_DIR or headphones.CONFIG.NZBGET_HOST)
|
NZB_DOWNLOADERS = (headphones.CONFIG.SAB_HOST or headphones.CONFIG.BLACKHOLE_DIR or headphones.CONFIG.NZBGET_HOST)
|
||||||
TORRENT_PROVIDERS = (headphones.CONFIG.KAT or headphones.CONFIG.PIRATEBAY or headphones.CONFIG.OLDPIRATEBAY or headphones.CONFIG.MININOVA or headphones.CONFIG.WAFFLES or headphones.CONFIG.RUTRACKER or headphones.CONFIG.WHATCD)
|
TORRENT_PROVIDERS = (headphones.CONFIG.TORZNAB or headphones.CONFIG.KAT or headphones.CONFIG.PIRATEBAY or headphones.CONFIG.OLDPIRATEBAY or headphones.CONFIG.MININOVA or headphones.CONFIG.WAFFLES or headphones.CONFIG.RUTRACKER or headphones.CONFIG.WHATCD or headphones.CONFIG.STRIKE)
|
||||||
|
|
||||||
results = []
|
results = []
|
||||||
myDB = db.DBConnection()
|
myDB = db.DBConnection()
|
||||||
@@ -780,10 +791,11 @@ def send_to_downloader(data, bestqual, album):
|
|||||||
# Randomize list of services
|
# Randomize list of services
|
||||||
services = TORRENT_TO_MAGNET_SERVICES[:]
|
services = TORRENT_TO_MAGNET_SERVICES[:]
|
||||||
random.shuffle(services)
|
random.shuffle(services)
|
||||||
|
headers = {'User-Agent': USER_AGENT}
|
||||||
|
|
||||||
for service in services:
|
for service in services:
|
||||||
data = request.request_content(service % torrent_hash)
|
|
||||||
|
|
||||||
|
data = request.request_content(service % torrent_hash, headers=headers)
|
||||||
if data and "torcache" in data:
|
if data and "torcache" in data:
|
||||||
if not torrent_to_file(download_path, data):
|
if not torrent_to_file(download_path, data):
|
||||||
return
|
return
|
||||||
@@ -805,15 +817,9 @@ def send_to_downloader(data, bestqual, album):
|
|||||||
"to open or convert magnet links")
|
"to open or convert magnet links")
|
||||||
return
|
return
|
||||||
else:
|
else:
|
||||||
if bestqual[3] == "rutracker.org":
|
|
||||||
download_path, _ = rutracker.get_torrent(bestqual[2],
|
|
||||||
headphones.CONFIG.TORRENTBLACKHOLE_DIR)
|
|
||||||
|
|
||||||
if not download_path:
|
if not torrent_to_file(download_path, data):
|
||||||
return
|
return
|
||||||
else:
|
|
||||||
if not torrent_to_file(download_path, data):
|
|
||||||
return
|
|
||||||
|
|
||||||
# Extract folder name from torrent
|
# Extract folder name from torrent
|
||||||
folder_name = read_torrent_name(download_path, bestqual[0])
|
folder_name = read_torrent_name(download_path, bestqual[0])
|
||||||
@@ -823,13 +829,11 @@ def send_to_downloader(data, bestqual, album):
|
|||||||
elif headphones.CONFIG.TORRENT_DOWNLOADER == 1:
|
elif headphones.CONFIG.TORRENT_DOWNLOADER == 1:
|
||||||
logger.info("Sending torrent to Transmission")
|
logger.info("Sending torrent to Transmission")
|
||||||
|
|
||||||
# rutracker needs cookies to be set, pass the .torrent file instead of url
|
# Add torrent
|
||||||
if bestqual[3] == 'rutracker.org':
|
if bestqual[3] == 'rutracker.org':
|
||||||
file_or_url, torrentid = rutracker.get_torrent(bestqual[2])
|
torrentid = transmission.addTorrent('', data)
|
||||||
else:
|
else:
|
||||||
file_or_url = bestqual[2]
|
torrentid = transmission.addTorrent(bestqual[2])
|
||||||
|
|
||||||
torrentid = transmission.addTorrent(file_or_url)
|
|
||||||
|
|
||||||
if not torrentid:
|
if not torrentid:
|
||||||
logger.error("Error sending torrent to Transmission. Are you sure it's running?")
|
logger.error("Error sending torrent to Transmission. Are you sure it's running?")
|
||||||
@@ -842,13 +846,6 @@ def send_to_downloader(data, bestqual, album):
|
|||||||
logger.error('Torrent folder name could not be determined')
|
logger.error('Torrent folder name could not be determined')
|
||||||
return
|
return
|
||||||
|
|
||||||
# remove temp .torrent file created above
|
|
||||||
if bestqual[3] == 'rutracker.org':
|
|
||||||
try:
|
|
||||||
shutil.rmtree(os.path.split(file_or_url)[0])
|
|
||||||
except Exception as e:
|
|
||||||
logger.exception("Unhandled exception")
|
|
||||||
|
|
||||||
# Set Seed Ratio
|
# Set Seed Ratio
|
||||||
seed_ratio = get_seed_ratio(bestqual[3])
|
seed_ratio = get_seed_ratio(bestqual[3])
|
||||||
if seed_ratio is not None:
|
if seed_ratio is not None:
|
||||||
@@ -857,29 +854,29 @@ def send_to_downloader(data, bestqual, album):
|
|||||||
else:# if headphones.CONFIG.TORRENT_DOWNLOADER == 2:
|
else:# if headphones.CONFIG.TORRENT_DOWNLOADER == 2:
|
||||||
logger.info("Sending torrent to uTorrent")
|
logger.info("Sending torrent to uTorrent")
|
||||||
|
|
||||||
# rutracker needs cookies to be set, pass the .torrent file instead of url
|
# Add torrent
|
||||||
if bestqual[3] == 'rutracker.org':
|
if bestqual[3] == 'rutracker.org':
|
||||||
file_or_url, torrentid = rutracker.get_torrent(bestqual[2])
|
ruobj.utorrent_add_file(data)
|
||||||
folder_name, cacheid = utorrent.dirTorrent(torrentid)
|
|
||||||
folder_name = os.path.basename(os.path.normpath(folder_name))
|
|
||||||
utorrent.labelTorrent(torrentid)
|
|
||||||
else:
|
else:
|
||||||
file_or_url = bestqual[2]
|
utorrent.addTorrent(bestqual[2])
|
||||||
torrentid = calculate_torrent_hash(file_or_url, data)
|
|
||||||
folder_name = utorrent.addTorrent(file_or_url, torrentid)
|
|
||||||
|
|
||||||
|
# Get hash
|
||||||
|
torrentid = calculate_torrent_hash(bestqual[2], data)
|
||||||
|
if not torrentid:
|
||||||
|
logger.error('Torrent id could not be determined')
|
||||||
|
return
|
||||||
|
|
||||||
|
# Get folder
|
||||||
|
folder_name = utorrent.getFolder(torrentid)
|
||||||
if folder_name:
|
if folder_name:
|
||||||
logger.info('Torrent folder name: %s' % folder_name)
|
logger.info('Torrent folder name: %s' % folder_name)
|
||||||
else:
|
else:
|
||||||
logger.error('Torrent folder name could not be determined')
|
logger.error('Torrent folder name could not be determined')
|
||||||
return
|
return
|
||||||
|
|
||||||
# remove temp .torrent file created above
|
# Set Label
|
||||||
if bestqual[3] == 'rutracker.org':
|
if headphones.CONFIG.UTORRENT_LABEL:
|
||||||
try:
|
utorrent.labelTorrent(torrentid)
|
||||||
shutil.rmtree(os.path.split(file_or_url)[0])
|
|
||||||
except Exception as e:
|
|
||||||
logger.exception("Unhandled exception")
|
|
||||||
|
|
||||||
# Set Seed Ratio
|
# Set Seed Ratio
|
||||||
seed_ratio = get_seed_ratio(bestqual[3])
|
seed_ratio = get_seed_ratio(bestqual[3])
|
||||||
@@ -919,7 +916,7 @@ def send_to_downloader(data, bestqual, album):
|
|||||||
if headphones.CONFIG.PUSHBULLET_ENABLED and headphones.CONFIG.PUSHBULLET_ONSNATCH:
|
if headphones.CONFIG.PUSHBULLET_ENABLED and headphones.CONFIG.PUSHBULLET_ONSNATCH:
|
||||||
logger.info(u"Sending PushBullet notification")
|
logger.info(u"Sending PushBullet notification")
|
||||||
pushbullet = notifiers.PUSHBULLET()
|
pushbullet = notifiers.PUSHBULLET()
|
||||||
pushbullet.notify(name + " has been snatched!", "Download started")
|
pushbullet.notify(name, "Download started")
|
||||||
if headphones.CONFIG.TWITTER_ENABLED and headphones.CONFIG.TWITTER_ONSNATCH:
|
if headphones.CONFIG.TWITTER_ENABLED and headphones.CONFIG.TWITTER_ONSNATCH:
|
||||||
logger.info(u"Sending Twitter notification")
|
logger.info(u"Sending Twitter notification")
|
||||||
twitter = notifiers.TwitterNotifier()
|
twitter = notifiers.TwitterNotifier()
|
||||||
@@ -1028,12 +1025,7 @@ def verifyresult(title, artistterm, term, lossless):
|
|||||||
|
|
||||||
def searchTorrent(album, new=False, losslessOnly=False, albumlength=None, choose_specific_download=False):
|
def searchTorrent(album, new=False, losslessOnly=False, albumlength=None, choose_specific_download=False):
|
||||||
global gazelle # persistent what.cd api object to reduce number of login attempts
|
global gazelle # persistent what.cd api object to reduce number of login attempts
|
||||||
|
global ruobj # and rutracker
|
||||||
# rutracker login
|
|
||||||
if headphones.CONFIG.RUTRACKER and album:
|
|
||||||
rulogin = rutracker.login(headphones.CONFIG.RUTRACKER_USER, headphones.CONFIG.RUTRACKER_PASSWORD)
|
|
||||||
if not rulogin:
|
|
||||||
logger.info(u'Could not login to rutracker, search results will exclude this provider')
|
|
||||||
|
|
||||||
albumid = album['AlbumID']
|
albumid = album['AlbumID']
|
||||||
reldate = album['ReleaseDate']
|
reldate = album['ReleaseDate']
|
||||||
@@ -1097,6 +1089,68 @@ def searchTorrent(album, new=False, losslessOnly=False, albumlength=None, choose
|
|||||||
|
|
||||||
return proxy_url
|
return proxy_url
|
||||||
|
|
||||||
|
if headphones.CONFIG.TORZNAB:
|
||||||
|
provider = "torznab"
|
||||||
|
torznab_hosts = []
|
||||||
|
|
||||||
|
if headphones.CONFIG.TORZNAB_HOST and headphones.CONFIG.TORZNAB_ENABLED:
|
||||||
|
torznab_hosts.append((headphones.CONFIG.TORZNAB_HOST, headphones.CONFIG.TORZNAB_APIKEY, headphones.CONFIG.TORZNAB_ENABLED))
|
||||||
|
|
||||||
|
for torznab_host in headphones.CONFIG.get_extra_torznabs():
|
||||||
|
if torznab_host[2] == '1' or torznab_host[2] == 1:
|
||||||
|
torznab_hosts.append(torznab_host)
|
||||||
|
|
||||||
|
if headphones.CONFIG.PREFERRED_QUALITY == 3 or losslessOnly:
|
||||||
|
categories = "3040"
|
||||||
|
elif headphones.CONFIG.PREFERRED_QUALITY == 1 or allow_lossless:
|
||||||
|
categories = "3040,3010"
|
||||||
|
else:
|
||||||
|
categories = "3010"
|
||||||
|
|
||||||
|
if album['Type'] == 'Other':
|
||||||
|
categories = "3030"
|
||||||
|
logger.info("Album type is audiobook/spokenword. Using audiobook category")
|
||||||
|
|
||||||
|
for torznab_host in torznab_hosts:
|
||||||
|
|
||||||
|
provider = torznab_host[0]
|
||||||
|
|
||||||
|
# Request results
|
||||||
|
logger.info('Parsing results from %s using search term: %s' % (torznab_host[0],term))
|
||||||
|
|
||||||
|
headers = {'User-Agent': USER_AGENT}
|
||||||
|
params = {
|
||||||
|
"t": "search",
|
||||||
|
"apikey": torznab_host[1],
|
||||||
|
"cat": categories,
|
||||||
|
"maxage": headphones.CONFIG.USENET_RETENTION,
|
||||||
|
"q": term
|
||||||
|
}
|
||||||
|
|
||||||
|
data = request.request_feed(
|
||||||
|
url=torznab_host[0] + '/api?',
|
||||||
|
params=params, headers=headers
|
||||||
|
)
|
||||||
|
|
||||||
|
# Process feed
|
||||||
|
if data:
|
||||||
|
if not len(data.entries):
|
||||||
|
logger.info(u"No results found from %s for %s", torznab_host[0], term)
|
||||||
|
else:
|
||||||
|
for item in data.entries:
|
||||||
|
try:
|
||||||
|
url = item.link
|
||||||
|
title = item.title
|
||||||
|
size = int(item.links[1]['length'])
|
||||||
|
if all(word.lower() in title.lower() for word in term.split()):
|
||||||
|
logger.info('Found %s. Size: %s' % (title, helpers.bytes_to_mb(size)))
|
||||||
|
resultlist.append((title, size, url, provider, 'torrent', True))
|
||||||
|
else:
|
||||||
|
logger.info('Skipping %s, not all search term words found' % title)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.exception("An unknown error occurred trying to parse the feed: %s" % e)
|
||||||
|
|
||||||
if headphones.CONFIG.KAT:
|
if headphones.CONFIG.KAT:
|
||||||
provider = "Kick Ass Torrents"
|
provider = "Kick Ass Torrents"
|
||||||
ka_term = term.replace("!", "")
|
ka_term = term.replace("!", "")
|
||||||
@@ -1129,7 +1183,8 @@ def searchTorrent(album, new=False, losslessOnly=False, albumlength=None, choose
|
|||||||
"field": "seeders",
|
"field": "seeders",
|
||||||
"sorder": "desc"
|
"sorder": "desc"
|
||||||
}
|
}
|
||||||
data = request.request_json(url=providerurl, params=params)
|
headers = {'User-Agent': USER_AGENT}
|
||||||
|
data = request.request_json(url=providerurl, params=params, headers=headers)
|
||||||
|
|
||||||
# Process feed
|
# Process feed
|
||||||
if data:
|
if data:
|
||||||
@@ -1145,7 +1200,7 @@ def searchTorrent(album, new=False, losslessOnly=False, albumlength=None, choose
|
|||||||
size = int(item['size'])
|
size = int(item['size'])
|
||||||
|
|
||||||
if format == "2":
|
if format == "2":
|
||||||
torrent = request.request_content(url)
|
torrent = request.request_content(url, headers=headers)
|
||||||
if not torrent or (int(torrent.find(".mp3")) > 0 and int(torrent.find(".flac")) < 1):
|
if not torrent or (int(torrent.find(".mp3")) > 0 and int(torrent.find(".flac")) < 1):
|
||||||
rightformat = False
|
rightformat = False
|
||||||
|
|
||||||
@@ -1226,45 +1281,38 @@ def searchTorrent(album, new=False, losslessOnly=False, albumlength=None, choose
|
|||||||
logger.error(u"An error occurred while trying to parse the response from Waffles.fm: %s", e)
|
logger.error(u"An error occurred while trying to parse the response from Waffles.fm: %s", e)
|
||||||
|
|
||||||
# rutracker.org
|
# rutracker.org
|
||||||
if headphones.CONFIG.RUTRACKER and rulogin:
|
if headphones.CONFIG.RUTRACKER:
|
||||||
provider = "rutracker.org"
|
provider = "rutracker.org"
|
||||||
|
|
||||||
# Ignore if release date not specified, results too unpredictable
|
# Ignore if release date not specified, results too unpredictable
|
||||||
if not year and not usersearchterm:
|
if not year and not usersearchterm:
|
||||||
logger.info(u'Release date not specified, ignoring for rutracker.org')
|
logger.info(u"Release date not specified, ignoring for rutracker.org")
|
||||||
else:
|
else:
|
||||||
|
|
||||||
if headphones.CONFIG.PREFERRED_QUALITY == 3 or losslessOnly:
|
if headphones.CONFIG.PREFERRED_QUALITY == 3 or losslessOnly:
|
||||||
format = 'lossless'
|
format = 'lossless'
|
||||||
maxsize = 10000000000
|
|
||||||
elif headphones.CONFIG.PREFERRED_QUALITY == 1 or allow_lossless:
|
elif headphones.CONFIG.PREFERRED_QUALITY == 1 or allow_lossless:
|
||||||
format = 'lossless+mp3'
|
format = 'lossless+mp3'
|
||||||
maxsize = 10000000000
|
|
||||||
else:
|
else:
|
||||||
format = 'mp3'
|
format = 'mp3'
|
||||||
maxsize = 300000000
|
|
||||||
|
|
||||||
# build search url based on above
|
# Login
|
||||||
if not usersearchterm:
|
if not ruobj or not ruobj.logged_in():
|
||||||
searchURL = rutracker.searchurl(artistterm, albumterm, year, format)
|
ruobj = rutracker.Rutracker()
|
||||||
else:
|
if not ruobj.login():
|
||||||
searchURL = rutracker.searchurl(usersearchterm, ' ', ' ', format)
|
ruobj = None
|
||||||
|
|
||||||
logger.info(u'Parsing results from <a href="%s">rutracker.org</a>' % searchURL)
|
if ruobj and ruobj.logged_in():
|
||||||
|
|
||||||
# parse results and get best match
|
# build search url
|
||||||
rulist = rutracker.search(searchURL, maxsize, minimumseeders, albumid)
|
if not usersearchterm:
|
||||||
|
searchURL = ruobj.searchurl(artistterm, albumterm, year, format)
|
||||||
|
else:
|
||||||
|
searchURL = ruobj.searchurl(usersearchterm, ' ', ' ', format)
|
||||||
|
|
||||||
# add best match to overall results list
|
# parse results
|
||||||
if rulist:
|
rulist = ruobj.search(searchURL)
|
||||||
for ru in rulist:
|
if rulist:
|
||||||
title = ru[0].decode('utf-8')
|
resultlist.extend(rulist)
|
||||||
size = ru[1]
|
|
||||||
url = ru[2]
|
|
||||||
resultlist.append((title, size, url, provider, 'torrent', True))
|
|
||||||
logger.info('Found %s. Size: %s' % (title, helpers.bytes_to_mb(size)))
|
|
||||||
else:
|
|
||||||
logger.info(u"No valid results found from %s" % (provider))
|
|
||||||
|
|
||||||
if headphones.CONFIG.WHATCD:
|
if headphones.CONFIG.WHATCD:
|
||||||
provider = "What.cd"
|
provider = "What.cd"
|
||||||
@@ -1280,6 +1328,12 @@ def searchTorrent(album, new=False, losslessOnly=False, albumlength=None, choose
|
|||||||
search_formats = [None] # should return all
|
search_formats = [None] # should return all
|
||||||
bitrate = headphones.CONFIG.PREFERRED_BITRATE
|
bitrate = headphones.CONFIG.PREFERRED_BITRATE
|
||||||
if bitrate:
|
if bitrate:
|
||||||
|
if 225 <= int(bitrate) < 256:
|
||||||
|
bitrate = 'V0'
|
||||||
|
elif 200 <= int(bitrate) < 225:
|
||||||
|
bitrate = 'V1'
|
||||||
|
elif 175 <= int(bitrate) < 200:
|
||||||
|
bitrate = 'V2'
|
||||||
for encoding_string in gazelleencoding.ALL_ENCODINGS:
|
for encoding_string in gazelleencoding.ALL_ENCODINGS:
|
||||||
if re.search(bitrate, encoding_string, flags=re.I):
|
if re.search(bitrate, encoding_string, flags=re.I):
|
||||||
bitrate_string = encoding_string
|
bitrate_string = encoding_string
|
||||||
@@ -1306,7 +1360,10 @@ def searchTorrent(album, new=False, losslessOnly=False, albumlength=None, choose
|
|||||||
logger.info(u"Searching %s..." % provider)
|
logger.info(u"Searching %s..." % provider)
|
||||||
all_torrents = []
|
all_torrents = []
|
||||||
for search_format in search_formats:
|
for search_format in search_formats:
|
||||||
all_torrents.extend(gazelle.search_torrents(artistname=semi_clean_artist_term,
|
if usersearchterm:
|
||||||
|
all_torrents.extend(gazelle.search_torrents(searchstr=usersearchterm, format=search_format, encoding=bitrate_string)['results'])
|
||||||
|
else:
|
||||||
|
all_torrents.extend(gazelle.search_torrents(artistname=semi_clean_artist_term,
|
||||||
groupname=semi_clean_album_term,
|
groupname=semi_clean_album_term,
|
||||||
format=search_format, encoding=bitrate_string)['results'])
|
format=search_format, encoding=bitrate_string)['results'])
|
||||||
|
|
||||||
@@ -1469,6 +1526,57 @@ def searchTorrent(album, new=False, losslessOnly=False, albumlength=None, choose
|
|||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.error(u"An unknown error occurred in the Old Pirate Bay parser: %s" % e)
|
logger.error(u"An unknown error occurred in the Old Pirate Bay parser: %s" % e)
|
||||||
|
|
||||||
|
# Strike
|
||||||
|
if headphones.CONFIG.STRIKE:
|
||||||
|
provider = "Strike"
|
||||||
|
s_term = term.replace("!", "")
|
||||||
|
providerurl = fix_url("https://getstrike.net/api/v2/torrents/search/?phrase=")
|
||||||
|
|
||||||
|
providerurl = providerurl + s_term + "&category=Music"
|
||||||
|
|
||||||
|
if headphones.CONFIG.PREFERRED_QUALITY == 3 or losslessOnly:
|
||||||
|
format = "2"
|
||||||
|
providerurl = providerurl + "&subcategory=Lossless"
|
||||||
|
maxsize = 10000000000
|
||||||
|
elif headphones.CONFIG.PREFERRED_QUALITY == 1 or allow_lossless:
|
||||||
|
format = "10" # MP3 and FLAC
|
||||||
|
maxsize = 10000000000
|
||||||
|
else:
|
||||||
|
format = "8" # MP3 only
|
||||||
|
maxsize = 300000000
|
||||||
|
|
||||||
|
logger.info("Searching %s using term: %s" % (provider, s_term))
|
||||||
|
data = request.request_json(url=providerurl)
|
||||||
|
|
||||||
|
if not data or not data.get('torrents'):
|
||||||
|
logger.info("No results found on %s using search term: %s" % (provider, s_term))
|
||||||
|
else:
|
||||||
|
for item in data['torrents']:
|
||||||
|
try:
|
||||||
|
rightformat = True
|
||||||
|
title = item['torrent_title']
|
||||||
|
seeders = item['seeds']
|
||||||
|
url = item['magnet_uri']
|
||||||
|
size = int(item['size'])
|
||||||
|
subcategory = item['sub_category']
|
||||||
|
|
||||||
|
if format == 2:
|
||||||
|
if subcategory != "Lossless":
|
||||||
|
rightformat = False
|
||||||
|
|
||||||
|
if rightformat and size < maxsize and minimumseeders < int(seeders):
|
||||||
|
match = True
|
||||||
|
logger.info('Found %s. Size: %s' % (title, helpers.bytes_to_mb(size)))
|
||||||
|
else:
|
||||||
|
match = False
|
||||||
|
logger.info(
|
||||||
|
'%s is larger than the maxsize, the wrong format or has too little seeders for this category, skipping. (Size: %i bytes, Seeders: %d, Format: %s)',
|
||||||
|
title, size, int(seeders), rightformat)
|
||||||
|
|
||||||
|
resultlist.append((title, size, url, provider, 'torrent', match))
|
||||||
|
except Exception as e:
|
||||||
|
logger.exception("Unhandled exception in the Strike parser")
|
||||||
|
|
||||||
# Mininova
|
# Mininova
|
||||||
if headphones.CONFIG.MININOVA:
|
if headphones.CONFIG.MININOVA:
|
||||||
provider = "Mininova"
|
provider = "Mininova"
|
||||||
@@ -1545,12 +1653,14 @@ def preprocess(resultlist):
|
|||||||
|
|
||||||
for result in resultlist:
|
for result in resultlist:
|
||||||
if result[4] == 'torrent':
|
if result[4] == 'torrent':
|
||||||
|
|
||||||
|
# rutracker always needs the torrent data
|
||||||
|
if result[3] == 'rutracker.org':
|
||||||
|
return ruobj.get_torrent_data(result[2]), result
|
||||||
|
|
||||||
#Get out of here if we're using Transmission
|
#Get out of here if we're using Transmission
|
||||||
if headphones.CONFIG.TORRENT_DOWNLOADER == 1: ## if not a magnet link still need the .torrent to generate hash... uTorrent support labeling
|
if headphones.CONFIG.TORRENT_DOWNLOADER == 1: ## if not a magnet link still need the .torrent to generate hash... uTorrent support labeling
|
||||||
return True, result
|
return True, result
|
||||||
# get outta here if rutracker
|
|
||||||
if result[3] == 'rutracker.org':
|
|
||||||
return True, result
|
|
||||||
# Get out of here if it's a magnet link
|
# Get out of here if it's a magnet link
|
||||||
if result[2].lower().startswith("magnet:"):
|
if result[2].lower().startswith("magnet:"):
|
||||||
return True, result
|
return True, result
|
||||||
@@ -1559,7 +1669,8 @@ def preprocess(resultlist):
|
|||||||
headers = {}
|
headers = {}
|
||||||
|
|
||||||
if result[3] == 'Kick Ass Torrents':
|
if result[3] == 'Kick Ass Torrents':
|
||||||
headers['Referer'] = 'http://kat.ph/'
|
#headers['Referer'] = 'http://kat.ph/'
|
||||||
|
headers['User-Agent'] = USER_AGENT
|
||||||
elif result[3] == 'What.cd':
|
elif result[3] == 'What.cd':
|
||||||
headers['User-Agent'] = 'Headphones'
|
headers['User-Agent'] = 'Headphones'
|
||||||
elif result[3] == "The Pirate Bay" or result[3] == "Old Pirate Bay":
|
elif result[3] == "The Pirate Bay" or result[3] == "Old Pirate Bay":
|
||||||
|
|||||||
@@ -1,349 +0,0 @@
|
|||||||
#!/usr/bin/env python
|
|
||||||
# coding=utf-8
|
|
||||||
|
|
||||||
# Headphones rutracker.org search
|
|
||||||
# Functions called from searcher.py
|
|
||||||
|
|
||||||
from bencode import bencode as bencode, bdecode
|
|
||||||
from urlparse import urlparse
|
|
||||||
from bs4 import BeautifulSoup
|
|
||||||
from tempfile import mkdtemp
|
|
||||||
from hashlib import sha1
|
|
||||||
|
|
||||||
import headphones
|
|
||||||
import requests
|
|
||||||
import cookielib
|
|
||||||
import urllib2
|
|
||||||
import urllib
|
|
||||||
import re
|
|
||||||
import os
|
|
||||||
|
|
||||||
from headphones import db, logger
|
|
||||||
|
|
||||||
|
|
||||||
class Rutracker():
|
|
||||||
|
|
||||||
logged_in = False
|
|
||||||
|
|
||||||
# Stores a number of login attempts to prevent recursion.
|
|
||||||
#login_counter = 0
|
|
||||||
|
|
||||||
def __init__(self):
|
|
||||||
|
|
||||||
self.cookiejar = cookielib.CookieJar()
|
|
||||||
self.opener = urllib2.build_opener(urllib2.HTTPCookieProcessor(self.cookiejar))
|
|
||||||
urllib2.install_opener(self.opener)
|
|
||||||
|
|
||||||
def login(self, login, password):
|
|
||||||
"""Implements tracker login procedure."""
|
|
||||||
|
|
||||||
self.logged_in = False
|
|
||||||
|
|
||||||
if login is None or password is None:
|
|
||||||
return False
|
|
||||||
|
|
||||||
#self.login_counter += 1
|
|
||||||
|
|
||||||
# No recursion wanted.
|
|
||||||
#if self.login_counter > 1:
|
|
||||||
# return False
|
|
||||||
|
|
||||||
params = urllib.urlencode({"login_username": login,
|
|
||||||
"login_password": password,
|
|
||||||
"login": "Вход"})
|
|
||||||
|
|
||||||
try:
|
|
||||||
self.opener.open("http://login.rutracker.org/forum/login.php", params)
|
|
||||||
except Exception:
|
|
||||||
pass
|
|
||||||
|
|
||||||
# Check if we're logged in
|
|
||||||
for cookie in self.cookiejar:
|
|
||||||
if cookie.name == 'bb_data':
|
|
||||||
self.logged_in = True
|
|
||||||
|
|
||||||
return self.logged_in
|
|
||||||
|
|
||||||
def searchurl(self, artist, album, year, format):
|
|
||||||
"""
|
|
||||||
Return the search url
|
|
||||||
"""
|
|
||||||
|
|
||||||
# Build search url
|
|
||||||
searchterm = ''
|
|
||||||
if artist != 'Various Artists':
|
|
||||||
searchterm = artist
|
|
||||||
searchterm = searchterm + ' '
|
|
||||||
searchterm = searchterm + album
|
|
||||||
searchterm = searchterm + ' '
|
|
||||||
searchterm = searchterm + year
|
|
||||||
|
|
||||||
providerurl = "http://rutracker.org/forum/tracker.php"
|
|
||||||
|
|
||||||
if format == 'lossless':
|
|
||||||
format = '+lossless'
|
|
||||||
elif format == 'lossless+mp3':
|
|
||||||
format = '+lossless||mp3||aac'
|
|
||||||
else:
|
|
||||||
format = '+mp3||aac'
|
|
||||||
|
|
||||||
# sort by size, descending.
|
|
||||||
sort = '&o=7&s=2'
|
|
||||||
|
|
||||||
searchurl = "%s?nm=%s%s%s" % (providerurl, urllib.quote(searchterm), format, sort)
|
|
||||||
|
|
||||||
return searchurl
|
|
||||||
|
|
||||||
def search(self, searchurl, maxsize, minseeders, albumid):
|
|
||||||
"""
|
|
||||||
Parse the search results and return valid torrent list
|
|
||||||
"""
|
|
||||||
|
|
||||||
titles = []
|
|
||||||
urls = []
|
|
||||||
seeders = []
|
|
||||||
sizes = []
|
|
||||||
torrentlist = []
|
|
||||||
rulist = []
|
|
||||||
|
|
||||||
try:
|
|
||||||
|
|
||||||
page = self.opener.open(searchurl, timeout=60)
|
|
||||||
soup = BeautifulSoup(page.read())
|
|
||||||
|
|
||||||
# Debug
|
|
||||||
#logger.debug (soup.prettify())
|
|
||||||
|
|
||||||
# Title
|
|
||||||
for link in soup.find_all('a', attrs={'class': 'med tLink hl-tags bold'}):
|
|
||||||
title = link.get_text()
|
|
||||||
titles.append(title)
|
|
||||||
|
|
||||||
# Download URL
|
|
||||||
for link in soup.find_all('a', attrs={'class': 'small tr-dl dl-stub'}):
|
|
||||||
url = link.get('href')
|
|
||||||
urls.append(url)
|
|
||||||
|
|
||||||
# Seeders
|
|
||||||
for link in soup.find_all('b', attrs={'class': 'seedmed'}):
|
|
||||||
seeder = link.get_text()
|
|
||||||
seeders.append(seeder)
|
|
||||||
|
|
||||||
# Size
|
|
||||||
for link in soup.find_all('td', attrs={'class': 'row4 small nowrap tor-size'}):
|
|
||||||
size = link.u.string
|
|
||||||
sizes.append(size)
|
|
||||||
|
|
||||||
except:
|
|
||||||
pass
|
|
||||||
|
|
||||||
# Combine lists
|
|
||||||
torrentlist = zip(titles, urls, seeders, sizes)
|
|
||||||
|
|
||||||
# return if nothing found
|
|
||||||
if not torrentlist:
|
|
||||||
return False
|
|
||||||
|
|
||||||
# don't bother checking track counts anymore, let searcher filter instead
|
|
||||||
# leave code in just in case
|
|
||||||
check_track_count = False
|
|
||||||
|
|
||||||
if check_track_count:
|
|
||||||
|
|
||||||
# get headphones track count for album, return if not found
|
|
||||||
myDB = db.DBConnection()
|
|
||||||
tracks = myDB.select('SELECT * from tracks WHERE AlbumID=?', [albumid])
|
|
||||||
hptrackcount = len(tracks)
|
|
||||||
|
|
||||||
if not hptrackcount:
|
|
||||||
logger.info('headphones track info not found, cannot compare to torrent')
|
|
||||||
return False
|
|
||||||
|
|
||||||
# Return all valid entries, ignored, required words now checked in searcher.py
|
|
||||||
|
|
||||||
#unwantedlist = ['promo', 'vinyl', '[lp]', 'songbook', 'tvrip', 'hdtv', 'dvd']
|
|
||||||
|
|
||||||
formatlist = ['ape', 'flac', 'ogg', 'm4a', 'aac', 'mp3', 'wav', 'aif']
|
|
||||||
deluxelist = ['deluxe', 'edition', 'japanese', 'exclusive']
|
|
||||||
|
|
||||||
for torrent in torrentlist:
|
|
||||||
|
|
||||||
returntitle = torrent[0].encode('utf-8')
|
|
||||||
url = torrent[1]
|
|
||||||
seeders = torrent[2]
|
|
||||||
size = torrent[3]
|
|
||||||
|
|
||||||
if int(size) <= maxsize and int(seeders) >= minseeders:
|
|
||||||
|
|
||||||
#Torrent topic page
|
|
||||||
torrent_id = dict([part.split('=') for part in urlparse(url)[4].split('&')])['t']
|
|
||||||
topicurl = 'http://rutracker.org/forum/viewtopic.php?t=' + torrent_id
|
|
||||||
|
|
||||||
# add to list
|
|
||||||
if not check_track_count:
|
|
||||||
valid = True
|
|
||||||
else:
|
|
||||||
|
|
||||||
# Check torrent info
|
|
||||||
self.cookiejar.set_cookie(cookielib.Cookie(version=0, name='bb_dl', value=torrent_id, port=None, port_specified=False, domain='.rutracker.org', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False))
|
|
||||||
|
|
||||||
# Debug
|
|
||||||
#for cookie in self.cookiejar:
|
|
||||||
# logger.debug ('Cookie: %s' % cookie)
|
|
||||||
|
|
||||||
try:
|
|
||||||
page = self.opener.open(url)
|
|
||||||
torrent = page.read()
|
|
||||||
if torrent:
|
|
||||||
decoded = bdecode(torrent)
|
|
||||||
metainfo = decoded['info']
|
|
||||||
page.close()
|
|
||||||
except Exception as e:
|
|
||||||
logger.error('Error getting torrent: %s' % e)
|
|
||||||
return False
|
|
||||||
|
|
||||||
# get torrent track count and check for cue
|
|
||||||
trackcount = 0
|
|
||||||
cuecount = 0
|
|
||||||
|
|
||||||
if 'files' in metainfo: # multi
|
|
||||||
for pathfile in metainfo['files']:
|
|
||||||
path = pathfile['path']
|
|
||||||
for file in path:
|
|
||||||
if any(file.lower().endswith('.' + x.lower()) for x in formatlist):
|
|
||||||
trackcount += 1
|
|
||||||
if '.cue' in file:
|
|
||||||
cuecount += 1
|
|
||||||
|
|
||||||
title = returntitle.lower()
|
|
||||||
logger.debug('torrent title: %s' % title)
|
|
||||||
logger.debug('headphones trackcount: %s' % hptrackcount)
|
|
||||||
logger.debug('rutracker trackcount: %s' % trackcount)
|
|
||||||
|
|
||||||
# If torrent track count less than headphones track count, and there's a cue, then attempt to get track count from log(s)
|
|
||||||
# This is for the case where we have a single .flac/.wav which can be split by cue
|
|
||||||
# Not great, but shouldn't be doing this too often
|
|
||||||
totallogcount = 0
|
|
||||||
if trackcount < hptrackcount and cuecount > 0 and cuecount < hptrackcount:
|
|
||||||
page = self.opener.open(topicurl, timeout=60)
|
|
||||||
soup = BeautifulSoup(page.read())
|
|
||||||
findtoc = soup.find_all(text='TOC of the extracted CD')
|
|
||||||
if not findtoc:
|
|
||||||
findtoc = soup.find_all(text='TOC извлечённого CD')
|
|
||||||
for toc in findtoc:
|
|
||||||
logcount = 0
|
|
||||||
for toccontent in toc.find_all_next(text=True):
|
|
||||||
cut_string = toccontent.split('|')
|
|
||||||
new_string = cut_string[0].lstrip().rstrip()
|
|
||||||
if new_string == '1' or new_string == '01':
|
|
||||||
logcount = 1
|
|
||||||
elif logcount > 0:
|
|
||||||
if new_string.isdigit():
|
|
||||||
logcount += 1
|
|
||||||
else:
|
|
||||||
break
|
|
||||||
totallogcount = totallogcount + logcount
|
|
||||||
|
|
||||||
if totallogcount > 0:
|
|
||||||
trackcount = totallogcount
|
|
||||||
logger.debug('rutracker logtrackcount: %s' % totallogcount)
|
|
||||||
|
|
||||||
# If torrent track count = hp track count then return torrent,
|
|
||||||
# if greater, check for deluxe/special/foreign editions
|
|
||||||
# if less, then allow if it's a single track with a cue
|
|
||||||
valid = False
|
|
||||||
|
|
||||||
if trackcount == hptrackcount:
|
|
||||||
valid = True
|
|
||||||
elif trackcount > hptrackcount:
|
|
||||||
if any(deluxe in title for deluxe in deluxelist):
|
|
||||||
valid = True
|
|
||||||
|
|
||||||
# Add to list
|
|
||||||
if valid:
|
|
||||||
rulist.append((returntitle, size, topicurl))
|
|
||||||
else:
|
|
||||||
if topicurl:
|
|
||||||
logger.info(u'<a href="%s">Torrent</a> found with %s tracks but the selected headphones release has %s tracks, skipping for rutracker.org' % (topicurl, trackcount, hptrackcount))
|
|
||||||
else:
|
|
||||||
logger.info('%s is larger than the maxsize or has too little seeders for this category, skipping. (Size: %i bytes, Seeders: %i)' % (returntitle, int(size), int(seeders)))
|
|
||||||
|
|
||||||
return rulist
|
|
||||||
|
|
||||||
def get_torrent(self, url, savelocation=None):
|
|
||||||
|
|
||||||
torrent_id = dict([part.split('=') for part in urlparse(url)[4].split('&')])['t']
|
|
||||||
self.cookiejar.set_cookie(cookielib.Cookie(version=0, name='bb_dl', value=torrent_id, port=None, port_specified=False, domain='.rutracker.org', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False))
|
|
||||||
downloadurl = 'http://dl.rutracker.org/forum/dl.php?t=' + torrent_id
|
|
||||||
torrent_name = torrent_id + '.torrent'
|
|
||||||
|
|
||||||
try:
|
|
||||||
prev = os.umask(headphones.UMASK)
|
|
||||||
page = self.opener.open(downloadurl)
|
|
||||||
torrent = page.read()
|
|
||||||
decoded = bdecode(torrent)
|
|
||||||
metainfo = decoded['info']
|
|
||||||
tor_hash = sha1(bencode(metainfo)).hexdigest()
|
|
||||||
if savelocation:
|
|
||||||
download_path = os.path.join(savelocation, torrent_name)
|
|
||||||
else:
|
|
||||||
tempdir = mkdtemp(suffix='_rutracker_torrents')
|
|
||||||
download_path = os.path.join(tempdir, torrent_name)
|
|
||||||
|
|
||||||
with open(download_path, 'wb') as f:
|
|
||||||
f.write(torrent)
|
|
||||||
os.umask(prev)
|
|
||||||
|
|
||||||
# Add file to utorrent
|
|
||||||
if headphones.CONFIG.TORRENT_DOWNLOADER == 2:
|
|
||||||
self.utorrent_add_file(download_path)
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
logger.error('Error getting torrent: %s', e)
|
|
||||||
return False
|
|
||||||
|
|
||||||
return download_path, tor_hash
|
|
||||||
|
|
||||||
#TODO get this working in utorrent.py
|
|
||||||
def utorrent_add_file(self, filename):
|
|
||||||
|
|
||||||
host = headphones.CONFIG.UTORRENT_HOST
|
|
||||||
if not host.startswith('http'):
|
|
||||||
host = 'http://' + host
|
|
||||||
if host.endswith('/'):
|
|
||||||
host = host[:-1]
|
|
||||||
if host.endswith('/gui'):
|
|
||||||
host = host[:-4]
|
|
||||||
|
|
||||||
base_url = host
|
|
||||||
username = headphones.CONFIG.UTORRENT_USERNAME
|
|
||||||
password = headphones.CONFIG.UTORRENT_PASSWORD
|
|
||||||
|
|
||||||
session = requests.Session()
|
|
||||||
url = base_url + '/gui/'
|
|
||||||
session.auth = (username, password)
|
|
||||||
|
|
||||||
try:
|
|
||||||
r = session.get(url + 'token.html')
|
|
||||||
except Exception:
|
|
||||||
logger.exception('Error getting token')
|
|
||||||
return
|
|
||||||
|
|
||||||
if r.status_code == '401':
|
|
||||||
logger.debug('Error reaching utorrent')
|
|
||||||
return
|
|
||||||
|
|
||||||
regex = re.search(r'.+>([^<]+)</div></html>', r.text)
|
|
||||||
if regex is None:
|
|
||||||
logger.debug('Error reading token')
|
|
||||||
return
|
|
||||||
|
|
||||||
session.params = {'token': regex.group(1)}
|
|
||||||
|
|
||||||
with open(filename, 'rb') as f:
|
|
||||||
try:
|
|
||||||
session.post(url, params={'action': 'add-file'},
|
|
||||||
files={'torrent_file': f})
|
|
||||||
except Exception:
|
|
||||||
logger.exception('Error adding file to utorrent')
|
|
||||||
return
|
|
||||||
@@ -28,12 +28,15 @@ import headphones
|
|||||||
# Store torrent id so we can check up on it
|
# Store torrent id so we can check up on it
|
||||||
|
|
||||||
|
|
||||||
def addTorrent(link):
|
def addTorrent(link, data=None):
|
||||||
method = 'torrent-add'
|
method = 'torrent-add'
|
||||||
|
|
||||||
if link.endswith('.torrent'):
|
if link.endswith('.torrent') or data:
|
||||||
with open(link, 'rb') as f:
|
if data:
|
||||||
metainfo = str(base64.b64encode(f.read()))
|
metainfo = str(base64.b64encode(data))
|
||||||
|
else:
|
||||||
|
with open(link, 'rb') as f:
|
||||||
|
metainfo = str(base64.b64encode(f.read()))
|
||||||
arguments = {'metainfo': metainfo, 'download-dir': headphones.CONFIG.DOWNLOAD_TORRENT_DIR}
|
arguments = {'metainfo': metainfo, 'download-dir': headphones.CONFIG.DOWNLOAD_TORRENT_DIR}
|
||||||
else:
|
else:
|
||||||
arguments = {'filename': link, 'download-dir': headphones.CONFIG.DOWNLOAD_TORRENT_DIR}
|
arguments = {'filename': link, 'download-dir': headphones.CONFIG.DOWNLOAD_TORRENT_DIR}
|
||||||
|
|||||||
+11
-6
@@ -73,6 +73,7 @@ class utorrentclient(object):
|
|||||||
except urllib2.HTTPError as err:
|
except urllib2.HTTPError as err:
|
||||||
logger.debug('URL: ' + str(url))
|
logger.debug('URL: ' + str(url))
|
||||||
logger.debug('Error getting Token. uTorrent responded with error: ' + str(err))
|
logger.debug('Error getting Token. uTorrent responded with error: ' + str(err))
|
||||||
|
return
|
||||||
match = re.search(utorrentclient.TOKEN_REGEX, response.read())
|
match = re.search(utorrentclient.TOKEN_REGEX, response.read())
|
||||||
return match.group(1)
|
return match.group(1)
|
||||||
|
|
||||||
@@ -147,6 +148,10 @@ class utorrentclient(object):
|
|||||||
return self._action(params)
|
return self._action(params)
|
||||||
|
|
||||||
def _action(self, params, body=None, content_type=None):
|
def _action(self, params, body=None, content_type=None):
|
||||||
|
|
||||||
|
if not self.token:
|
||||||
|
return
|
||||||
|
|
||||||
url = self.base_url + '/gui/' + '?token=' + self.token + '&' + urllib.urlencode(params)
|
url = self.base_url + '/gui/' + '?token=' + self.token + '&' + urllib.urlencode(params)
|
||||||
request = urllib2.Request(url)
|
request = urllib2.Request(url)
|
||||||
|
|
||||||
@@ -215,7 +220,7 @@ def dirTorrent(hash, cacheid=None, return_name=None):
|
|||||||
cacheid = torrentList['torrentc']
|
cacheid = torrentList['torrentc']
|
||||||
|
|
||||||
for torrent in torrents:
|
for torrent in torrents:
|
||||||
if torrent[0].upper() == hash:
|
if torrent[0].upper() == hash.upper():
|
||||||
if not return_name:
|
if not return_name:
|
||||||
return torrent[26], cacheid
|
return torrent[26], cacheid
|
||||||
else:
|
else:
|
||||||
@@ -223,8 +228,12 @@ def dirTorrent(hash, cacheid=None, return_name=None):
|
|||||||
|
|
||||||
return None, None
|
return None, None
|
||||||
|
|
||||||
|
def addTorrent(link):
|
||||||
|
uTorrentClient = utorrentclient()
|
||||||
|
uTorrentClient.add_url(link)
|
||||||
|
|
||||||
def addTorrent(link, hash):
|
|
||||||
|
def getFolder(hash):
|
||||||
uTorrentClient = utorrentclient()
|
uTorrentClient = utorrentclient()
|
||||||
|
|
||||||
# Get Active Directory from settings
|
# Get Active Directory from settings
|
||||||
@@ -234,8 +243,6 @@ def addTorrent(link, hash):
|
|||||||
logger.error('Could not get "Put new downloads in:" directory from uTorrent settings, please ensure it is set')
|
logger.error('Could not get "Put new downloads in:" directory from uTorrent settings, please ensure it is set')
|
||||||
return None
|
return None
|
||||||
|
|
||||||
uTorrentClient.add_url(link)
|
|
||||||
|
|
||||||
# Get Torrent Folder Name
|
# Get Torrent Folder Name
|
||||||
torrent_folder, cacheid = dirTorrent(hash)
|
torrent_folder, cacheid = dirTorrent(hash)
|
||||||
|
|
||||||
@@ -249,10 +256,8 @@ def addTorrent(link, hash):
|
|||||||
|
|
||||||
if torrent_folder == active_dir or not torrent_folder:
|
if torrent_folder == active_dir or not torrent_folder:
|
||||||
torrent_folder, cacheid = dirTorrent(hash, cacheid, return_name=True)
|
torrent_folder, cacheid = dirTorrent(hash, cacheid, return_name=True)
|
||||||
labelTorrent(hash)
|
|
||||||
return torrent_folder
|
return torrent_folder
|
||||||
else:
|
else:
|
||||||
labelTorrent(hash)
|
|
||||||
if headphones.SYS_PLATFORM != "win32":
|
if headphones.SYS_PLATFORM != "win32":
|
||||||
torrent_folder = torrent_folder.replace('\\', '/')
|
torrent_folder = torrent_folder.replace('\\', '/')
|
||||||
return os.path.basename(os.path.normpath(torrent_folder))
|
return os.path.basename(os.path.normpath(torrent_folder))
|
||||||
|
|||||||
+129
-15
@@ -32,8 +32,10 @@ import random
|
|||||||
import urllib
|
import urllib
|
||||||
import json
|
import json
|
||||||
import time
|
import time
|
||||||
|
import cgi
|
||||||
import sys
|
import sys
|
||||||
import os
|
import os
|
||||||
|
import re
|
||||||
|
|
||||||
try:
|
try:
|
||||||
# pylint:disable=E0611
|
# pylint:disable=E0611
|
||||||
@@ -149,7 +151,7 @@ class WebInterface(object):
|
|||||||
searchresults = mb.findRelease(name, limit=100)
|
searchresults = mb.findRelease(name, limit=100)
|
||||||
else:
|
else:
|
||||||
searchresults = mb.findSeries(name, limit=100)
|
searchresults = mb.findSeries(name, limit=100)
|
||||||
return serve_template(templatename="searchresults.html", title='Search Results for: "' + name + '"', searchresults=searchresults, name=name, type=type)
|
return serve_template(templatename="searchresults.html", title='Search Results for: "' + cgi.escape(name) + '"', searchresults=searchresults, name=cgi.escape(name), type=type)
|
||||||
|
|
||||||
@cherrypy.expose
|
@cherrypy.expose
|
||||||
def addArtist(self, artistid):
|
def addArtist(self, artistid):
|
||||||
@@ -230,11 +232,11 @@ class WebInterface(object):
|
|||||||
raise cherrypy.HTTPRedirect("artistPage?ArtistID=%s" % ArtistID)
|
raise cherrypy.HTTPRedirect("artistPage?ArtistID=%s" % ArtistID)
|
||||||
|
|
||||||
def removeArtist(self, ArtistID):
|
def removeArtist(self, ArtistID):
|
||||||
logger.info(u"Deleting all traces of artist: " + ArtistID)
|
|
||||||
myDB = db.DBConnection()
|
myDB = db.DBConnection()
|
||||||
namecheck = myDB.select('SELECT ArtistName from artists where ArtistID=?', [ArtistID])
|
namecheck = myDB.select('SELECT ArtistName from artists where ArtistID=?', [ArtistID])
|
||||||
for name in namecheck:
|
for name in namecheck:
|
||||||
artistname = name['ArtistName']
|
artistname = name['ArtistName']
|
||||||
|
logger.info(u"Deleting all traces of artist: " + artistname)
|
||||||
myDB.action('DELETE from artists WHERE ArtistID=?', [ArtistID])
|
myDB.action('DELETE from artists WHERE ArtistID=?', [ArtistID])
|
||||||
|
|
||||||
from headphones import cache
|
from headphones import cache
|
||||||
@@ -265,14 +267,72 @@ class WebInterface(object):
|
|||||||
|
|
||||||
@cherrypy.expose
|
@cherrypy.expose
|
||||||
def scanArtist(self, ArtistID):
|
def scanArtist(self, ArtistID):
|
||||||
logger.info(u"Scanning artist: %s", ArtistID)
|
|
||||||
myDB = db.DBConnection()
|
myDB = db.DBConnection()
|
||||||
artistname = myDB.select('SELECT DISTINCT ArtistName FROM artists WHERE ArtistID=?', [ArtistID])
|
artist_name = myDB.select('SELECT DISTINCT ArtistName FROM artists WHERE ArtistID=?', [ArtistID])[0][0]
|
||||||
artistfolder = os.path.join(headphones.CONFIG.DESTINATION_DIR, artistname[0][0])
|
|
||||||
try:
|
logger.info(u"Scanning artist: %s", artist_name)
|
||||||
threading.Thread(target=librarysync.libraryScan(dir=artistfolder)).start()
|
|
||||||
except Exception as e:
|
full_folder_format = headphones.CONFIG.FOLDER_FORMAT
|
||||||
logger.error('Unable to complete the scan: %s', e)
|
folder_format = re.findall(r'(.*?[Aa]rtist?)\.*', full_folder_format)[0]
|
||||||
|
|
||||||
|
acceptable_formats = ["$artist","$sortartist","$first/$artist","$first/$sortartist"]
|
||||||
|
|
||||||
|
if not folder_format.lower() in acceptable_formats:
|
||||||
|
logger.info("Can't determine the artist folder from the configured folder_format. Not scanning")
|
||||||
|
return
|
||||||
|
|
||||||
|
# Format the folder to match the settings
|
||||||
|
artist = artist_name.replace('/', '_')
|
||||||
|
|
||||||
|
if headphones.CONFIG.FILE_UNDERSCORES:
|
||||||
|
artist = artist.replace(' ', '_')
|
||||||
|
|
||||||
|
if artist.startswith('The '):
|
||||||
|
sortname = artist[4:] + ", The"
|
||||||
|
else:
|
||||||
|
sortname = artist
|
||||||
|
|
||||||
|
if sortname[0].isdigit():
|
||||||
|
firstchar = u'0-9'
|
||||||
|
else:
|
||||||
|
firstchar = sortname[0]
|
||||||
|
|
||||||
|
values = {'$Artist': artist,
|
||||||
|
'$SortArtist': sortname,
|
||||||
|
'$First': firstchar.upper(),
|
||||||
|
'$artist': artist.lower(),
|
||||||
|
'$sortartist': sortname.lower(),
|
||||||
|
'$first': firstchar.lower(),
|
||||||
|
}
|
||||||
|
|
||||||
|
folder = helpers.replace_all(folder_format.strip(), values, normalize=True)
|
||||||
|
|
||||||
|
folder = helpers.replace_illegal_chars(folder, type="folder")
|
||||||
|
folder = folder.replace('./', '_/').replace('/.', '/_')
|
||||||
|
|
||||||
|
if folder.endswith('.'):
|
||||||
|
folder = folder[:-1] + '_'
|
||||||
|
|
||||||
|
if folder.startswith('.'):
|
||||||
|
folder = '_' + folder[1:]
|
||||||
|
|
||||||
|
dirs = []
|
||||||
|
if headphones.CONFIG.MUSIC_DIR:
|
||||||
|
dirs.append(headphones.CONFIG.MUSIC_DIR)
|
||||||
|
if headphones.CONFIG.DESTINATION_DIR:
|
||||||
|
dirs.append(headphones.CONFIG.DESTINATION_DIR)
|
||||||
|
if headphones.CONFIG.LOSSLESS_DESTINATION_DIR:
|
||||||
|
dirs.append(headphones.CONFIG.LOSSLESS_DESTINATION_DIR)
|
||||||
|
|
||||||
|
dirs = set(dirs)
|
||||||
|
|
||||||
|
for dir in dirs:
|
||||||
|
artistfolder = os.path.join(dir, folder)
|
||||||
|
if not os.path.isdir(artistfolder):
|
||||||
|
logger.debug("Cannot find directory: " + artistfolder)
|
||||||
|
continue
|
||||||
|
threading.Thread(target=librarysync.libraryScan, kwargs={"dir":artistfolder, "artistScan":True, "ArtistID":ArtistID, "ArtistName":artist_name}).start()
|
||||||
raise cherrypy.HTTPRedirect("artistPage?ArtistID=%s" % ArtistID)
|
raise cherrypy.HTTPRedirect("artistPage?ArtistID=%s" % ArtistID)
|
||||||
|
|
||||||
@cherrypy.expose
|
@cherrypy.expose
|
||||||
@@ -740,9 +800,9 @@ class WebInterface(object):
|
|||||||
raise cherrypy.HTTPRedirect("home")
|
raise cherrypy.HTTPRedirect("home")
|
||||||
|
|
||||||
@cherrypy.expose
|
@cherrypy.expose
|
||||||
def forcePostProcess(self, dir=None, album_dir=None):
|
def forcePostProcess(self, dir=None, album_dir=None, keep_original_folder=False):
|
||||||
from headphones import postprocessor
|
from headphones import postprocessor
|
||||||
threading.Thread(target=postprocessor.forcePostProcess, kwargs={'dir': dir, 'album_dir': album_dir}).start()
|
threading.Thread(target=postprocessor.forcePostProcess, kwargs={'dir': dir, 'album_dir': album_dir, 'keep_original_folder':keep_original_folder == 'True'}).start()
|
||||||
raise cherrypy.HTTPRedirect("home")
|
raise cherrypy.HTTPRedirect("home")
|
||||||
|
|
||||||
@cherrypy.expose
|
@cherrypy.expose
|
||||||
@@ -1005,6 +1065,11 @@ class WebInterface(object):
|
|||||||
"newznab_apikey": headphones.CONFIG.NEWZNAB_APIKEY,
|
"newznab_apikey": headphones.CONFIG.NEWZNAB_APIKEY,
|
||||||
"newznab_enabled": checked(headphones.CONFIG.NEWZNAB_ENABLED),
|
"newznab_enabled": checked(headphones.CONFIG.NEWZNAB_ENABLED),
|
||||||
"extra_newznabs": headphones.CONFIG.get_extra_newznabs(),
|
"extra_newznabs": headphones.CONFIG.get_extra_newznabs(),
|
||||||
|
"use_torznab": checked(headphones.CONFIG.TORZNAB),
|
||||||
|
"torznab_host": headphones.CONFIG.TORZNAB_HOST,
|
||||||
|
"torznab_apikey": headphones.CONFIG.TORZNAB_APIKEY,
|
||||||
|
"torznab_enabled": checked(headphones.CONFIG.TORZNAB_ENABLED),
|
||||||
|
"extra_torznabs": headphones.CONFIG.get_extra_torznabs(),
|
||||||
"use_nzbsorg": checked(headphones.CONFIG.NZBSORG),
|
"use_nzbsorg": checked(headphones.CONFIG.NZBSORG),
|
||||||
"nzbsorg_uid": headphones.CONFIG.NZBSORG_UID,
|
"nzbsorg_uid": headphones.CONFIG.NZBSORG_UID,
|
||||||
"nzbsorg_hash": headphones.CONFIG.NZBSORG_HASH,
|
"nzbsorg_hash": headphones.CONFIG.NZBSORG_HASH,
|
||||||
@@ -1041,6 +1106,8 @@ class WebInterface(object):
|
|||||||
"whatcd_username": headphones.CONFIG.WHATCD_USERNAME,
|
"whatcd_username": headphones.CONFIG.WHATCD_USERNAME,
|
||||||
"whatcd_password": headphones.CONFIG.WHATCD_PASSWORD,
|
"whatcd_password": headphones.CONFIG.WHATCD_PASSWORD,
|
||||||
"whatcd_ratio": headphones.CONFIG.WHATCD_RATIO,
|
"whatcd_ratio": headphones.CONFIG.WHATCD_RATIO,
|
||||||
|
"use_strike": checked(headphones.CONFIG.STRIKE),
|
||||||
|
"strike_ratio": headphones.CONFIG.STRIKE_RATIO,
|
||||||
"pref_qual_0": radio(headphones.CONFIG.PREFERRED_QUALITY, 0),
|
"pref_qual_0": radio(headphones.CONFIG.PREFERRED_QUALITY, 0),
|
||||||
"pref_qual_1": radio(headphones.CONFIG.PREFERRED_QUALITY, 1),
|
"pref_qual_1": radio(headphones.CONFIG.PREFERRED_QUALITY, 1),
|
||||||
"pref_qual_2": radio(headphones.CONFIG.PREFERRED_QUALITY, 2),
|
"pref_qual_2": radio(headphones.CONFIG.PREFERRED_QUALITY, 2),
|
||||||
@@ -1066,15 +1133,19 @@ class WebInterface(object):
|
|||||||
"embed_album_art": checked(headphones.CONFIG.EMBED_ALBUM_ART),
|
"embed_album_art": checked(headphones.CONFIG.EMBED_ALBUM_ART),
|
||||||
"embed_lyrics": checked(headphones.CONFIG.EMBED_LYRICS),
|
"embed_lyrics": checked(headphones.CONFIG.EMBED_LYRICS),
|
||||||
"replace_existing_folders": checked(headphones.CONFIG.REPLACE_EXISTING_FOLDERS),
|
"replace_existing_folders": checked(headphones.CONFIG.REPLACE_EXISTING_FOLDERS),
|
||||||
|
"keep_original_folder" : checked(headphones.CONFIG.KEEP_ORIGINAL_FOLDER),
|
||||||
"destination_dir": headphones.CONFIG.DESTINATION_DIR,
|
"destination_dir": headphones.CONFIG.DESTINATION_DIR,
|
||||||
"lossless_destination_dir": headphones.CONFIG.LOSSLESS_DESTINATION_DIR,
|
"lossless_destination_dir": headphones.CONFIG.LOSSLESS_DESTINATION_DIR,
|
||||||
"folder_format": headphones.CONFIG.FOLDER_FORMAT,
|
"folder_format": headphones.CONFIG.FOLDER_FORMAT,
|
||||||
"file_format": headphones.CONFIG.FILE_FORMAT,
|
"file_format": headphones.CONFIG.FILE_FORMAT,
|
||||||
"file_underscores": checked(headphones.CONFIG.FILE_UNDERSCORES),
|
"file_underscores": checked(headphones.CONFIG.FILE_UNDERSCORES),
|
||||||
"include_extras": checked(headphones.CONFIG.INCLUDE_EXTRAS),
|
"include_extras": checked(headphones.CONFIG.INCLUDE_EXTRAS),
|
||||||
|
"official_releases_only": checked(headphones.CONFIG.OFFICIAL_RELEASES_ONLY),
|
||||||
|
"wait_until_release_date": checked(headphones.CONFIG.WAIT_UNTIL_RELEASE_DATE),
|
||||||
"autowant_upcoming": checked(headphones.CONFIG.AUTOWANT_UPCOMING),
|
"autowant_upcoming": checked(headphones.CONFIG.AUTOWANT_UPCOMING),
|
||||||
"autowant_all": checked(headphones.CONFIG.AUTOWANT_ALL),
|
"autowant_all": checked(headphones.CONFIG.AUTOWANT_ALL),
|
||||||
"autowant_manually_added": checked(headphones.CONFIG.AUTOWANT_MANUALLY_ADDED),
|
"autowant_manually_added": checked(headphones.CONFIG.AUTOWANT_MANUALLY_ADDED),
|
||||||
|
"do_not_process_unmatched": checked(headphones.CONFIG.DO_NOT_PROCESS_UNMATCHED),
|
||||||
"keep_torrent_files": checked(headphones.CONFIG.KEEP_TORRENT_FILES),
|
"keep_torrent_files": checked(headphones.CONFIG.KEEP_TORRENT_FILES),
|
||||||
"prefer_torrents_0": radio(headphones.CONFIG.PREFER_TORRENTS, 0),
|
"prefer_torrents_0": radio(headphones.CONFIG.PREFER_TORRENTS, 0),
|
||||||
"prefer_torrents_1": radio(headphones.CONFIG.PREFER_TORRENTS, 1),
|
"prefer_torrents_1": radio(headphones.CONFIG.PREFER_TORRENTS, 1),
|
||||||
@@ -1120,6 +1191,7 @@ class WebInterface(object):
|
|||||||
"plex_client_host": headphones.CONFIG.PLEX_CLIENT_HOST,
|
"plex_client_host": headphones.CONFIG.PLEX_CLIENT_HOST,
|
||||||
"plex_username": headphones.CONFIG.PLEX_USERNAME,
|
"plex_username": headphones.CONFIG.PLEX_USERNAME,
|
||||||
"plex_password": headphones.CONFIG.PLEX_PASSWORD,
|
"plex_password": headphones.CONFIG.PLEX_PASSWORD,
|
||||||
|
"plex_token": headphones.CONFIG.PLEX_TOKEN,
|
||||||
"plex_update": checked(headphones.CONFIG.PLEX_UPDATE),
|
"plex_update": checked(headphones.CONFIG.PLEX_UPDATE),
|
||||||
"plex_notify": checked(headphones.CONFIG.PLEX_NOTIFY),
|
"plex_notify": checked(headphones.CONFIG.PLEX_NOTIFY),
|
||||||
"nma_enabled": checked(headphones.CONFIG.NMA_ENABLED),
|
"nma_enabled": checked(headphones.CONFIG.NMA_ENABLED),
|
||||||
@@ -1214,11 +1286,12 @@ class WebInterface(object):
|
|||||||
# Handle the variable config options. Note - keys with False values aren't getting passed
|
# Handle the variable config options. Note - keys with False values aren't getting passed
|
||||||
|
|
||||||
checked_configs = [
|
checked_configs = [
|
||||||
"launch_browser", "enable_https", "api_enabled", "use_blackhole", "headphones_indexer", "use_newznab", "newznab_enabled",
|
"launch_browser", "enable_https", "api_enabled", "use_blackhole", "headphones_indexer", "use_newznab", "newznab_enabled", "use_torznab", "torznab_enabled",
|
||||||
"use_nzbsorg", "use_omgwtfnzbs", "use_kat", "use_piratebay", "use_oldpiratebay", "use_mininova", "use_waffles", "use_rutracker",
|
"use_nzbsorg", "use_omgwtfnzbs", "use_kat", "use_piratebay", "use_oldpiratebay", "use_mininova", "use_waffles", "use_rutracker",
|
||||||
"use_whatcd", "preferred_bitrate_allow_lossless", "detect_bitrate", "ignore_clean_releases", "freeze_db", "cue_split", "move_files", "rename_files",
|
"use_whatcd", "use_strike", "preferred_bitrate_allow_lossless", "detect_bitrate", "ignore_clean_releases", "freeze_db", "cue_split", "move_files",
|
||||||
"correct_metadata", "cleanup_files", "keep_nfo", "add_album_art", "embed_album_art", "embed_lyrics", "replace_existing_folders",
|
"rename_files", "correct_metadata", "cleanup_files", "keep_nfo", "add_album_art", "embed_album_art", "embed_lyrics",
|
||||||
"file_underscores", "include_extras", "autowant_upcoming", "autowant_all", "autowant_manually_added", "keep_torrent_files", "music_encoder",
|
"replace_existing_folders", "keep_original_folder", "file_underscores", "include_extras", "official_releases_only",
|
||||||
|
"wait_until_release_date", "autowant_upcoming", "autowant_all", "autowant_manually_added", "do_not_process_unmatched", "keep_torrent_files", "music_encoder",
|
||||||
"encoderlossless", "encoder_multicore", "delete_lossless_files", "growl_enabled", "growl_onsnatch", "prowl_enabled",
|
"encoderlossless", "encoder_multicore", "delete_lossless_files", "growl_enabled", "growl_onsnatch", "prowl_enabled",
|
||||||
"prowl_onsnatch", "xbmc_enabled", "xbmc_update", "xbmc_notify", "lms_enabled", "plex_enabled", "plex_update", "plex_notify",
|
"prowl_onsnatch", "xbmc_enabled", "xbmc_update", "xbmc_notify", "lms_enabled", "plex_enabled", "plex_update", "plex_notify",
|
||||||
"nma_enabled", "nma_onsnatch", "pushalot_enabled", "pushalot_onsnatch", "synoindex_enabled", "pushover_enabled",
|
"nma_enabled", "nma_onsnatch", "pushalot_enabled", "pushalot_onsnatch", "synoindex_enabled", "pushover_enabled",
|
||||||
@@ -1251,6 +1324,21 @@ class WebInterface(object):
|
|||||||
del kwargs[key]
|
del kwargs[key]
|
||||||
extra_newznabs.append((newznab_host, newznab_api, newznab_enabled))
|
extra_newznabs.append((newznab_host, newznab_api, newznab_enabled))
|
||||||
|
|
||||||
|
extra_torznabs = []
|
||||||
|
for kwarg in [x for x in kwargs if x.startswith('torznab_host')]:
|
||||||
|
torznab_host_key = kwarg
|
||||||
|
torznab_number = kwarg[12:]
|
||||||
|
if len(torznab_number):
|
||||||
|
torznab_api_key = 'torznab_api' + torznab_number
|
||||||
|
torznab_enabled_key = 'torznab_enabled' + torznab_number
|
||||||
|
torznab_host = kwargs.get(torznab_host_key, '')
|
||||||
|
torznab_api = kwargs.get(torznab_api_key, '')
|
||||||
|
torznab_enabled = int(kwargs.get(torznab_enabled_key, 0))
|
||||||
|
for key in [torznab_host_key, torznab_api_key, torznab_enabled_key]:
|
||||||
|
if key in kwargs:
|
||||||
|
del kwargs[key]
|
||||||
|
extra_torznabs.append((torznab_host, torznab_api, torznab_enabled))
|
||||||
|
|
||||||
# Convert the extras to list then string. Coming in as 0 or 1 (append new extras to the end)
|
# Convert the extras to list then string. Coming in as 0 or 1 (append new extras to the end)
|
||||||
temp_extras_list = []
|
temp_extras_list = []
|
||||||
|
|
||||||
@@ -1276,11 +1364,18 @@ class WebInterface(object):
|
|||||||
del kwargs[extra]
|
del kwargs[extra]
|
||||||
|
|
||||||
headphones.CONFIG.EXTRAS = ','.join(str(n) for n in temp_extras_list)
|
headphones.CONFIG.EXTRAS = ','.join(str(n) for n in temp_extras_list)
|
||||||
|
|
||||||
headphones.CONFIG.clear_extra_newznabs()
|
headphones.CONFIG.clear_extra_newznabs()
|
||||||
|
headphones.CONFIG.clear_extra_torznabs()
|
||||||
|
|
||||||
headphones.CONFIG.process_kwargs(kwargs)
|
headphones.CONFIG.process_kwargs(kwargs)
|
||||||
|
|
||||||
for extra_newznab in extra_newznabs:
|
for extra_newznab in extra_newznabs:
|
||||||
headphones.CONFIG.add_extra_newznab(extra_newznab)
|
headphones.CONFIG.add_extra_newznab(extra_newznab)
|
||||||
|
|
||||||
|
for extra_torznab in extra_torznabs:
|
||||||
|
headphones.CONFIG.add_extra_torznab(extra_torznab)
|
||||||
|
|
||||||
# Sanity checking
|
# Sanity checking
|
||||||
if headphones.CONFIG.SEARCH_INTERVAL and headphones.CONFIG.SEARCH_INTERVAL < 360:
|
if headphones.CONFIG.SEARCH_INTERVAL and headphones.CONFIG.SEARCH_INTERVAL < 360:
|
||||||
logger.info("Search interval too low. Resetting to 6 hour minimum")
|
logger.info("Search interval too low. Resetting to 6 hour minimum")
|
||||||
@@ -1424,6 +1519,25 @@ class WebInterface(object):
|
|||||||
logger.warn(msg)
|
logger.warn(msg)
|
||||||
return msg
|
return msg
|
||||||
|
|
||||||
|
@cherrypy.expose
|
||||||
|
def testPushover(self):
|
||||||
|
logger.info(u"Sending Pushover notification")
|
||||||
|
pushover = notifiers.PUSHOVER()
|
||||||
|
result = pushover.notify("hooray!", "This is a test")
|
||||||
|
return str(result)
|
||||||
|
|
||||||
|
@cherrypy.expose
|
||||||
|
def testPlex(self):
|
||||||
|
logger.info(u"Testing plex notifications")
|
||||||
|
plex = notifiers.Plex()
|
||||||
|
plex.notify("hellooooo", "test album!", "")
|
||||||
|
|
||||||
|
@cherrypy.expose
|
||||||
|
def testPushbullet(self):
|
||||||
|
logger.info("Testing Pushbullet notifications")
|
||||||
|
pushbullet = notifiers.PUSHBULLET()
|
||||||
|
pushbullet.notify("it works!")
|
||||||
|
|
||||||
class Artwork(object):
|
class Artwork(object):
|
||||||
@cherrypy.expose
|
@cherrypy.expose
|
||||||
def index(self):
|
def index(self):
|
||||||
|
|||||||
+77
-15
@@ -17,8 +17,8 @@ http://www.crummy.com/software/BeautifulSoup/bs4/doc/
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
__author__ = "Leonard Richardson (leonardr@segfault.org)"
|
__author__ = "Leonard Richardson (leonardr@segfault.org)"
|
||||||
__version__ = "4.3.2"
|
__version__ = "4.4.0"
|
||||||
__copyright__ = "Copyright (c) 2004-2013 Leonard Richardson"
|
__copyright__ = "Copyright (c) 2004-2015 Leonard Richardson"
|
||||||
__license__ = "MIT"
|
__license__ = "MIT"
|
||||||
|
|
||||||
__all__ = ['BeautifulSoup']
|
__all__ = ['BeautifulSoup']
|
||||||
@@ -45,7 +45,7 @@ from .element import (
|
|||||||
|
|
||||||
# The very first thing we do is give a useful error if someone is
|
# The very first thing we do is give a useful error if someone is
|
||||||
# running this code under Python 3 without converting it.
|
# running this code under Python 3 without converting it.
|
||||||
syntax_error = u'You are trying to run the Python 2 version of Beautiful Soup under Python 3. This will not work. You need to convert the code, either by installing it (`python setup.py install`) or by running 2to3 (`2to3 -w bs4`).'
|
'You are trying to run the Python 2 version of Beautiful Soup under Python 3. This will not work.'<>'You need to convert the code, either by installing it (`python setup.py install`) or by running 2to3 (`2to3 -w bs4`).'
|
||||||
|
|
||||||
class BeautifulSoup(Tag):
|
class BeautifulSoup(Tag):
|
||||||
"""
|
"""
|
||||||
@@ -77,8 +77,11 @@ class BeautifulSoup(Tag):
|
|||||||
|
|
||||||
ASCII_SPACES = '\x20\x0a\x09\x0c\x0d'
|
ASCII_SPACES = '\x20\x0a\x09\x0c\x0d'
|
||||||
|
|
||||||
|
NO_PARSER_SPECIFIED_WARNING = "No parser was explicitly specified, so I'm using the best available %(markup_type)s parser for this system (\"%(parser)s\"). This usually isn't a problem, but if you run this code on another system, or in a different virtual environment, it may use a different parser and behave differently.\n\nTo get rid of this warning, change this:\n\n BeautifulSoup([your markup])\n\nto this:\n\n BeautifulSoup([your markup], \"%(parser)s\")\n"
|
||||||
|
|
||||||
def __init__(self, markup="", features=None, builder=None,
|
def __init__(self, markup="", features=None, builder=None,
|
||||||
parse_only=None, from_encoding=None, **kwargs):
|
parse_only=None, from_encoding=None, exclude_encodings=None,
|
||||||
|
**kwargs):
|
||||||
"""The Soup object is initialized as the 'root tag', and the
|
"""The Soup object is initialized as the 'root tag', and the
|
||||||
provided markup (which can be a string or a file-like object)
|
provided markup (which can be a string or a file-like object)
|
||||||
is fed into the underlying parser."""
|
is fed into the underlying parser."""
|
||||||
@@ -114,9 +117,9 @@ class BeautifulSoup(Tag):
|
|||||||
del kwargs['isHTML']
|
del kwargs['isHTML']
|
||||||
warnings.warn(
|
warnings.warn(
|
||||||
"BS4 does not respect the isHTML argument to the "
|
"BS4 does not respect the isHTML argument to the "
|
||||||
"BeautifulSoup constructor. You can pass in features='html' "
|
"BeautifulSoup constructor. Suggest you use "
|
||||||
"or features='xml' to get a builder capable of handling "
|
"features='lxml' for HTML and features='lxml-xml' for "
|
||||||
"one or the other.")
|
"XML.")
|
||||||
|
|
||||||
def deprecated_argument(old_name, new_name):
|
def deprecated_argument(old_name, new_name):
|
||||||
if old_name in kwargs:
|
if old_name in kwargs:
|
||||||
@@ -140,6 +143,7 @@ class BeautifulSoup(Tag):
|
|||||||
"__init__() got an unexpected keyword argument '%s'" % arg)
|
"__init__() got an unexpected keyword argument '%s'" % arg)
|
||||||
|
|
||||||
if builder is None:
|
if builder is None:
|
||||||
|
original_features = features
|
||||||
if isinstance(features, basestring):
|
if isinstance(features, basestring):
|
||||||
features = [features]
|
features = [features]
|
||||||
if features is None or len(features) == 0:
|
if features is None or len(features) == 0:
|
||||||
@@ -151,6 +155,16 @@ class BeautifulSoup(Tag):
|
|||||||
"requested: %s. Do you need to install a parser library?"
|
"requested: %s. Do you need to install a parser library?"
|
||||||
% ",".join(features))
|
% ",".join(features))
|
||||||
builder = builder_class()
|
builder = builder_class()
|
||||||
|
if not (original_features == builder.NAME or
|
||||||
|
original_features in builder.ALTERNATE_NAMES):
|
||||||
|
if builder.is_xml:
|
||||||
|
markup_type = "XML"
|
||||||
|
else:
|
||||||
|
markup_type = "HTML"
|
||||||
|
warnings.warn(self.NO_PARSER_SPECIFIED_WARNING % dict(
|
||||||
|
parser=builder.NAME,
|
||||||
|
markup_type=markup_type))
|
||||||
|
|
||||||
self.builder = builder
|
self.builder = builder
|
||||||
self.is_xml = builder.is_xml
|
self.is_xml = builder.is_xml
|
||||||
self.builder.soup = self
|
self.builder.soup = self
|
||||||
@@ -178,6 +192,8 @@ class BeautifulSoup(Tag):
|
|||||||
# system. Just let it go.
|
# system. Just let it go.
|
||||||
pass
|
pass
|
||||||
if is_file:
|
if is_file:
|
||||||
|
if isinstance(markup, unicode):
|
||||||
|
markup = markup.encode("utf8")
|
||||||
warnings.warn(
|
warnings.warn(
|
||||||
'"%s" looks like a filename, not markup. You should probably open this file and pass the filehandle into Beautiful Soup.' % markup)
|
'"%s" looks like a filename, not markup. You should probably open this file and pass the filehandle into Beautiful Soup.' % markup)
|
||||||
if markup[:5] == "http:" or markup[:6] == "https:":
|
if markup[:5] == "http:" or markup[:6] == "https:":
|
||||||
@@ -185,12 +201,15 @@ class BeautifulSoup(Tag):
|
|||||||
# Python 3 otherwise.
|
# Python 3 otherwise.
|
||||||
if ((isinstance(markup, bytes) and not b' ' in markup)
|
if ((isinstance(markup, bytes) and not b' ' in markup)
|
||||||
or (isinstance(markup, unicode) and not u' ' in markup)):
|
or (isinstance(markup, unicode) and not u' ' in markup)):
|
||||||
|
if isinstance(markup, unicode):
|
||||||
|
markup = markup.encode("utf8")
|
||||||
warnings.warn(
|
warnings.warn(
|
||||||
'"%s" looks like a URL. Beautiful Soup is not an HTTP client. You should probably use an HTTP client to get the document behind the URL, and feed that document to Beautiful Soup.' % markup)
|
'"%s" looks like a URL. Beautiful Soup is not an HTTP client. You should probably use an HTTP client to get the document behind the URL, and feed that document to Beautiful Soup.' % markup)
|
||||||
|
|
||||||
for (self.markup, self.original_encoding, self.declared_html_encoding,
|
for (self.markup, self.original_encoding, self.declared_html_encoding,
|
||||||
self.contains_replacement_characters) in (
|
self.contains_replacement_characters) in (
|
||||||
self.builder.prepare_markup(markup, from_encoding)):
|
self.builder.prepare_markup(
|
||||||
|
markup, from_encoding, exclude_encodings=exclude_encodings)):
|
||||||
self.reset()
|
self.reset()
|
||||||
try:
|
try:
|
||||||
self._feed()
|
self._feed()
|
||||||
@@ -203,6 +222,16 @@ class BeautifulSoup(Tag):
|
|||||||
self.markup = None
|
self.markup = None
|
||||||
self.builder.soup = None
|
self.builder.soup = None
|
||||||
|
|
||||||
|
def __copy__(self):
|
||||||
|
return type(self)(self.encode(), builder=self.builder)
|
||||||
|
|
||||||
|
def __getstate__(self):
|
||||||
|
# Frequently a tree builder can't be pickled.
|
||||||
|
d = dict(self.__dict__)
|
||||||
|
if 'builder' in d and not self.builder.picklable:
|
||||||
|
del d['builder']
|
||||||
|
return d
|
||||||
|
|
||||||
def _feed(self):
|
def _feed(self):
|
||||||
# Convert the document to Unicode.
|
# Convert the document to Unicode.
|
||||||
self.builder.reset()
|
self.builder.reset()
|
||||||
@@ -229,9 +258,7 @@ class BeautifulSoup(Tag):
|
|||||||
|
|
||||||
def new_string(self, s, subclass=NavigableString):
|
def new_string(self, s, subclass=NavigableString):
|
||||||
"""Create a new NavigableString associated with this soup."""
|
"""Create a new NavigableString associated with this soup."""
|
||||||
navigable = subclass(s)
|
return subclass(s)
|
||||||
navigable.setup()
|
|
||||||
return navigable
|
|
||||||
|
|
||||||
def insert_before(self, successor):
|
def insert_before(self, successor):
|
||||||
raise NotImplementedError("BeautifulSoup objects don't support insert_before().")
|
raise NotImplementedError("BeautifulSoup objects don't support insert_before().")
|
||||||
@@ -290,14 +317,49 @@ class BeautifulSoup(Tag):
|
|||||||
def object_was_parsed(self, o, parent=None, most_recent_element=None):
|
def object_was_parsed(self, o, parent=None, most_recent_element=None):
|
||||||
"""Add an object to the parse tree."""
|
"""Add an object to the parse tree."""
|
||||||
parent = parent or self.currentTag
|
parent = parent or self.currentTag
|
||||||
most_recent_element = most_recent_element or self._most_recent_element
|
previous_element = most_recent_element or self._most_recent_element
|
||||||
o.setup(parent, most_recent_element)
|
|
||||||
|
next_element = previous_sibling = next_sibling = None
|
||||||
|
if isinstance(o, Tag):
|
||||||
|
next_element = o.next_element
|
||||||
|
next_sibling = o.next_sibling
|
||||||
|
previous_sibling = o.previous_sibling
|
||||||
|
if not previous_element:
|
||||||
|
previous_element = o.previous_element
|
||||||
|
|
||||||
|
o.setup(parent, previous_element, next_element, previous_sibling, next_sibling)
|
||||||
|
|
||||||
if most_recent_element is not None:
|
|
||||||
most_recent_element.next_element = o
|
|
||||||
self._most_recent_element = o
|
self._most_recent_element = o
|
||||||
parent.contents.append(o)
|
parent.contents.append(o)
|
||||||
|
|
||||||
|
if parent.next_sibling:
|
||||||
|
# This node is being inserted into an element that has
|
||||||
|
# already been parsed. Deal with any dangling references.
|
||||||
|
index = parent.contents.index(o)
|
||||||
|
if index == 0:
|
||||||
|
previous_element = parent
|
||||||
|
previous_sibling = None
|
||||||
|
else:
|
||||||
|
previous_element = previous_sibling = parent.contents[index-1]
|
||||||
|
if index == len(parent.contents)-1:
|
||||||
|
next_element = parent.next_sibling
|
||||||
|
next_sibling = None
|
||||||
|
else:
|
||||||
|
next_element = next_sibling = parent.contents[index+1]
|
||||||
|
|
||||||
|
o.previous_element = previous_element
|
||||||
|
if previous_element:
|
||||||
|
previous_element.next_element = o
|
||||||
|
o.next_element = next_element
|
||||||
|
if next_element:
|
||||||
|
next_element.previous_element = o
|
||||||
|
o.next_sibling = next_sibling
|
||||||
|
if next_sibling:
|
||||||
|
next_sibling.previous_sibling = o
|
||||||
|
o.previous_sibling = previous_sibling
|
||||||
|
if previous_sibling:
|
||||||
|
previous_sibling.next_sibling = o
|
||||||
|
|
||||||
def _popToTag(self, name, nsprefix=None, inclusivePop=True):
|
def _popToTag(self, name, nsprefix=None, inclusivePop=True):
|
||||||
"""Pops the tag stack up to and including the most recent
|
"""Pops the tag stack up to and including the most recent
|
||||||
instance of the given tag. If inclusivePop is false, pops the tag
|
instance of the given tag. If inclusivePop is false, pops the tag
|
||||||
|
|||||||
@@ -80,9 +80,12 @@ builder_registry = TreeBuilderRegistry()
|
|||||||
class TreeBuilder(object):
|
class TreeBuilder(object):
|
||||||
"""Turn a document into a Beautiful Soup object tree."""
|
"""Turn a document into a Beautiful Soup object tree."""
|
||||||
|
|
||||||
|
NAME = "[Unknown tree builder]"
|
||||||
|
ALTERNATE_NAMES = []
|
||||||
features = []
|
features = []
|
||||||
|
|
||||||
is_xml = False
|
is_xml = False
|
||||||
|
picklable = False
|
||||||
preserve_whitespace_tags = set()
|
preserve_whitespace_tags = set()
|
||||||
empty_element_tags = None # A tag will be considered an empty-element
|
empty_element_tags = None # A tag will be considered an empty-element
|
||||||
# tag when and only when it has no contents.
|
# tag when and only when it has no contents.
|
||||||
|
|||||||
@@ -2,6 +2,7 @@ __all__ = [
|
|||||||
'HTML5TreeBuilder',
|
'HTML5TreeBuilder',
|
||||||
]
|
]
|
||||||
|
|
||||||
|
from pdb import set_trace
|
||||||
import warnings
|
import warnings
|
||||||
from bs4.builder import (
|
from bs4.builder import (
|
||||||
PERMISSIVE,
|
PERMISSIVE,
|
||||||
@@ -9,7 +10,10 @@ from bs4.builder import (
|
|||||||
HTML_5,
|
HTML_5,
|
||||||
HTMLTreeBuilder,
|
HTMLTreeBuilder,
|
||||||
)
|
)
|
||||||
from bs4.element import NamespacedAttribute
|
from bs4.element import (
|
||||||
|
NamespacedAttribute,
|
||||||
|
whitespace_re,
|
||||||
|
)
|
||||||
import html5lib
|
import html5lib
|
||||||
from html5lib.constants import namespaces
|
from html5lib.constants import namespaces
|
||||||
from bs4.element import (
|
from bs4.element import (
|
||||||
@@ -22,11 +26,20 @@ from bs4.element import (
|
|||||||
class HTML5TreeBuilder(HTMLTreeBuilder):
|
class HTML5TreeBuilder(HTMLTreeBuilder):
|
||||||
"""Use html5lib to build a tree."""
|
"""Use html5lib to build a tree."""
|
||||||
|
|
||||||
features = ['html5lib', PERMISSIVE, HTML_5, HTML]
|
NAME = "html5lib"
|
||||||
|
|
||||||
def prepare_markup(self, markup, user_specified_encoding):
|
features = [NAME, PERMISSIVE, HTML_5, HTML]
|
||||||
|
|
||||||
|
def prepare_markup(self, markup, user_specified_encoding,
|
||||||
|
document_declared_encoding=None, exclude_encodings=None):
|
||||||
# Store the user-specified encoding for use later on.
|
# Store the user-specified encoding for use later on.
|
||||||
self.user_specified_encoding = user_specified_encoding
|
self.user_specified_encoding = user_specified_encoding
|
||||||
|
|
||||||
|
# document_declared_encoding and exclude_encodings aren't used
|
||||||
|
# ATM because the html5lib TreeBuilder doesn't use
|
||||||
|
# UnicodeDammit.
|
||||||
|
if exclude_encodings:
|
||||||
|
warnings.warn("You provided a value for exclude_encoding, but the html5lib tree builder doesn't support exclude_encoding.")
|
||||||
yield (markup, None, None, False)
|
yield (markup, None, None, False)
|
||||||
|
|
||||||
# These methods are defined by Beautiful Soup.
|
# These methods are defined by Beautiful Soup.
|
||||||
@@ -101,7 +114,13 @@ class AttrList(object):
|
|||||||
def __iter__(self):
|
def __iter__(self):
|
||||||
return list(self.attrs.items()).__iter__()
|
return list(self.attrs.items()).__iter__()
|
||||||
def __setitem__(self, name, value):
|
def __setitem__(self, name, value):
|
||||||
"set attr", name, value
|
# If this attribute is a multi-valued attribute for this element,
|
||||||
|
# turn its value into a list.
|
||||||
|
list_attr = HTML5TreeBuilder.cdata_list_attributes
|
||||||
|
if (name in list_attr['*']
|
||||||
|
or (self.element.name in list_attr
|
||||||
|
and name in list_attr[self.element.name])):
|
||||||
|
value = whitespace_re.split(value)
|
||||||
self.element[name] = value
|
self.element[name] = value
|
||||||
def items(self):
|
def items(self):
|
||||||
return list(self.attrs.items())
|
return list(self.attrs.items())
|
||||||
@@ -161,6 +180,12 @@ class Element(html5lib.treebuilders._base.Node):
|
|||||||
# immediately after the parent, if it has no children.)
|
# immediately after the parent, if it has no children.)
|
||||||
if self.element.contents:
|
if self.element.contents:
|
||||||
most_recent_element = self.element._last_descendant(False)
|
most_recent_element = self.element._last_descendant(False)
|
||||||
|
elif self.element.next_element is not None:
|
||||||
|
# Something from further ahead in the parse tree is
|
||||||
|
# being inserted into this earlier element. This is
|
||||||
|
# very annoying because it means an expensive search
|
||||||
|
# for the last element in the tree.
|
||||||
|
most_recent_element = self.soup._last_descendant()
|
||||||
else:
|
else:
|
||||||
most_recent_element = self.element
|
most_recent_element = self.element
|
||||||
|
|
||||||
@@ -172,6 +197,7 @@ class Element(html5lib.treebuilders._base.Node):
|
|||||||
return AttrList(self.element)
|
return AttrList(self.element)
|
||||||
|
|
||||||
def setAttributes(self, attributes):
|
def setAttributes(self, attributes):
|
||||||
|
|
||||||
if attributes is not None and len(attributes) > 0:
|
if attributes is not None and len(attributes) > 0:
|
||||||
|
|
||||||
converted_attributes = []
|
converted_attributes = []
|
||||||
@@ -218,6 +244,9 @@ class Element(html5lib.treebuilders._base.Node):
|
|||||||
|
|
||||||
def reparentChildren(self, new_parent):
|
def reparentChildren(self, new_parent):
|
||||||
"""Move all of this tag's children into another tag."""
|
"""Move all of this tag's children into another tag."""
|
||||||
|
# print "MOVE", self.element.contents
|
||||||
|
# print "FROM", self.element
|
||||||
|
# print "TO", new_parent.element
|
||||||
element = self.element
|
element = self.element
|
||||||
new_parent_element = new_parent.element
|
new_parent_element = new_parent.element
|
||||||
# Determine what this tag's next_element will be once all the children
|
# Determine what this tag's next_element will be once all the children
|
||||||
@@ -236,17 +265,28 @@ class Element(html5lib.treebuilders._base.Node):
|
|||||||
new_parents_last_descendant_next_element = new_parent_element.next_element
|
new_parents_last_descendant_next_element = new_parent_element.next_element
|
||||||
|
|
||||||
to_append = element.contents
|
to_append = element.contents
|
||||||
append_after = new_parent.element.contents
|
append_after = new_parent_element.contents
|
||||||
if len(to_append) > 0:
|
if len(to_append) > 0:
|
||||||
# Set the first child's previous_element and previous_sibling
|
# Set the first child's previous_element and previous_sibling
|
||||||
# to elements within the new parent
|
# to elements within the new parent
|
||||||
first_child = to_append[0]
|
first_child = to_append[0]
|
||||||
first_child.previous_element = new_parents_last_descendant
|
if new_parents_last_descendant:
|
||||||
|
first_child.previous_element = new_parents_last_descendant
|
||||||
|
else:
|
||||||
|
first_child.previous_element = new_parent_element
|
||||||
first_child.previous_sibling = new_parents_last_child
|
first_child.previous_sibling = new_parents_last_child
|
||||||
|
if new_parents_last_descendant:
|
||||||
|
new_parents_last_descendant.next_element = first_child
|
||||||
|
else:
|
||||||
|
new_parent_element.next_element = first_child
|
||||||
|
if new_parents_last_child:
|
||||||
|
new_parents_last_child.next_sibling = first_child
|
||||||
|
|
||||||
# Fix the last child's next_element and next_sibling
|
# Fix the last child's next_element and next_sibling
|
||||||
last_child = to_append[-1]
|
last_child = to_append[-1]
|
||||||
last_child.next_element = new_parents_last_descendant_next_element
|
last_child.next_element = new_parents_last_descendant_next_element
|
||||||
|
if new_parents_last_descendant_next_element:
|
||||||
|
new_parents_last_descendant_next_element.previous_element = last_child
|
||||||
last_child.next_sibling = None
|
last_child.next_sibling = None
|
||||||
|
|
||||||
for child in to_append:
|
for child in to_append:
|
||||||
@@ -257,6 +297,10 @@ class Element(html5lib.treebuilders._base.Node):
|
|||||||
element.contents = []
|
element.contents = []
|
||||||
element.next_element = final_next_element
|
element.next_element = final_next_element
|
||||||
|
|
||||||
|
# print "DONE WITH MOVE"
|
||||||
|
# print "FROM", self.element
|
||||||
|
# print "TO", new_parent_element
|
||||||
|
|
||||||
def cloneNode(self):
|
def cloneNode(self):
|
||||||
tag = self.soup.new_tag(self.element.name, self.namespace)
|
tag = self.soup.new_tag(self.element.name, self.namespace)
|
||||||
node = Element(tag, self.soup, self.namespace)
|
node = Element(tag, self.soup, self.namespace)
|
||||||
@@ -268,7 +312,7 @@ class Element(html5lib.treebuilders._base.Node):
|
|||||||
return self.element.contents
|
return self.element.contents
|
||||||
|
|
||||||
def getNameTuple(self):
|
def getNameTuple(self):
|
||||||
if self.namespace is None:
|
if self.namespace == None:
|
||||||
return namespaces["html"], self.name
|
return namespaces["html"], self.name
|
||||||
else:
|
else:
|
||||||
return self.namespace, self.name
|
return self.namespace, self.name
|
||||||
|
|||||||
@@ -4,10 +4,16 @@ __all__ = [
|
|||||||
'HTMLParserTreeBuilder',
|
'HTMLParserTreeBuilder',
|
||||||
]
|
]
|
||||||
|
|
||||||
from HTMLParser import (
|
from HTMLParser import HTMLParser
|
||||||
HTMLParser,
|
|
||||||
HTMLParseError,
|
try:
|
||||||
)
|
from HTMLParser import HTMLParseError
|
||||||
|
except ImportError, e:
|
||||||
|
# HTMLParseError is removed in Python 3.5. Since it can never be
|
||||||
|
# thrown in 3.5, we can just define our own class as a placeholder.
|
||||||
|
class HTMLParseError(Exception):
|
||||||
|
pass
|
||||||
|
|
||||||
import sys
|
import sys
|
||||||
import warnings
|
import warnings
|
||||||
|
|
||||||
@@ -19,10 +25,10 @@ import warnings
|
|||||||
# At the end of this file, we monkeypatch HTMLParser so that
|
# At the end of this file, we monkeypatch HTMLParser so that
|
||||||
# strict=True works well on Python 3.2.2.
|
# strict=True works well on Python 3.2.2.
|
||||||
major, minor, release = sys.version_info[:3]
|
major, minor, release = sys.version_info[:3]
|
||||||
CONSTRUCTOR_TAKES_STRICT = (
|
CONSTRUCTOR_TAKES_STRICT = major == 3 and minor == 2 and release >= 3
|
||||||
major > 3
|
CONSTRUCTOR_STRICT_IS_DEPRECATED = major == 3 and minor == 3
|
||||||
or (major == 3 and minor > 2)
|
CONSTRUCTOR_TAKES_CONVERT_CHARREFS = major == 3 and minor >= 4
|
||||||
or (major == 3 and minor == 2 and release >= 3))
|
|
||||||
|
|
||||||
from bs4.element import (
|
from bs4.element import (
|
||||||
CData,
|
CData,
|
||||||
@@ -63,7 +69,8 @@ class BeautifulSoupHTMLParser(HTMLParser):
|
|||||||
|
|
||||||
def handle_charref(self, name):
|
def handle_charref(self, name):
|
||||||
# XXX workaround for a bug in HTMLParser. Remove this once
|
# XXX workaround for a bug in HTMLParser. Remove this once
|
||||||
# it's fixed.
|
# it's fixed in all supported versions.
|
||||||
|
# http://bugs.python.org/issue13633
|
||||||
if name.startswith('x'):
|
if name.startswith('x'):
|
||||||
real_name = int(name.lstrip('x'), 16)
|
real_name = int(name.lstrip('x'), 16)
|
||||||
elif name.startswith('X'):
|
elif name.startswith('X'):
|
||||||
@@ -113,14 +120,6 @@ class BeautifulSoupHTMLParser(HTMLParser):
|
|||||||
|
|
||||||
def handle_pi(self, data):
|
def handle_pi(self, data):
|
||||||
self.soup.endData()
|
self.soup.endData()
|
||||||
if data.endswith("?") and data.lower().startswith("xml"):
|
|
||||||
# "An XHTML processing instruction using the trailing '?'
|
|
||||||
# will cause the '?' to be included in data." - HTMLParser
|
|
||||||
# docs.
|
|
||||||
#
|
|
||||||
# Strip the question mark so we don't end up with two
|
|
||||||
# question marks.
|
|
||||||
data = data[:-1]
|
|
||||||
self.soup.handle_data(data)
|
self.soup.handle_data(data)
|
||||||
self.soup.endData(ProcessingInstruction)
|
self.soup.endData(ProcessingInstruction)
|
||||||
|
|
||||||
@@ -128,15 +127,19 @@ class BeautifulSoupHTMLParser(HTMLParser):
|
|||||||
class HTMLParserTreeBuilder(HTMLTreeBuilder):
|
class HTMLParserTreeBuilder(HTMLTreeBuilder):
|
||||||
|
|
||||||
is_xml = False
|
is_xml = False
|
||||||
features = [HTML, STRICT, HTMLPARSER]
|
picklable = True
|
||||||
|
NAME = HTMLPARSER
|
||||||
|
features = [NAME, HTML, STRICT]
|
||||||
|
|
||||||
def __init__(self, *args, **kwargs):
|
def __init__(self, *args, **kwargs):
|
||||||
if CONSTRUCTOR_TAKES_STRICT:
|
if CONSTRUCTOR_TAKES_STRICT and not CONSTRUCTOR_STRICT_IS_DEPRECATED:
|
||||||
kwargs['strict'] = False
|
kwargs['strict'] = False
|
||||||
|
if CONSTRUCTOR_TAKES_CONVERT_CHARREFS:
|
||||||
|
kwargs['convert_charrefs'] = False
|
||||||
self.parser_args = (args, kwargs)
|
self.parser_args = (args, kwargs)
|
||||||
|
|
||||||
def prepare_markup(self, markup, user_specified_encoding=None,
|
def prepare_markup(self, markup, user_specified_encoding=None,
|
||||||
document_declared_encoding=None):
|
document_declared_encoding=None, exclude_encodings=None):
|
||||||
"""
|
"""
|
||||||
:return: A 4-tuple (markup, original encoding, encoding
|
:return: A 4-tuple (markup, original encoding, encoding
|
||||||
declared within markup, whether any characters had to be
|
declared within markup, whether any characters had to be
|
||||||
@@ -147,7 +150,8 @@ class HTMLParserTreeBuilder(HTMLTreeBuilder):
|
|||||||
return
|
return
|
||||||
|
|
||||||
try_encodings = [user_specified_encoding, document_declared_encoding]
|
try_encodings = [user_specified_encoding, document_declared_encoding]
|
||||||
dammit = UnicodeDammit(markup, try_encodings, is_html=True)
|
dammit = UnicodeDammit(markup, try_encodings, is_html=True,
|
||||||
|
exclude_encodings=exclude_encodings)
|
||||||
yield (dammit.markup, dammit.original_encoding,
|
yield (dammit.markup, dammit.original_encoding,
|
||||||
dammit.declared_html_encoding,
|
dammit.declared_html_encoding,
|
||||||
dammit.contains_replacement_characters)
|
dammit.contains_replacement_characters)
|
||||||
|
|||||||
@@ -7,7 +7,12 @@ from io import BytesIO
|
|||||||
from StringIO import StringIO
|
from StringIO import StringIO
|
||||||
import collections
|
import collections
|
||||||
from lxml import etree
|
from lxml import etree
|
||||||
from bs4.element import Comment, Doctype, NamespacedAttribute
|
from bs4.element import (
|
||||||
|
Comment,
|
||||||
|
Doctype,
|
||||||
|
NamespacedAttribute,
|
||||||
|
ProcessingInstruction,
|
||||||
|
)
|
||||||
from bs4.builder import (
|
from bs4.builder import (
|
||||||
FAST,
|
FAST,
|
||||||
HTML,
|
HTML,
|
||||||
@@ -25,8 +30,11 @@ class LXMLTreeBuilderForXML(TreeBuilder):
|
|||||||
|
|
||||||
is_xml = True
|
is_xml = True
|
||||||
|
|
||||||
|
NAME = "lxml-xml"
|
||||||
|
ALTERNATE_NAMES = ["xml"]
|
||||||
|
|
||||||
# Well, it's permissive by XML parser standards.
|
# Well, it's permissive by XML parser standards.
|
||||||
features = [LXML, XML, FAST, PERMISSIVE]
|
features = [NAME, LXML, XML, FAST, PERMISSIVE]
|
||||||
|
|
||||||
CHUNK_SIZE = 512
|
CHUNK_SIZE = 512
|
||||||
|
|
||||||
@@ -70,6 +78,7 @@ class LXMLTreeBuilderForXML(TreeBuilder):
|
|||||||
return (None, tag)
|
return (None, tag)
|
||||||
|
|
||||||
def prepare_markup(self, markup, user_specified_encoding=None,
|
def prepare_markup(self, markup, user_specified_encoding=None,
|
||||||
|
exclude_encodings=None,
|
||||||
document_declared_encoding=None):
|
document_declared_encoding=None):
|
||||||
"""
|
"""
|
||||||
:yield: A series of 4-tuples.
|
:yield: A series of 4-tuples.
|
||||||
@@ -95,7 +104,8 @@ class LXMLTreeBuilderForXML(TreeBuilder):
|
|||||||
# the document as each one in turn.
|
# the document as each one in turn.
|
||||||
is_html = not self.is_xml
|
is_html = not self.is_xml
|
||||||
try_encodings = [user_specified_encoding, document_declared_encoding]
|
try_encodings = [user_specified_encoding, document_declared_encoding]
|
||||||
detector = EncodingDetector(markup, try_encodings, is_html)
|
detector = EncodingDetector(
|
||||||
|
markup, try_encodings, is_html, exclude_encodings)
|
||||||
for encoding in detector.encodings:
|
for encoding in detector.encodings:
|
||||||
yield (detector.markup, encoding, document_declared_encoding, False)
|
yield (detector.markup, encoding, document_declared_encoding, False)
|
||||||
|
|
||||||
@@ -189,7 +199,9 @@ class LXMLTreeBuilderForXML(TreeBuilder):
|
|||||||
self.nsmaps.pop()
|
self.nsmaps.pop()
|
||||||
|
|
||||||
def pi(self, target, data):
|
def pi(self, target, data):
|
||||||
pass
|
self.soup.endData()
|
||||||
|
self.soup.handle_data(target + ' ' + data)
|
||||||
|
self.soup.endData(ProcessingInstruction)
|
||||||
|
|
||||||
def data(self, content):
|
def data(self, content):
|
||||||
self.soup.handle_data(content)
|
self.soup.handle_data(content)
|
||||||
@@ -212,7 +224,10 @@ class LXMLTreeBuilderForXML(TreeBuilder):
|
|||||||
|
|
||||||
class LXMLTreeBuilder(HTMLTreeBuilder, LXMLTreeBuilderForXML):
|
class LXMLTreeBuilder(HTMLTreeBuilder, LXMLTreeBuilderForXML):
|
||||||
|
|
||||||
features = [LXML, HTML, FAST, PERMISSIVE]
|
NAME = LXML
|
||||||
|
ALTERNATE_NAMES = ["lxml-html"]
|
||||||
|
|
||||||
|
features = ALTERNATE_NAMES + [NAME, HTML, FAST, PERMISSIVE]
|
||||||
is_xml = False
|
is_xml = False
|
||||||
|
|
||||||
def default_parser(self, encoding):
|
def default_parser(self, encoding):
|
||||||
|
|||||||
+15
-5
@@ -3,10 +3,11 @@
|
|||||||
|
|
||||||
This library converts a bytestream to Unicode through any means
|
This library converts a bytestream to Unicode through any means
|
||||||
necessary. It is heavily based on code from Mark Pilgrim's Universal
|
necessary. It is heavily based on code from Mark Pilgrim's Universal
|
||||||
Feed Parser. It works best on XML and XML, but it does not rewrite the
|
Feed Parser. It works best on XML and HTML, but it does not rewrite the
|
||||||
XML or HTML to reflect a new encoding; that's the tree builder's job.
|
XML or HTML to reflect a new encoding; that's the tree builder's job.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
from pdb import set_trace
|
||||||
import codecs
|
import codecs
|
||||||
from htmlentitydefs import codepoint2name
|
from htmlentitydefs import codepoint2name
|
||||||
import re
|
import re
|
||||||
@@ -212,8 +213,11 @@ class EncodingDetector:
|
|||||||
|
|
||||||
5. Windows-1252.
|
5. Windows-1252.
|
||||||
"""
|
"""
|
||||||
def __init__(self, markup, override_encodings=None, is_html=False):
|
def __init__(self, markup, override_encodings=None, is_html=False,
|
||||||
|
exclude_encodings=None):
|
||||||
self.override_encodings = override_encodings or []
|
self.override_encodings = override_encodings or []
|
||||||
|
exclude_encodings = exclude_encodings or []
|
||||||
|
self.exclude_encodings = set([x.lower() for x in exclude_encodings])
|
||||||
self.chardet_encoding = None
|
self.chardet_encoding = None
|
||||||
self.is_html = is_html
|
self.is_html = is_html
|
||||||
self.declared_encoding = None
|
self.declared_encoding = None
|
||||||
@@ -224,6 +228,8 @@ class EncodingDetector:
|
|||||||
def _usable(self, encoding, tried):
|
def _usable(self, encoding, tried):
|
||||||
if encoding is not None:
|
if encoding is not None:
|
||||||
encoding = encoding.lower()
|
encoding = encoding.lower()
|
||||||
|
if encoding in self.exclude_encodings:
|
||||||
|
return False
|
||||||
if encoding not in tried:
|
if encoding not in tried:
|
||||||
tried.add(encoding)
|
tried.add(encoding)
|
||||||
return True
|
return True
|
||||||
@@ -266,6 +272,9 @@ class EncodingDetector:
|
|||||||
def strip_byte_order_mark(cls, data):
|
def strip_byte_order_mark(cls, data):
|
||||||
"""If a byte-order mark is present, strip it and return the encoding it implies."""
|
"""If a byte-order mark is present, strip it and return the encoding it implies."""
|
||||||
encoding = None
|
encoding = None
|
||||||
|
if isinstance(data, unicode):
|
||||||
|
# Unicode data cannot have a byte-order mark.
|
||||||
|
return data, encoding
|
||||||
if (len(data) >= 4) and (data[:2] == b'\xfe\xff') \
|
if (len(data) >= 4) and (data[:2] == b'\xfe\xff') \
|
||||||
and (data[2:4] != '\x00\x00'):
|
and (data[2:4] != '\x00\x00'):
|
||||||
encoding = 'utf-16be'
|
encoding = 'utf-16be'
|
||||||
@@ -306,7 +315,7 @@ class EncodingDetector:
|
|||||||
declared_encoding_match = html_meta_re.search(markup, endpos=html_endpos)
|
declared_encoding_match = html_meta_re.search(markup, endpos=html_endpos)
|
||||||
if declared_encoding_match is not None:
|
if declared_encoding_match is not None:
|
||||||
declared_encoding = declared_encoding_match.groups()[0].decode(
|
declared_encoding = declared_encoding_match.groups()[0].decode(
|
||||||
'ascii')
|
'ascii', 'replace')
|
||||||
if declared_encoding:
|
if declared_encoding:
|
||||||
return declared_encoding.lower()
|
return declared_encoding.lower()
|
||||||
return None
|
return None
|
||||||
@@ -331,13 +340,14 @@ class UnicodeDammit:
|
|||||||
]
|
]
|
||||||
|
|
||||||
def __init__(self, markup, override_encodings=[],
|
def __init__(self, markup, override_encodings=[],
|
||||||
smart_quotes_to=None, is_html=False):
|
smart_quotes_to=None, is_html=False, exclude_encodings=[]):
|
||||||
self.smart_quotes_to = smart_quotes_to
|
self.smart_quotes_to = smart_quotes_to
|
||||||
self.tried_encodings = []
|
self.tried_encodings = []
|
||||||
self.contains_replacement_characters = False
|
self.contains_replacement_characters = False
|
||||||
self.is_html = is_html
|
self.is_html = is_html
|
||||||
|
|
||||||
self.detector = EncodingDetector(markup, override_encodings, is_html)
|
self.detector = EncodingDetector(
|
||||||
|
markup, override_encodings, is_html, exclude_encodings)
|
||||||
|
|
||||||
# Short-circuit if the data is in Unicode to begin with.
|
# Short-circuit if the data is in Unicode to begin with.
|
||||||
if isinstance(markup, unicode) or markup == '':
|
if isinstance(markup, unicode) or markup == '':
|
||||||
|
|||||||
+13
-4
@@ -33,12 +33,21 @@ def diagnose(data):
|
|||||||
|
|
||||||
if 'lxml' in basic_parsers:
|
if 'lxml' in basic_parsers:
|
||||||
basic_parsers.append(["lxml", "xml"])
|
basic_parsers.append(["lxml", "xml"])
|
||||||
from lxml import etree
|
try:
|
||||||
print "Found lxml version %s" % ".".join(map(str,etree.LXML_VERSION))
|
from lxml import etree
|
||||||
|
print "Found lxml version %s" % ".".join(map(str,etree.LXML_VERSION))
|
||||||
|
except ImportError, e:
|
||||||
|
print (
|
||||||
|
"lxml is not installed or couldn't be imported.")
|
||||||
|
|
||||||
|
|
||||||
if 'html5lib' in basic_parsers:
|
if 'html5lib' in basic_parsers:
|
||||||
import html5lib
|
try:
|
||||||
print "Found html5lib version %s" % html5lib.__version__
|
import html5lib
|
||||||
|
print "Found html5lib version %s" % html5lib.__version__
|
||||||
|
except ImportError, e:
|
||||||
|
print (
|
||||||
|
"html5lib is not installed or couldn't be imported.")
|
||||||
|
|
||||||
if hasattr(data, 'read'):
|
if hasattr(data, 'read'):
|
||||||
data = data.read()
|
data = data.read()
|
||||||
|
|||||||
+271
-169
@@ -1,3 +1,4 @@
|
|||||||
|
from pdb import set_trace
|
||||||
import collections
|
import collections
|
||||||
import re
|
import re
|
||||||
import sys
|
import sys
|
||||||
@@ -185,24 +186,40 @@ class PageElement(object):
|
|||||||
return self.HTML_FORMATTERS.get(
|
return self.HTML_FORMATTERS.get(
|
||||||
name, HTMLAwareEntitySubstitution.substitute_xml)
|
name, HTMLAwareEntitySubstitution.substitute_xml)
|
||||||
|
|
||||||
def setup(self, parent=None, previous_element=None):
|
def setup(self, parent=None, previous_element=None, next_element=None,
|
||||||
|
previous_sibling=None, next_sibling=None):
|
||||||
"""Sets up the initial relations between this element and
|
"""Sets up the initial relations between this element and
|
||||||
other elements."""
|
other elements."""
|
||||||
self.parent = parent
|
self.parent = parent
|
||||||
|
|
||||||
self.previous_element = previous_element
|
self.previous_element = previous_element
|
||||||
if previous_element is not None:
|
if previous_element is not None:
|
||||||
self.previous_element.next_element = self
|
self.previous_element.next_element = self
|
||||||
self.next_element = None
|
|
||||||
self.previous_sibling = None
|
self.next_element = next_element
|
||||||
self.next_sibling = None
|
if self.next_element:
|
||||||
if self.parent is not None and self.parent.contents:
|
self.next_element.previous_element = self
|
||||||
self.previous_sibling = self.parent.contents[-1]
|
|
||||||
|
self.next_sibling = next_sibling
|
||||||
|
if self.next_sibling:
|
||||||
|
self.next_sibling.previous_sibling = self
|
||||||
|
|
||||||
|
if (not previous_sibling
|
||||||
|
and self.parent is not None and self.parent.contents):
|
||||||
|
previous_sibling = self.parent.contents[-1]
|
||||||
|
|
||||||
|
self.previous_sibling = previous_sibling
|
||||||
|
if previous_sibling:
|
||||||
self.previous_sibling.next_sibling = self
|
self.previous_sibling.next_sibling = self
|
||||||
|
|
||||||
nextSibling = _alias("next_sibling") # BS3
|
nextSibling = _alias("next_sibling") # BS3
|
||||||
previousSibling = _alias("previous_sibling") # BS3
|
previousSibling = _alias("previous_sibling") # BS3
|
||||||
|
|
||||||
def replace_with(self, replace_with):
|
def replace_with(self, replace_with):
|
||||||
|
if not self.parent:
|
||||||
|
raise ValueError(
|
||||||
|
"Cannot replace one element with another when the"
|
||||||
|
"element to be replaced is not part of a tree.")
|
||||||
if replace_with is self:
|
if replace_with is self:
|
||||||
return
|
return
|
||||||
if replace_with is self.parent:
|
if replace_with is self.parent:
|
||||||
@@ -216,6 +233,10 @@ class PageElement(object):
|
|||||||
|
|
||||||
def unwrap(self):
|
def unwrap(self):
|
||||||
my_parent = self.parent
|
my_parent = self.parent
|
||||||
|
if not self.parent:
|
||||||
|
raise ValueError(
|
||||||
|
"Cannot replace an element with its contents when that"
|
||||||
|
"element is not part of a tree.")
|
||||||
my_index = self.parent.index(self)
|
my_index = self.parent.index(self)
|
||||||
self.extract()
|
self.extract()
|
||||||
for child in reversed(self.contents[:]):
|
for child in reversed(self.contents[:]):
|
||||||
@@ -240,17 +261,20 @@ class PageElement(object):
|
|||||||
last_child = self._last_descendant()
|
last_child = self._last_descendant()
|
||||||
next_element = last_child.next_element
|
next_element = last_child.next_element
|
||||||
|
|
||||||
if self.previous_element is not None:
|
if (self.previous_element is not None and
|
||||||
|
self.previous_element != next_element):
|
||||||
self.previous_element.next_element = next_element
|
self.previous_element.next_element = next_element
|
||||||
if next_element is not None:
|
if next_element is not None and next_element != self.previous_element:
|
||||||
next_element.previous_element = self.previous_element
|
next_element.previous_element = self.previous_element
|
||||||
self.previous_element = None
|
self.previous_element = None
|
||||||
last_child.next_element = None
|
last_child.next_element = None
|
||||||
|
|
||||||
self.parent = None
|
self.parent = None
|
||||||
if self.previous_sibling is not None:
|
if (self.previous_sibling is not None
|
||||||
|
and self.previous_sibling != self.next_sibling):
|
||||||
self.previous_sibling.next_sibling = self.next_sibling
|
self.previous_sibling.next_sibling = self.next_sibling
|
||||||
if self.next_sibling is not None:
|
if (self.next_sibling is not None
|
||||||
|
and self.next_sibling != self.previous_sibling):
|
||||||
self.next_sibling.previous_sibling = self.previous_sibling
|
self.next_sibling.previous_sibling = self.previous_sibling
|
||||||
self.previous_sibling = self.next_sibling = None
|
self.previous_sibling = self.next_sibling = None
|
||||||
return self
|
return self
|
||||||
@@ -478,6 +502,10 @@ class PageElement(object):
|
|||||||
def _find_all(self, name, attrs, text, limit, generator, **kwargs):
|
def _find_all(self, name, attrs, text, limit, generator, **kwargs):
|
||||||
"Iterates over a generator looking for things that match."
|
"Iterates over a generator looking for things that match."
|
||||||
|
|
||||||
|
if text is None and 'string' in kwargs:
|
||||||
|
text = kwargs['string']
|
||||||
|
del kwargs['string']
|
||||||
|
|
||||||
if isinstance(name, SoupStrainer):
|
if isinstance(name, SoupStrainer):
|
||||||
strainer = name
|
strainer = name
|
||||||
else:
|
else:
|
||||||
@@ -548,17 +576,17 @@ class PageElement(object):
|
|||||||
|
|
||||||
# Methods for supporting CSS selectors.
|
# Methods for supporting CSS selectors.
|
||||||
|
|
||||||
tag_name_re = re.compile('^[a-z0-9]+$')
|
tag_name_re = re.compile('^[a-zA-Z0-9][-.a-zA-Z0-9:_]*$')
|
||||||
|
|
||||||
# /^(\w+)\[(\w+)([=~\|\^\$\*]?)=?"?([^\]"]*)"?\]$/
|
# /^([a-zA-Z0-9][-.a-zA-Z0-9:_]*)\[(\w+)([=~\|\^\$\*]?)=?"?([^\]"]*)"?\]$/
|
||||||
# \---/ \---/\-------------/ \-------/
|
# \---------------------------/ \---/\-------------/ \-------/
|
||||||
# | | | |
|
# | | | |
|
||||||
# | | | The value
|
# | | | The value
|
||||||
# | | ~,|,^,$,* or =
|
# | | ~,|,^,$,* or =
|
||||||
# | Attribute
|
# | Attribute
|
||||||
# Tag
|
# Tag
|
||||||
attribselect_re = re.compile(
|
attribselect_re = re.compile(
|
||||||
r'^(?P<tag>\w+)?\[(?P<attribute>\w+)(?P<operator>[=~\|\^\$\*]?)' +
|
r'^(?P<tag>[a-zA-Z0-9][-.a-zA-Z0-9:_]*)?\[(?P<attribute>[\w-]+)(?P<operator>[=~\|\^\$\*]?)' +
|
||||||
r'=?"?(?P<value>[^\]"]*)"?\]$'
|
r'=?"?(?P<value>[^\]"]*)"?\]$'
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -654,11 +682,17 @@ class NavigableString(unicode, PageElement):
|
|||||||
how to handle non-ASCII characters.
|
how to handle non-ASCII characters.
|
||||||
"""
|
"""
|
||||||
if isinstance(value, unicode):
|
if isinstance(value, unicode):
|
||||||
return unicode.__new__(cls, value)
|
u = unicode.__new__(cls, value)
|
||||||
return unicode.__new__(cls, value, DEFAULT_OUTPUT_ENCODING)
|
else:
|
||||||
|
u = unicode.__new__(cls, value, DEFAULT_OUTPUT_ENCODING)
|
||||||
|
u.setup()
|
||||||
|
return u
|
||||||
|
|
||||||
def __copy__(self):
|
def __copy__(self):
|
||||||
return self
|
"""A copy of a NavigableString has the same contents and class
|
||||||
|
as the original, but it is not connected to the parse tree.
|
||||||
|
"""
|
||||||
|
return type(self)(self)
|
||||||
|
|
||||||
def __getnewargs__(self):
|
def __getnewargs__(self):
|
||||||
return (unicode(self),)
|
return (unicode(self),)
|
||||||
@@ -707,7 +741,7 @@ class CData(PreformattedString):
|
|||||||
class ProcessingInstruction(PreformattedString):
|
class ProcessingInstruction(PreformattedString):
|
||||||
|
|
||||||
PREFIX = u'<?'
|
PREFIX = u'<?'
|
||||||
SUFFIX = u'?>'
|
SUFFIX = u'>'
|
||||||
|
|
||||||
class Comment(PreformattedString):
|
class Comment(PreformattedString):
|
||||||
|
|
||||||
@@ -759,9 +793,12 @@ class Tag(PageElement):
|
|||||||
self.prefix = prefix
|
self.prefix = prefix
|
||||||
if attrs is None:
|
if attrs is None:
|
||||||
attrs = {}
|
attrs = {}
|
||||||
elif attrs and builder.cdata_list_attributes:
|
elif attrs:
|
||||||
attrs = builder._replace_cdata_list_attribute_values(
|
if builder is not None and builder.cdata_list_attributes:
|
||||||
self.name, attrs)
|
attrs = builder._replace_cdata_list_attribute_values(
|
||||||
|
self.name, attrs)
|
||||||
|
else:
|
||||||
|
attrs = dict(attrs)
|
||||||
else:
|
else:
|
||||||
attrs = dict(attrs)
|
attrs = dict(attrs)
|
||||||
self.attrs = attrs
|
self.attrs = attrs
|
||||||
@@ -778,6 +815,18 @@ class Tag(PageElement):
|
|||||||
|
|
||||||
parserClass = _alias("parser_class") # BS3
|
parserClass = _alias("parser_class") # BS3
|
||||||
|
|
||||||
|
def __copy__(self):
|
||||||
|
"""A copy of a Tag is a new Tag, unconnected to the parse tree.
|
||||||
|
Its contents are a copy of the old Tag's contents.
|
||||||
|
"""
|
||||||
|
clone = type(self)(None, self.builder, self.name, self.namespace,
|
||||||
|
self.nsprefix, self.attrs)
|
||||||
|
for attr in ('can_be_empty_element', 'hidden'):
|
||||||
|
setattr(clone, attr, getattr(self, attr))
|
||||||
|
for child in self.contents:
|
||||||
|
clone.append(child.__copy__())
|
||||||
|
return clone
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def is_empty_element(self):
|
def is_empty_element(self):
|
||||||
"""Is this tag an empty-element tag? (aka a self-closing tag)
|
"""Is this tag an empty-element tag? (aka a self-closing tag)
|
||||||
@@ -971,15 +1020,25 @@ class Tag(PageElement):
|
|||||||
as defined in __eq__."""
|
as defined in __eq__."""
|
||||||
return not self == other
|
return not self == other
|
||||||
|
|
||||||
def __repr__(self, encoding=DEFAULT_OUTPUT_ENCODING):
|
def __repr__(self, encoding="unicode-escape"):
|
||||||
"""Renders this tag as a string."""
|
"""Renders this tag as a string."""
|
||||||
return self.encode(encoding)
|
if PY3K:
|
||||||
|
# "The return value must be a string object", i.e. Unicode
|
||||||
|
return self.decode()
|
||||||
|
else:
|
||||||
|
# "The return value must be a string object", i.e. a bytestring.
|
||||||
|
# By convention, the return value of __repr__ should also be
|
||||||
|
# an ASCII string.
|
||||||
|
return self.encode(encoding)
|
||||||
|
|
||||||
def __unicode__(self):
|
def __unicode__(self):
|
||||||
return self.decode()
|
return self.decode()
|
||||||
|
|
||||||
def __str__(self):
|
def __str__(self):
|
||||||
return self.encode()
|
if PY3K:
|
||||||
|
return self.decode()
|
||||||
|
else:
|
||||||
|
return self.encode()
|
||||||
|
|
||||||
if PY3K:
|
if PY3K:
|
||||||
__str__ = __repr__ = __unicode__
|
__str__ = __repr__ = __unicode__
|
||||||
@@ -1103,12 +1162,18 @@ class Tag(PageElement):
|
|||||||
formatter="minimal"):
|
formatter="minimal"):
|
||||||
"""Renders the contents of this tag as a Unicode string.
|
"""Renders the contents of this tag as a Unicode string.
|
||||||
|
|
||||||
|
:param indent_level: Each line of the rendering will be
|
||||||
|
indented this many spaces.
|
||||||
|
|
||||||
:param eventual_encoding: The tag is destined to be
|
:param eventual_encoding: The tag is destined to be
|
||||||
encoded into this encoding. This method is _not_
|
encoded into this encoding. This method is _not_
|
||||||
responsible for performing that encoding. This information
|
responsible for performing that encoding. This information
|
||||||
is passed in so that it can be substituted in if the
|
is passed in so that it can be substituted in if the
|
||||||
document contains a <META> tag that mentions the document's
|
document contains a <META> tag that mentions the document's
|
||||||
encoding.
|
encoding.
|
||||||
|
|
||||||
|
:param formatter: The output formatter responsible for converting
|
||||||
|
entities to Unicode characters.
|
||||||
"""
|
"""
|
||||||
# First off, turn a string formatter into a function. This
|
# First off, turn a string formatter into a function. This
|
||||||
# will stop the lookup from happening over and over again.
|
# will stop the lookup from happening over and over again.
|
||||||
@@ -1137,7 +1202,17 @@ class Tag(PageElement):
|
|||||||
def encode_contents(
|
def encode_contents(
|
||||||
self, indent_level=None, encoding=DEFAULT_OUTPUT_ENCODING,
|
self, indent_level=None, encoding=DEFAULT_OUTPUT_ENCODING,
|
||||||
formatter="minimal"):
|
formatter="minimal"):
|
||||||
"""Renders the contents of this tag as a bytestring."""
|
"""Renders the contents of this tag as a bytestring.
|
||||||
|
|
||||||
|
:param indent_level: Each line of the rendering will be
|
||||||
|
indented this many spaces.
|
||||||
|
|
||||||
|
:param eventual_encoding: The bytestring will be in this encoding.
|
||||||
|
|
||||||
|
:param formatter: The output formatter responsible for converting
|
||||||
|
entities to Unicode characters.
|
||||||
|
"""
|
||||||
|
|
||||||
contents = self.decode_contents(indent_level, encoding, formatter)
|
contents = self.decode_contents(indent_level, encoding, formatter)
|
||||||
return contents.encode(encoding)
|
return contents.encode(encoding)
|
||||||
|
|
||||||
@@ -1201,63 +1276,89 @@ class Tag(PageElement):
|
|||||||
|
|
||||||
_selector_combinators = ['>', '+', '~']
|
_selector_combinators = ['>', '+', '~']
|
||||||
_select_debug = False
|
_select_debug = False
|
||||||
def select(self, selector, _candidate_generator=None):
|
def select_one(self, selector):
|
||||||
"""Perform a CSS selection operation on the current element."""
|
"""Perform a CSS selection operation on the current element."""
|
||||||
tokens = selector.split()
|
value = self.select(selector, limit=1)
|
||||||
|
if value:
|
||||||
|
return value[0]
|
||||||
|
return None
|
||||||
|
|
||||||
|
def select(self, selector, _candidate_generator=None, limit=None):
|
||||||
|
"""Perform a CSS selection operation on the current element."""
|
||||||
|
|
||||||
|
# Remove whitespace directly after the grouping operator ','
|
||||||
|
# then split into tokens.
|
||||||
|
tokens = re.sub(',[\s]*',',', selector).split()
|
||||||
current_context = [self]
|
current_context = [self]
|
||||||
|
|
||||||
if tokens[-1] in self._selector_combinators:
|
if tokens[-1] in self._selector_combinators:
|
||||||
raise ValueError(
|
raise ValueError(
|
||||||
'Final combinator "%s" is missing an argument.' % tokens[-1])
|
'Final combinator "%s" is missing an argument.' % tokens[-1])
|
||||||
|
|
||||||
if self._select_debug:
|
if self._select_debug:
|
||||||
print 'Running CSS selector "%s"' % selector
|
print 'Running CSS selector "%s"' % selector
|
||||||
for index, token in enumerate(tokens):
|
|
||||||
if self._select_debug:
|
for index, token_group in enumerate(tokens):
|
||||||
print ' Considering token "%s"' % token
|
new_context = []
|
||||||
recursive_candidate_generator = None
|
new_context_ids = set([])
|
||||||
tag_name = None
|
|
||||||
|
# Grouping selectors, ie: p,a
|
||||||
|
grouped_tokens = token_group.split(',')
|
||||||
|
if '' in grouped_tokens:
|
||||||
|
raise ValueError('Invalid group selection syntax: %s' % token_group)
|
||||||
|
|
||||||
if tokens[index-1] in self._selector_combinators:
|
if tokens[index-1] in self._selector_combinators:
|
||||||
# This token was consumed by the previous combinator. Skip it.
|
# This token was consumed by the previous combinator. Skip it.
|
||||||
if self._select_debug:
|
if self._select_debug:
|
||||||
print ' Token was consumed by the previous combinator.'
|
print ' Token was consumed by the previous combinator.'
|
||||||
continue
|
continue
|
||||||
# Each operation corresponds to a checker function, a rule
|
|
||||||
# for determining whether a candidate matches the
|
|
||||||
# selector. Candidates are generated by the active
|
|
||||||
# iterator.
|
|
||||||
checker = None
|
|
||||||
|
|
||||||
m = self.attribselect_re.match(token)
|
for token in grouped_tokens:
|
||||||
if m is not None:
|
if self._select_debug:
|
||||||
# Attribute selector
|
print ' Considering token "%s"' % token
|
||||||
tag_name, attribute, operator, value = m.groups()
|
recursive_candidate_generator = None
|
||||||
checker = self._attribute_checker(operator, attribute, value)
|
tag_name = None
|
||||||
|
|
||||||
elif '#' in token:
|
# Each operation corresponds to a checker function, a rule
|
||||||
# ID selector
|
# for determining whether a candidate matches the
|
||||||
tag_name, tag_id = token.split('#', 1)
|
# selector. Candidates are generated by the active
|
||||||
def id_matches(tag):
|
# iterator.
|
||||||
return tag.get('id', None) == tag_id
|
checker = None
|
||||||
checker = id_matches
|
|
||||||
|
|
||||||
elif '.' in token:
|
m = self.attribselect_re.match(token)
|
||||||
# Class selector
|
if m is not None:
|
||||||
tag_name, klass = token.split('.', 1)
|
# Attribute selector
|
||||||
classes = set(klass.split('.'))
|
tag_name, attribute, operator, value = m.groups()
|
||||||
def classes_match(candidate):
|
checker = self._attribute_checker(operator, attribute, value)
|
||||||
return classes.issubset(candidate.get('class', []))
|
|
||||||
checker = classes_match
|
|
||||||
|
|
||||||
elif ':' in token:
|
elif '#' in token:
|
||||||
# Pseudo-class
|
# ID selector
|
||||||
tag_name, pseudo = token.split(':', 1)
|
tag_name, tag_id = token.split('#', 1)
|
||||||
if tag_name == '':
|
def id_matches(tag):
|
||||||
raise ValueError(
|
return tag.get('id', None) == tag_id
|
||||||
"A pseudo-class must be prefixed with a tag name.")
|
checker = id_matches
|
||||||
pseudo_attributes = re.match('([a-zA-Z\d-]+)\(([a-zA-Z\d]+)\)', pseudo)
|
|
||||||
found = []
|
elif '.' in token:
|
||||||
if pseudo_attributes is not None:
|
# Class selector
|
||||||
pseudo_type, pseudo_value = pseudo_attributes.groups()
|
tag_name, klass = token.split('.', 1)
|
||||||
|
classes = set(klass.split('.'))
|
||||||
|
def classes_match(candidate):
|
||||||
|
return classes.issubset(candidate.get('class', []))
|
||||||
|
checker = classes_match
|
||||||
|
|
||||||
|
elif ':' in token:
|
||||||
|
# Pseudo-class
|
||||||
|
tag_name, pseudo = token.split(':', 1)
|
||||||
|
if tag_name == '':
|
||||||
|
raise ValueError(
|
||||||
|
"A pseudo-class must be prefixed with a tag name.")
|
||||||
|
pseudo_attributes = re.match('([a-zA-Z\d-]+)\(([a-zA-Z\d]+)\)', pseudo)
|
||||||
|
found = []
|
||||||
|
if pseudo_attributes is None:
|
||||||
|
pseudo_type = pseudo
|
||||||
|
pseudo_value = None
|
||||||
|
else:
|
||||||
|
pseudo_type, pseudo_value = pseudo_attributes.groups()
|
||||||
if pseudo_type == 'nth-of-type':
|
if pseudo_type == 'nth-of-type':
|
||||||
try:
|
try:
|
||||||
pseudo_value = int(pseudo_value)
|
pseudo_value = int(pseudo_value)
|
||||||
@@ -1286,109 +1387,110 @@ class Tag(PageElement):
|
|||||||
raise NotImplementedError(
|
raise NotImplementedError(
|
||||||
'Only the following pseudo-classes are implemented: nth-of-type.')
|
'Only the following pseudo-classes are implemented: nth-of-type.')
|
||||||
|
|
||||||
elif token == '*':
|
elif token == '*':
|
||||||
# Star selector -- matches everything
|
# Star selector -- matches everything
|
||||||
pass
|
pass
|
||||||
elif token == '>':
|
elif token == '>':
|
||||||
# Run the next token as a CSS selector against the
|
# Run the next token as a CSS selector against the
|
||||||
# direct children of each tag in the current context.
|
# direct children of each tag in the current context.
|
||||||
recursive_candidate_generator = lambda tag: tag.children
|
recursive_candidate_generator = lambda tag: tag.children
|
||||||
elif token == '~':
|
elif token == '~':
|
||||||
# Run the next token as a CSS selector against the
|
# Run the next token as a CSS selector against the
|
||||||
# siblings of each tag in the current context.
|
# siblings of each tag in the current context.
|
||||||
recursive_candidate_generator = lambda tag: tag.next_siblings
|
recursive_candidate_generator = lambda tag: tag.next_siblings
|
||||||
elif token == '+':
|
elif token == '+':
|
||||||
# For each tag in the current context, run the next
|
# For each tag in the current context, run the next
|
||||||
# token as a CSS selector against the tag's next
|
# token as a CSS selector against the tag's next
|
||||||
# sibling that's a tag.
|
# sibling that's a tag.
|
||||||
def next_tag_sibling(tag):
|
def next_tag_sibling(tag):
|
||||||
yield tag.find_next_sibling(True)
|
yield tag.find_next_sibling(True)
|
||||||
recursive_candidate_generator = next_tag_sibling
|
recursive_candidate_generator = next_tag_sibling
|
||||||
|
|
||||||
elif self.tag_name_re.match(token):
|
elif self.tag_name_re.match(token):
|
||||||
# Just a tag name.
|
# Just a tag name.
|
||||||
tag_name = token
|
tag_name = token
|
||||||
else:
|
|
||||||
raise ValueError(
|
|
||||||
'Unsupported or invalid CSS selector: "%s"' % token)
|
|
||||||
|
|
||||||
if recursive_candidate_generator:
|
|
||||||
# This happens when the selector looks like "> foo".
|
|
||||||
#
|
|
||||||
# The generator calls select() recursively on every
|
|
||||||
# member of the current context, passing in a different
|
|
||||||
# candidate generator and a different selector.
|
|
||||||
#
|
|
||||||
# In the case of "> foo", the candidate generator is
|
|
||||||
# one that yields a tag's direct children (">"), and
|
|
||||||
# the selector is "foo".
|
|
||||||
next_token = tokens[index+1]
|
|
||||||
def recursive_select(tag):
|
|
||||||
if self._select_debug:
|
|
||||||
print ' Calling select("%s") recursively on %s %s' % (next_token, tag.name, tag.attrs)
|
|
||||||
print '-' * 40
|
|
||||||
for i in tag.select(next_token, recursive_candidate_generator):
|
|
||||||
if self._select_debug:
|
|
||||||
print '(Recursive select picked up candidate %s %s)' % (i.name, i.attrs)
|
|
||||||
yield i
|
|
||||||
if self._select_debug:
|
|
||||||
print '-' * 40
|
|
||||||
_use_candidate_generator = recursive_select
|
|
||||||
elif _candidate_generator is None:
|
|
||||||
# By default, a tag's candidates are all of its
|
|
||||||
# children. If tag_name is defined, only yield tags
|
|
||||||
# with that name.
|
|
||||||
if self._select_debug:
|
|
||||||
if tag_name:
|
|
||||||
check = "[any]"
|
|
||||||
else:
|
|
||||||
check = tag_name
|
|
||||||
print ' Default candidate generator, tag name="%s"' % check
|
|
||||||
if self._select_debug:
|
|
||||||
# This is redundant with later code, but it stops
|
|
||||||
# a bunch of bogus tags from cluttering up the
|
|
||||||
# debug log.
|
|
||||||
def default_candidate_generator(tag):
|
|
||||||
for child in tag.descendants:
|
|
||||||
if not isinstance(child, Tag):
|
|
||||||
continue
|
|
||||||
if tag_name and not child.name == tag_name:
|
|
||||||
continue
|
|
||||||
yield child
|
|
||||||
_use_candidate_generator = default_candidate_generator
|
|
||||||
else:
|
else:
|
||||||
_use_candidate_generator = lambda tag: tag.descendants
|
raise ValueError(
|
||||||
else:
|
'Unsupported or invalid CSS selector: "%s"' % token)
|
||||||
_use_candidate_generator = _candidate_generator
|
if recursive_candidate_generator:
|
||||||
|
# This happens when the selector looks like "> foo".
|
||||||
new_context = []
|
#
|
||||||
new_context_ids = set([])
|
# The generator calls select() recursively on every
|
||||||
for tag in current_context:
|
# member of the current context, passing in a different
|
||||||
if self._select_debug:
|
# candidate generator and a different selector.
|
||||||
print " Running candidate generator on %s %s" % (
|
#
|
||||||
tag.name, repr(tag.attrs))
|
# In the case of "> foo", the candidate generator is
|
||||||
for candidate in _use_candidate_generator(tag):
|
# one that yields a tag's direct children (">"), and
|
||||||
if not isinstance(candidate, Tag):
|
# the selector is "foo".
|
||||||
continue
|
next_token = tokens[index+1]
|
||||||
if tag_name and candidate.name != tag_name:
|
def recursive_select(tag):
|
||||||
continue
|
|
||||||
if checker is not None:
|
|
||||||
try:
|
|
||||||
result = checker(candidate)
|
|
||||||
except StopIteration:
|
|
||||||
# The checker has decided we should no longer
|
|
||||||
# run the generator.
|
|
||||||
break
|
|
||||||
if checker is None or result:
|
|
||||||
if self._select_debug:
|
if self._select_debug:
|
||||||
print " SUCCESS %s %s" % (candidate.name, repr(candidate.attrs))
|
print ' Calling select("%s") recursively on %s %s' % (next_token, tag.name, tag.attrs)
|
||||||
if id(candidate) not in new_context_ids:
|
print '-' * 40
|
||||||
# If a tag matches a selector more than once,
|
for i in tag.select(next_token, recursive_candidate_generator):
|
||||||
# don't include it in the context more than once.
|
if self._select_debug:
|
||||||
new_context.append(candidate)
|
print '(Recursive select picked up candidate %s %s)' % (i.name, i.attrs)
|
||||||
new_context_ids.add(id(candidate))
|
yield i
|
||||||
elif self._select_debug:
|
if self._select_debug:
|
||||||
print " FAILURE %s %s" % (candidate.name, repr(candidate.attrs))
|
print '-' * 40
|
||||||
|
_use_candidate_generator = recursive_select
|
||||||
|
elif _candidate_generator is None:
|
||||||
|
# By default, a tag's candidates are all of its
|
||||||
|
# children. If tag_name is defined, only yield tags
|
||||||
|
# with that name.
|
||||||
|
if self._select_debug:
|
||||||
|
if tag_name:
|
||||||
|
check = "[any]"
|
||||||
|
else:
|
||||||
|
check = tag_name
|
||||||
|
print ' Default candidate generator, tag name="%s"' % check
|
||||||
|
if self._select_debug:
|
||||||
|
# This is redundant with later code, but it stops
|
||||||
|
# a bunch of bogus tags from cluttering up the
|
||||||
|
# debug log.
|
||||||
|
def default_candidate_generator(tag):
|
||||||
|
for child in tag.descendants:
|
||||||
|
if not isinstance(child, Tag):
|
||||||
|
continue
|
||||||
|
if tag_name and not child.name == tag_name:
|
||||||
|
continue
|
||||||
|
yield child
|
||||||
|
_use_candidate_generator = default_candidate_generator
|
||||||
|
else:
|
||||||
|
_use_candidate_generator = lambda tag: tag.descendants
|
||||||
|
else:
|
||||||
|
_use_candidate_generator = _candidate_generator
|
||||||
|
|
||||||
|
count = 0
|
||||||
|
for tag in current_context:
|
||||||
|
if self._select_debug:
|
||||||
|
print " Running candidate generator on %s %s" % (
|
||||||
|
tag.name, repr(tag.attrs))
|
||||||
|
for candidate in _use_candidate_generator(tag):
|
||||||
|
if not isinstance(candidate, Tag):
|
||||||
|
continue
|
||||||
|
if tag_name and candidate.name != tag_name:
|
||||||
|
continue
|
||||||
|
if checker is not None:
|
||||||
|
try:
|
||||||
|
result = checker(candidate)
|
||||||
|
except StopIteration:
|
||||||
|
# The checker has decided we should no longer
|
||||||
|
# run the generator.
|
||||||
|
break
|
||||||
|
if checker is None or result:
|
||||||
|
if self._select_debug:
|
||||||
|
print " SUCCESS %s %s" % (candidate.name, repr(candidate.attrs))
|
||||||
|
if id(candidate) not in new_context_ids:
|
||||||
|
# If a tag matches a selector more than once,
|
||||||
|
# don't include it in the context more than once.
|
||||||
|
new_context.append(candidate)
|
||||||
|
new_context_ids.add(id(candidate))
|
||||||
|
if limit and len(new_context) >= limit:
|
||||||
|
break
|
||||||
|
elif self._select_debug:
|
||||||
|
print " FAILURE %s %s" % (candidate.name, repr(candidate.attrs))
|
||||||
|
|
||||||
|
|
||||||
current_context = new_context
|
current_context = new_context
|
||||||
|
|
||||||
|
|||||||
+89
-1
@@ -1,5 +1,6 @@
|
|||||||
"""Helper classes for tests."""
|
"""Helper classes for tests."""
|
||||||
|
|
||||||
|
import pickle
|
||||||
import copy
|
import copy
|
||||||
import functools
|
import functools
|
||||||
import unittest
|
import unittest
|
||||||
@@ -43,6 +44,16 @@ class SoupTest(unittest.TestCase):
|
|||||||
|
|
||||||
self.assertEqual(obj.decode(), self.document_for(compare_parsed_to))
|
self.assertEqual(obj.decode(), self.document_for(compare_parsed_to))
|
||||||
|
|
||||||
|
def assertConnectedness(self, element):
|
||||||
|
"""Ensure that next_element and previous_element are properly
|
||||||
|
set for all descendants of the given element.
|
||||||
|
"""
|
||||||
|
earlier = None
|
||||||
|
for e in element.descendants:
|
||||||
|
if earlier:
|
||||||
|
self.assertEqual(e, earlier.next_element)
|
||||||
|
self.assertEqual(earlier, e.previous_element)
|
||||||
|
earlier = e
|
||||||
|
|
||||||
class HTMLTreeBuilderSmokeTest(object):
|
class HTMLTreeBuilderSmokeTest(object):
|
||||||
|
|
||||||
@@ -54,6 +65,15 @@ class HTMLTreeBuilderSmokeTest(object):
|
|||||||
markup in these tests, there's not much room for interpretation.
|
markup in these tests, there's not much room for interpretation.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
def test_pickle_and_unpickle_identity(self):
|
||||||
|
# Pickling a tree, then unpickling it, yields a tree identical
|
||||||
|
# to the original.
|
||||||
|
tree = self.soup("<a><b>foo</a>")
|
||||||
|
dumped = pickle.dumps(tree, 2)
|
||||||
|
loaded = pickle.loads(dumped)
|
||||||
|
self.assertEqual(loaded.__class__, BeautifulSoup)
|
||||||
|
self.assertEqual(loaded.decode(), tree.decode())
|
||||||
|
|
||||||
def assertDoctypeHandled(self, doctype_fragment):
|
def assertDoctypeHandled(self, doctype_fragment):
|
||||||
"""Assert that a given doctype string is handled correctly."""
|
"""Assert that a given doctype string is handled correctly."""
|
||||||
doctype_str, soup = self._document_with_doctype(doctype_fragment)
|
doctype_str, soup = self._document_with_doctype(doctype_fragment)
|
||||||
@@ -114,6 +134,11 @@ class HTMLTreeBuilderSmokeTest(object):
|
|||||||
soup.encode("utf-8").replace(b"\n", b""),
|
soup.encode("utf-8").replace(b"\n", b""),
|
||||||
markup.replace(b"\n", b""))
|
markup.replace(b"\n", b""))
|
||||||
|
|
||||||
|
def test_processing_instruction(self):
|
||||||
|
markup = b"""<?PITarget PIContent?>"""
|
||||||
|
soup = self.soup(markup)
|
||||||
|
self.assertEqual(markup, soup.encode("utf8"))
|
||||||
|
|
||||||
def test_deepcopy(self):
|
def test_deepcopy(self):
|
||||||
"""Make sure you can copy the tree builder.
|
"""Make sure you can copy the tree builder.
|
||||||
|
|
||||||
@@ -155,6 +180,23 @@ class HTMLTreeBuilderSmokeTest(object):
|
|||||||
def test_nested_formatting_elements(self):
|
def test_nested_formatting_elements(self):
|
||||||
self.assertSoupEquals("<em><em></em></em>")
|
self.assertSoupEquals("<em><em></em></em>")
|
||||||
|
|
||||||
|
def test_double_head(self):
|
||||||
|
html = '''<!DOCTYPE html>
|
||||||
|
<html>
|
||||||
|
<head>
|
||||||
|
<title>Ordinary HEAD element test</title>
|
||||||
|
</head>
|
||||||
|
<script type="text/javascript">
|
||||||
|
alert("Help!");
|
||||||
|
</script>
|
||||||
|
<body>
|
||||||
|
Hello, world!
|
||||||
|
</body>
|
||||||
|
</html>
|
||||||
|
'''
|
||||||
|
soup = self.soup(html)
|
||||||
|
self.assertEqual("text/javascript", soup.find('script')['type'])
|
||||||
|
|
||||||
def test_comment(self):
|
def test_comment(self):
|
||||||
# Comments are represented as Comment objects.
|
# Comments are represented as Comment objects.
|
||||||
markup = "<p>foo<!--foobar-->baz</p>"
|
markup = "<p>foo<!--foobar-->baz</p>"
|
||||||
@@ -221,6 +263,14 @@ class HTMLTreeBuilderSmokeTest(object):
|
|||||||
soup = self.soup(markup)
|
soup = self.soup(markup)
|
||||||
self.assertEqual(["css"], soup.div.div['class'])
|
self.assertEqual(["css"], soup.div.div['class'])
|
||||||
|
|
||||||
|
def test_multivalued_attribute_on_html(self):
|
||||||
|
# html5lib uses a different API to set the attributes ot the
|
||||||
|
# <html> tag. This has caused problems with multivalued
|
||||||
|
# attributes.
|
||||||
|
markup = '<html class="a b"></html>'
|
||||||
|
soup = self.soup(markup)
|
||||||
|
self.assertEqual(["a", "b"], soup.html['class'])
|
||||||
|
|
||||||
def test_angle_brackets_in_attribute_values_are_escaped(self):
|
def test_angle_brackets_in_attribute_values_are_escaped(self):
|
||||||
self.assertSoupEquals('<a b="<a>"></a>', '<a b="<a>"></a>')
|
self.assertSoupEquals('<a b="<a>"></a>', '<a b="<a>"></a>')
|
||||||
|
|
||||||
@@ -253,6 +303,35 @@ class HTMLTreeBuilderSmokeTest(object):
|
|||||||
soup = self.soup("<html><h2>\nfoo</h2><p></p></html>")
|
soup = self.soup("<html><h2>\nfoo</h2><p></p></html>")
|
||||||
self.assertEqual("p", soup.h2.string.next_element.name)
|
self.assertEqual("p", soup.h2.string.next_element.name)
|
||||||
self.assertEqual("p", soup.p.name)
|
self.assertEqual("p", soup.p.name)
|
||||||
|
self.assertConnectedness(soup)
|
||||||
|
|
||||||
|
def test_head_tag_between_head_and_body(self):
|
||||||
|
"Prevent recurrence of a bug in the html5lib treebuilder."
|
||||||
|
content = """<html><head></head>
|
||||||
|
<link></link>
|
||||||
|
<body>foo</body>
|
||||||
|
</html>
|
||||||
|
"""
|
||||||
|
soup = self.soup(content)
|
||||||
|
self.assertNotEqual(None, soup.html.body)
|
||||||
|
self.assertConnectedness(soup)
|
||||||
|
|
||||||
|
def test_multiple_copies_of_a_tag(self):
|
||||||
|
"Prevent recurrence of a bug in the html5lib treebuilder."
|
||||||
|
content = """<!DOCTYPE html>
|
||||||
|
<html>
|
||||||
|
<body>
|
||||||
|
<article id="a" >
|
||||||
|
<div><a href="1"></div>
|
||||||
|
<footer>
|
||||||
|
<a href="2"></a>
|
||||||
|
</footer>
|
||||||
|
</article>
|
||||||
|
</body>
|
||||||
|
</html>
|
||||||
|
"""
|
||||||
|
soup = self.soup(content)
|
||||||
|
self.assertConnectedness(soup.article)
|
||||||
|
|
||||||
def test_basic_namespaces(self):
|
def test_basic_namespaces(self):
|
||||||
"""Parsers don't need to *understand* namespaces, but at the
|
"""Parsers don't need to *understand* namespaces, but at the
|
||||||
@@ -463,6 +542,15 @@ class HTMLTreeBuilderSmokeTest(object):
|
|||||||
|
|
||||||
class XMLTreeBuilderSmokeTest(object):
|
class XMLTreeBuilderSmokeTest(object):
|
||||||
|
|
||||||
|
def test_pickle_and_unpickle_identity(self):
|
||||||
|
# Pickling a tree, then unpickling it, yields a tree identical
|
||||||
|
# to the original.
|
||||||
|
tree = self.soup("<a><b>foo</a>")
|
||||||
|
dumped = pickle.dumps(tree, 2)
|
||||||
|
loaded = pickle.loads(dumped)
|
||||||
|
self.assertEqual(loaded.__class__, BeautifulSoup)
|
||||||
|
self.assertEqual(loaded.decode(), tree.decode())
|
||||||
|
|
||||||
def test_docstring_generated(self):
|
def test_docstring_generated(self):
|
||||||
soup = self.soup("<root/>")
|
soup = self.soup("<root/>")
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
@@ -485,7 +573,7 @@ class XMLTreeBuilderSmokeTest(object):
|
|||||||
<script type="text/javascript">
|
<script type="text/javascript">
|
||||||
</script>
|
</script>
|
||||||
"""
|
"""
|
||||||
soup = BeautifulSoup(doc, "xml")
|
soup = BeautifulSoup(doc, "lxml-xml")
|
||||||
# lxml would have stripped this while parsing, but we can add
|
# lxml would have stripped this while parsing, but we can add
|
||||||
# it later.
|
# it later.
|
||||||
soup.script.string = 'console.log("< < hey > > ");'
|
soup.script.string = 'console.log("< < hey > > ");'
|
||||||
|
|||||||
@@ -20,4 +20,6 @@ from .serializer import serialize
|
|||||||
|
|
||||||
__all__ = ["HTMLParser", "parse", "parseFragment", "getTreeBuilder",
|
__all__ = ["HTMLParser", "parse", "parseFragment", "getTreeBuilder",
|
||||||
"getTreeWalker", "serialize"]
|
"getTreeWalker", "serialize"]
|
||||||
__version__ = "0.999"
|
|
||||||
|
# this has to be at the top level, see how setup.py parses this
|
||||||
|
__version__ = "0.999999"
|
||||||
|
|||||||
+193
-195
@@ -1,292 +1,290 @@
|
|||||||
from __future__ import absolute_import, division, unicode_literals
|
from __future__ import absolute_import, division, unicode_literals
|
||||||
|
|
||||||
import string
|
import string
|
||||||
import gettext
|
|
||||||
_ = gettext.gettext
|
|
||||||
|
|
||||||
EOF = None
|
EOF = None
|
||||||
|
|
||||||
E = {
|
E = {
|
||||||
"null-character":
|
"null-character":
|
||||||
_("Null character in input stream, replaced with U+FFFD."),
|
"Null character in input stream, replaced with U+FFFD.",
|
||||||
"invalid-codepoint":
|
"invalid-codepoint":
|
||||||
_("Invalid codepoint in stream."),
|
"Invalid codepoint in stream.",
|
||||||
"incorrectly-placed-solidus":
|
"incorrectly-placed-solidus":
|
||||||
_("Solidus (/) incorrectly placed in tag."),
|
"Solidus (/) incorrectly placed in tag.",
|
||||||
"incorrect-cr-newline-entity":
|
"incorrect-cr-newline-entity":
|
||||||
_("Incorrect CR newline entity, replaced with LF."),
|
"Incorrect CR newline entity, replaced with LF.",
|
||||||
"illegal-windows-1252-entity":
|
"illegal-windows-1252-entity":
|
||||||
_("Entity used with illegal number (windows-1252 reference)."),
|
"Entity used with illegal number (windows-1252 reference).",
|
||||||
"cant-convert-numeric-entity":
|
"cant-convert-numeric-entity":
|
||||||
_("Numeric entity couldn't be converted to character "
|
"Numeric entity couldn't be converted to character "
|
||||||
"(codepoint U+%(charAsInt)08x)."),
|
"(codepoint U+%(charAsInt)08x).",
|
||||||
"illegal-codepoint-for-numeric-entity":
|
"illegal-codepoint-for-numeric-entity":
|
||||||
_("Numeric entity represents an illegal codepoint: "
|
"Numeric entity represents an illegal codepoint: "
|
||||||
"U+%(charAsInt)08x."),
|
"U+%(charAsInt)08x.",
|
||||||
"numeric-entity-without-semicolon":
|
"numeric-entity-without-semicolon":
|
||||||
_("Numeric entity didn't end with ';'."),
|
"Numeric entity didn't end with ';'.",
|
||||||
"expected-numeric-entity-but-got-eof":
|
"expected-numeric-entity-but-got-eof":
|
||||||
_("Numeric entity expected. Got end of file instead."),
|
"Numeric entity expected. Got end of file instead.",
|
||||||
"expected-numeric-entity":
|
"expected-numeric-entity":
|
||||||
_("Numeric entity expected but none found."),
|
"Numeric entity expected but none found.",
|
||||||
"named-entity-without-semicolon":
|
"named-entity-without-semicolon":
|
||||||
_("Named entity didn't end with ';'."),
|
"Named entity didn't end with ';'.",
|
||||||
"expected-named-entity":
|
"expected-named-entity":
|
||||||
_("Named entity expected. Got none."),
|
"Named entity expected. Got none.",
|
||||||
"attributes-in-end-tag":
|
"attributes-in-end-tag":
|
||||||
_("End tag contains unexpected attributes."),
|
"End tag contains unexpected attributes.",
|
||||||
'self-closing-flag-on-end-tag':
|
'self-closing-flag-on-end-tag':
|
||||||
_("End tag contains unexpected self-closing flag."),
|
"End tag contains unexpected self-closing flag.",
|
||||||
"expected-tag-name-but-got-right-bracket":
|
"expected-tag-name-but-got-right-bracket":
|
||||||
_("Expected tag name. Got '>' instead."),
|
"Expected tag name. Got '>' instead.",
|
||||||
"expected-tag-name-but-got-question-mark":
|
"expected-tag-name-but-got-question-mark":
|
||||||
_("Expected tag name. Got '?' instead. (HTML doesn't "
|
"Expected tag name. Got '?' instead. (HTML doesn't "
|
||||||
"support processing instructions.)"),
|
"support processing instructions.)",
|
||||||
"expected-tag-name":
|
"expected-tag-name":
|
||||||
_("Expected tag name. Got something else instead"),
|
"Expected tag name. Got something else instead",
|
||||||
"expected-closing-tag-but-got-right-bracket":
|
"expected-closing-tag-but-got-right-bracket":
|
||||||
_("Expected closing tag. Got '>' instead. Ignoring '</>'."),
|
"Expected closing tag. Got '>' instead. Ignoring '</>'.",
|
||||||
"expected-closing-tag-but-got-eof":
|
"expected-closing-tag-but-got-eof":
|
||||||
_("Expected closing tag. Unexpected end of file."),
|
"Expected closing tag. Unexpected end of file.",
|
||||||
"expected-closing-tag-but-got-char":
|
"expected-closing-tag-but-got-char":
|
||||||
_("Expected closing tag. Unexpected character '%(data)s' found."),
|
"Expected closing tag. Unexpected character '%(data)s' found.",
|
||||||
"eof-in-tag-name":
|
"eof-in-tag-name":
|
||||||
_("Unexpected end of file in the tag name."),
|
"Unexpected end of file in the tag name.",
|
||||||
"expected-attribute-name-but-got-eof":
|
"expected-attribute-name-but-got-eof":
|
||||||
_("Unexpected end of file. Expected attribute name instead."),
|
"Unexpected end of file. Expected attribute name instead.",
|
||||||
"eof-in-attribute-name":
|
"eof-in-attribute-name":
|
||||||
_("Unexpected end of file in attribute name."),
|
"Unexpected end of file in attribute name.",
|
||||||
"invalid-character-in-attribute-name":
|
"invalid-character-in-attribute-name":
|
||||||
_("Invalid character in attribute name"),
|
"Invalid character in attribute name",
|
||||||
"duplicate-attribute":
|
"duplicate-attribute":
|
||||||
_("Dropped duplicate attribute on tag."),
|
"Dropped duplicate attribute on tag.",
|
||||||
"expected-end-of-tag-name-but-got-eof":
|
"expected-end-of-tag-name-but-got-eof":
|
||||||
_("Unexpected end of file. Expected = or end of tag."),
|
"Unexpected end of file. Expected = or end of tag.",
|
||||||
"expected-attribute-value-but-got-eof":
|
"expected-attribute-value-but-got-eof":
|
||||||
_("Unexpected end of file. Expected attribute value."),
|
"Unexpected end of file. Expected attribute value.",
|
||||||
"expected-attribute-value-but-got-right-bracket":
|
"expected-attribute-value-but-got-right-bracket":
|
||||||
_("Expected attribute value. Got '>' instead."),
|
"Expected attribute value. Got '>' instead.",
|
||||||
'equals-in-unquoted-attribute-value':
|
'equals-in-unquoted-attribute-value':
|
||||||
_("Unexpected = in unquoted attribute"),
|
"Unexpected = in unquoted attribute",
|
||||||
'unexpected-character-in-unquoted-attribute-value':
|
'unexpected-character-in-unquoted-attribute-value':
|
||||||
_("Unexpected character in unquoted attribute"),
|
"Unexpected character in unquoted attribute",
|
||||||
"invalid-character-after-attribute-name":
|
"invalid-character-after-attribute-name":
|
||||||
_("Unexpected character after attribute name."),
|
"Unexpected character after attribute name.",
|
||||||
"unexpected-character-after-attribute-value":
|
"unexpected-character-after-attribute-value":
|
||||||
_("Unexpected character after attribute value."),
|
"Unexpected character after attribute value.",
|
||||||
"eof-in-attribute-value-double-quote":
|
"eof-in-attribute-value-double-quote":
|
||||||
_("Unexpected end of file in attribute value (\")."),
|
"Unexpected end of file in attribute value (\").",
|
||||||
"eof-in-attribute-value-single-quote":
|
"eof-in-attribute-value-single-quote":
|
||||||
_("Unexpected end of file in attribute value (')."),
|
"Unexpected end of file in attribute value (').",
|
||||||
"eof-in-attribute-value-no-quotes":
|
"eof-in-attribute-value-no-quotes":
|
||||||
_("Unexpected end of file in attribute value."),
|
"Unexpected end of file in attribute value.",
|
||||||
"unexpected-EOF-after-solidus-in-tag":
|
"unexpected-EOF-after-solidus-in-tag":
|
||||||
_("Unexpected end of file in tag. Expected >"),
|
"Unexpected end of file in tag. Expected >",
|
||||||
"unexpected-character-after-solidus-in-tag":
|
"unexpected-character-after-solidus-in-tag":
|
||||||
_("Unexpected character after / in tag. Expected >"),
|
"Unexpected character after / in tag. Expected >",
|
||||||
"expected-dashes-or-doctype":
|
"expected-dashes-or-doctype":
|
||||||
_("Expected '--' or 'DOCTYPE'. Not found."),
|
"Expected '--' or 'DOCTYPE'. Not found.",
|
||||||
"unexpected-bang-after-double-dash-in-comment":
|
"unexpected-bang-after-double-dash-in-comment":
|
||||||
_("Unexpected ! after -- in comment"),
|
"Unexpected ! after -- in comment",
|
||||||
"unexpected-space-after-double-dash-in-comment":
|
"unexpected-space-after-double-dash-in-comment":
|
||||||
_("Unexpected space after -- in comment"),
|
"Unexpected space after -- in comment",
|
||||||
"incorrect-comment":
|
"incorrect-comment":
|
||||||
_("Incorrect comment."),
|
"Incorrect comment.",
|
||||||
"eof-in-comment":
|
"eof-in-comment":
|
||||||
_("Unexpected end of file in comment."),
|
"Unexpected end of file in comment.",
|
||||||
"eof-in-comment-end-dash":
|
"eof-in-comment-end-dash":
|
||||||
_("Unexpected end of file in comment (-)"),
|
"Unexpected end of file in comment (-)",
|
||||||
"unexpected-dash-after-double-dash-in-comment":
|
"unexpected-dash-after-double-dash-in-comment":
|
||||||
_("Unexpected '-' after '--' found in comment."),
|
"Unexpected '-' after '--' found in comment.",
|
||||||
"eof-in-comment-double-dash":
|
"eof-in-comment-double-dash":
|
||||||
_("Unexpected end of file in comment (--)."),
|
"Unexpected end of file in comment (--).",
|
||||||
"eof-in-comment-end-space-state":
|
"eof-in-comment-end-space-state":
|
||||||
_("Unexpected end of file in comment."),
|
"Unexpected end of file in comment.",
|
||||||
"eof-in-comment-end-bang-state":
|
"eof-in-comment-end-bang-state":
|
||||||
_("Unexpected end of file in comment."),
|
"Unexpected end of file in comment.",
|
||||||
"unexpected-char-in-comment":
|
"unexpected-char-in-comment":
|
||||||
_("Unexpected character in comment found."),
|
"Unexpected character in comment found.",
|
||||||
"need-space-after-doctype":
|
"need-space-after-doctype":
|
||||||
_("No space after literal string 'DOCTYPE'."),
|
"No space after literal string 'DOCTYPE'.",
|
||||||
"expected-doctype-name-but-got-right-bracket":
|
"expected-doctype-name-but-got-right-bracket":
|
||||||
_("Unexpected > character. Expected DOCTYPE name."),
|
"Unexpected > character. Expected DOCTYPE name.",
|
||||||
"expected-doctype-name-but-got-eof":
|
"expected-doctype-name-but-got-eof":
|
||||||
_("Unexpected end of file. Expected DOCTYPE name."),
|
"Unexpected end of file. Expected DOCTYPE name.",
|
||||||
"eof-in-doctype-name":
|
"eof-in-doctype-name":
|
||||||
_("Unexpected end of file in DOCTYPE name."),
|
"Unexpected end of file in DOCTYPE name.",
|
||||||
"eof-in-doctype":
|
"eof-in-doctype":
|
||||||
_("Unexpected end of file in DOCTYPE."),
|
"Unexpected end of file in DOCTYPE.",
|
||||||
"expected-space-or-right-bracket-in-doctype":
|
"expected-space-or-right-bracket-in-doctype":
|
||||||
_("Expected space or '>'. Got '%(data)s'"),
|
"Expected space or '>'. Got '%(data)s'",
|
||||||
"unexpected-end-of-doctype":
|
"unexpected-end-of-doctype":
|
||||||
_("Unexpected end of DOCTYPE."),
|
"Unexpected end of DOCTYPE.",
|
||||||
"unexpected-char-in-doctype":
|
"unexpected-char-in-doctype":
|
||||||
_("Unexpected character in DOCTYPE."),
|
"Unexpected character in DOCTYPE.",
|
||||||
"eof-in-innerhtml":
|
"eof-in-innerhtml":
|
||||||
_("XXX innerHTML EOF"),
|
"XXX innerHTML EOF",
|
||||||
"unexpected-doctype":
|
"unexpected-doctype":
|
||||||
_("Unexpected DOCTYPE. Ignored."),
|
"Unexpected DOCTYPE. Ignored.",
|
||||||
"non-html-root":
|
"non-html-root":
|
||||||
_("html needs to be the first start tag."),
|
"html needs to be the first start tag.",
|
||||||
"expected-doctype-but-got-eof":
|
"expected-doctype-but-got-eof":
|
||||||
_("Unexpected End of file. Expected DOCTYPE."),
|
"Unexpected End of file. Expected DOCTYPE.",
|
||||||
"unknown-doctype":
|
"unknown-doctype":
|
||||||
_("Erroneous DOCTYPE."),
|
"Erroneous DOCTYPE.",
|
||||||
"expected-doctype-but-got-chars":
|
"expected-doctype-but-got-chars":
|
||||||
_("Unexpected non-space characters. Expected DOCTYPE."),
|
"Unexpected non-space characters. Expected DOCTYPE.",
|
||||||
"expected-doctype-but-got-start-tag":
|
"expected-doctype-but-got-start-tag":
|
||||||
_("Unexpected start tag (%(name)s). Expected DOCTYPE."),
|
"Unexpected start tag (%(name)s). Expected DOCTYPE.",
|
||||||
"expected-doctype-but-got-end-tag":
|
"expected-doctype-but-got-end-tag":
|
||||||
_("Unexpected end tag (%(name)s). Expected DOCTYPE."),
|
"Unexpected end tag (%(name)s). Expected DOCTYPE.",
|
||||||
"end-tag-after-implied-root":
|
"end-tag-after-implied-root":
|
||||||
_("Unexpected end tag (%(name)s) after the (implied) root element."),
|
"Unexpected end tag (%(name)s) after the (implied) root element.",
|
||||||
"expected-named-closing-tag-but-got-eof":
|
"expected-named-closing-tag-but-got-eof":
|
||||||
_("Unexpected end of file. Expected end tag (%(name)s)."),
|
"Unexpected end of file. Expected end tag (%(name)s).",
|
||||||
"two-heads-are-not-better-than-one":
|
"two-heads-are-not-better-than-one":
|
||||||
_("Unexpected start tag head in existing head. Ignored."),
|
"Unexpected start tag head in existing head. Ignored.",
|
||||||
"unexpected-end-tag":
|
"unexpected-end-tag":
|
||||||
_("Unexpected end tag (%(name)s). Ignored."),
|
"Unexpected end tag (%(name)s). Ignored.",
|
||||||
"unexpected-start-tag-out-of-my-head":
|
"unexpected-start-tag-out-of-my-head":
|
||||||
_("Unexpected start tag (%(name)s) that can be in head. Moved."),
|
"Unexpected start tag (%(name)s) that can be in head. Moved.",
|
||||||
"unexpected-start-tag":
|
"unexpected-start-tag":
|
||||||
_("Unexpected start tag (%(name)s)."),
|
"Unexpected start tag (%(name)s).",
|
||||||
"missing-end-tag":
|
"missing-end-tag":
|
||||||
_("Missing end tag (%(name)s)."),
|
"Missing end tag (%(name)s).",
|
||||||
"missing-end-tags":
|
"missing-end-tags":
|
||||||
_("Missing end tags (%(name)s)."),
|
"Missing end tags (%(name)s).",
|
||||||
"unexpected-start-tag-implies-end-tag":
|
"unexpected-start-tag-implies-end-tag":
|
||||||
_("Unexpected start tag (%(startName)s) "
|
"Unexpected start tag (%(startName)s) "
|
||||||
"implies end tag (%(endName)s)."),
|
"implies end tag (%(endName)s).",
|
||||||
"unexpected-start-tag-treated-as":
|
"unexpected-start-tag-treated-as":
|
||||||
_("Unexpected start tag (%(originalName)s). Treated as %(newName)s."),
|
"Unexpected start tag (%(originalName)s). Treated as %(newName)s.",
|
||||||
"deprecated-tag":
|
"deprecated-tag":
|
||||||
_("Unexpected start tag %(name)s. Don't use it!"),
|
"Unexpected start tag %(name)s. Don't use it!",
|
||||||
"unexpected-start-tag-ignored":
|
"unexpected-start-tag-ignored":
|
||||||
_("Unexpected start tag %(name)s. Ignored."),
|
"Unexpected start tag %(name)s. Ignored.",
|
||||||
"expected-one-end-tag-but-got-another":
|
"expected-one-end-tag-but-got-another":
|
||||||
_("Unexpected end tag (%(gotName)s). "
|
"Unexpected end tag (%(gotName)s). "
|
||||||
"Missing end tag (%(expectedName)s)."),
|
"Missing end tag (%(expectedName)s).",
|
||||||
"end-tag-too-early":
|
"end-tag-too-early":
|
||||||
_("End tag (%(name)s) seen too early. Expected other end tag."),
|
"End tag (%(name)s) seen too early. Expected other end tag.",
|
||||||
"end-tag-too-early-named":
|
"end-tag-too-early-named":
|
||||||
_("Unexpected end tag (%(gotName)s). Expected end tag (%(expectedName)s)."),
|
"Unexpected end tag (%(gotName)s). Expected end tag (%(expectedName)s).",
|
||||||
"end-tag-too-early-ignored":
|
"end-tag-too-early-ignored":
|
||||||
_("End tag (%(name)s) seen too early. Ignored."),
|
"End tag (%(name)s) seen too early. Ignored.",
|
||||||
"adoption-agency-1.1":
|
"adoption-agency-1.1":
|
||||||
_("End tag (%(name)s) violates step 1, "
|
"End tag (%(name)s) violates step 1, "
|
||||||
"paragraph 1 of the adoption agency algorithm."),
|
"paragraph 1 of the adoption agency algorithm.",
|
||||||
"adoption-agency-1.2":
|
"adoption-agency-1.2":
|
||||||
_("End tag (%(name)s) violates step 1, "
|
"End tag (%(name)s) violates step 1, "
|
||||||
"paragraph 2 of the adoption agency algorithm."),
|
"paragraph 2 of the adoption agency algorithm.",
|
||||||
"adoption-agency-1.3":
|
"adoption-agency-1.3":
|
||||||
_("End tag (%(name)s) violates step 1, "
|
"End tag (%(name)s) violates step 1, "
|
||||||
"paragraph 3 of the adoption agency algorithm."),
|
"paragraph 3 of the adoption agency algorithm.",
|
||||||
"adoption-agency-4.4":
|
"adoption-agency-4.4":
|
||||||
_("End tag (%(name)s) violates step 4, "
|
"End tag (%(name)s) violates step 4, "
|
||||||
"paragraph 4 of the adoption agency algorithm."),
|
"paragraph 4 of the adoption agency algorithm.",
|
||||||
"unexpected-end-tag-treated-as":
|
"unexpected-end-tag-treated-as":
|
||||||
_("Unexpected end tag (%(originalName)s). Treated as %(newName)s."),
|
"Unexpected end tag (%(originalName)s). Treated as %(newName)s.",
|
||||||
"no-end-tag":
|
"no-end-tag":
|
||||||
_("This element (%(name)s) has no end tag."),
|
"This element (%(name)s) has no end tag.",
|
||||||
"unexpected-implied-end-tag-in-table":
|
"unexpected-implied-end-tag-in-table":
|
||||||
_("Unexpected implied end tag (%(name)s) in the table phase."),
|
"Unexpected implied end tag (%(name)s) in the table phase.",
|
||||||
"unexpected-implied-end-tag-in-table-body":
|
"unexpected-implied-end-tag-in-table-body":
|
||||||
_("Unexpected implied end tag (%(name)s) in the table body phase."),
|
"Unexpected implied end tag (%(name)s) in the table body phase.",
|
||||||
"unexpected-char-implies-table-voodoo":
|
"unexpected-char-implies-table-voodoo":
|
||||||
_("Unexpected non-space characters in "
|
"Unexpected non-space characters in "
|
||||||
"table context caused voodoo mode."),
|
"table context caused voodoo mode.",
|
||||||
"unexpected-hidden-input-in-table":
|
"unexpected-hidden-input-in-table":
|
||||||
_("Unexpected input with type hidden in table context."),
|
"Unexpected input with type hidden in table context.",
|
||||||
"unexpected-form-in-table":
|
"unexpected-form-in-table":
|
||||||
_("Unexpected form in table context."),
|
"Unexpected form in table context.",
|
||||||
"unexpected-start-tag-implies-table-voodoo":
|
"unexpected-start-tag-implies-table-voodoo":
|
||||||
_("Unexpected start tag (%(name)s) in "
|
"Unexpected start tag (%(name)s) in "
|
||||||
"table context caused voodoo mode."),
|
"table context caused voodoo mode.",
|
||||||
"unexpected-end-tag-implies-table-voodoo":
|
"unexpected-end-tag-implies-table-voodoo":
|
||||||
_("Unexpected end tag (%(name)s) in "
|
"Unexpected end tag (%(name)s) in "
|
||||||
"table context caused voodoo mode."),
|
"table context caused voodoo mode.",
|
||||||
"unexpected-cell-in-table-body":
|
"unexpected-cell-in-table-body":
|
||||||
_("Unexpected table cell start tag (%(name)s) "
|
"Unexpected table cell start tag (%(name)s) "
|
||||||
"in the table body phase."),
|
"in the table body phase.",
|
||||||
"unexpected-cell-end-tag":
|
"unexpected-cell-end-tag":
|
||||||
_("Got table cell end tag (%(name)s) "
|
"Got table cell end tag (%(name)s) "
|
||||||
"while required end tags are missing."),
|
"while required end tags are missing.",
|
||||||
"unexpected-end-tag-in-table-body":
|
"unexpected-end-tag-in-table-body":
|
||||||
_("Unexpected end tag (%(name)s) in the table body phase. Ignored."),
|
"Unexpected end tag (%(name)s) in the table body phase. Ignored.",
|
||||||
"unexpected-implied-end-tag-in-table-row":
|
"unexpected-implied-end-tag-in-table-row":
|
||||||
_("Unexpected implied end tag (%(name)s) in the table row phase."),
|
"Unexpected implied end tag (%(name)s) in the table row phase.",
|
||||||
"unexpected-end-tag-in-table-row":
|
"unexpected-end-tag-in-table-row":
|
||||||
_("Unexpected end tag (%(name)s) in the table row phase. Ignored."),
|
"Unexpected end tag (%(name)s) in the table row phase. Ignored.",
|
||||||
"unexpected-select-in-select":
|
"unexpected-select-in-select":
|
||||||
_("Unexpected select start tag in the select phase "
|
"Unexpected select start tag in the select phase "
|
||||||
"treated as select end tag."),
|
"treated as select end tag.",
|
||||||
"unexpected-input-in-select":
|
"unexpected-input-in-select":
|
||||||
_("Unexpected input start tag in the select phase."),
|
"Unexpected input start tag in the select phase.",
|
||||||
"unexpected-start-tag-in-select":
|
"unexpected-start-tag-in-select":
|
||||||
_("Unexpected start tag token (%(name)s in the select phase. "
|
"Unexpected start tag token (%(name)s in the select phase. "
|
||||||
"Ignored."),
|
"Ignored.",
|
||||||
"unexpected-end-tag-in-select":
|
"unexpected-end-tag-in-select":
|
||||||
_("Unexpected end tag (%(name)s) in the select phase. Ignored."),
|
"Unexpected end tag (%(name)s) in the select phase. Ignored.",
|
||||||
"unexpected-table-element-start-tag-in-select-in-table":
|
"unexpected-table-element-start-tag-in-select-in-table":
|
||||||
_("Unexpected table element start tag (%(name)s) in the select in table phase."),
|
"Unexpected table element start tag (%(name)s) in the select in table phase.",
|
||||||
"unexpected-table-element-end-tag-in-select-in-table":
|
"unexpected-table-element-end-tag-in-select-in-table":
|
||||||
_("Unexpected table element end tag (%(name)s) in the select in table phase."),
|
"Unexpected table element end tag (%(name)s) in the select in table phase.",
|
||||||
"unexpected-char-after-body":
|
"unexpected-char-after-body":
|
||||||
_("Unexpected non-space characters in the after body phase."),
|
"Unexpected non-space characters in the after body phase.",
|
||||||
"unexpected-start-tag-after-body":
|
"unexpected-start-tag-after-body":
|
||||||
_("Unexpected start tag token (%(name)s)"
|
"Unexpected start tag token (%(name)s)"
|
||||||
" in the after body phase."),
|
" in the after body phase.",
|
||||||
"unexpected-end-tag-after-body":
|
"unexpected-end-tag-after-body":
|
||||||
_("Unexpected end tag token (%(name)s)"
|
"Unexpected end tag token (%(name)s)"
|
||||||
" in the after body phase."),
|
" in the after body phase.",
|
||||||
"unexpected-char-in-frameset":
|
"unexpected-char-in-frameset":
|
||||||
_("Unexpected characters in the frameset phase. Characters ignored."),
|
"Unexpected characters in the frameset phase. Characters ignored.",
|
||||||
"unexpected-start-tag-in-frameset":
|
"unexpected-start-tag-in-frameset":
|
||||||
_("Unexpected start tag token (%(name)s)"
|
"Unexpected start tag token (%(name)s)"
|
||||||
" in the frameset phase. Ignored."),
|
" in the frameset phase. Ignored.",
|
||||||
"unexpected-frameset-in-frameset-innerhtml":
|
"unexpected-frameset-in-frameset-innerhtml":
|
||||||
_("Unexpected end tag token (frameset) "
|
"Unexpected end tag token (frameset) "
|
||||||
"in the frameset phase (innerHTML)."),
|
"in the frameset phase (innerHTML).",
|
||||||
"unexpected-end-tag-in-frameset":
|
"unexpected-end-tag-in-frameset":
|
||||||
_("Unexpected end tag token (%(name)s)"
|
"Unexpected end tag token (%(name)s)"
|
||||||
" in the frameset phase. Ignored."),
|
" in the frameset phase. Ignored.",
|
||||||
"unexpected-char-after-frameset":
|
"unexpected-char-after-frameset":
|
||||||
_("Unexpected non-space characters in the "
|
"Unexpected non-space characters in the "
|
||||||
"after frameset phase. Ignored."),
|
"after frameset phase. Ignored.",
|
||||||
"unexpected-start-tag-after-frameset":
|
"unexpected-start-tag-after-frameset":
|
||||||
_("Unexpected start tag (%(name)s)"
|
"Unexpected start tag (%(name)s)"
|
||||||
" in the after frameset phase. Ignored."),
|
" in the after frameset phase. Ignored.",
|
||||||
"unexpected-end-tag-after-frameset":
|
"unexpected-end-tag-after-frameset":
|
||||||
_("Unexpected end tag (%(name)s)"
|
"Unexpected end tag (%(name)s)"
|
||||||
" in the after frameset phase. Ignored."),
|
" in the after frameset phase. Ignored.",
|
||||||
"unexpected-end-tag-after-body-innerhtml":
|
"unexpected-end-tag-after-body-innerhtml":
|
||||||
_("Unexpected end tag after body(innerHtml)"),
|
"Unexpected end tag after body(innerHtml)",
|
||||||
"expected-eof-but-got-char":
|
"expected-eof-but-got-char":
|
||||||
_("Unexpected non-space characters. Expected end of file."),
|
"Unexpected non-space characters. Expected end of file.",
|
||||||
"expected-eof-but-got-start-tag":
|
"expected-eof-but-got-start-tag":
|
||||||
_("Unexpected start tag (%(name)s)"
|
"Unexpected start tag (%(name)s)"
|
||||||
". Expected end of file."),
|
". Expected end of file.",
|
||||||
"expected-eof-but-got-end-tag":
|
"expected-eof-but-got-end-tag":
|
||||||
_("Unexpected end tag (%(name)s)"
|
"Unexpected end tag (%(name)s)"
|
||||||
". Expected end of file."),
|
". Expected end of file.",
|
||||||
"eof-in-table":
|
"eof-in-table":
|
||||||
_("Unexpected end of file. Expected table content."),
|
"Unexpected end of file. Expected table content.",
|
||||||
"eof-in-select":
|
"eof-in-select":
|
||||||
_("Unexpected end of file. Expected select content."),
|
"Unexpected end of file. Expected select content.",
|
||||||
"eof-in-frameset":
|
"eof-in-frameset":
|
||||||
_("Unexpected end of file. Expected frameset content."),
|
"Unexpected end of file. Expected frameset content.",
|
||||||
"eof-in-script-in-script":
|
"eof-in-script-in-script":
|
||||||
_("Unexpected end of file. Expected script content."),
|
"Unexpected end of file. Expected script content.",
|
||||||
"eof-in-foreign-lands":
|
"eof-in-foreign-lands":
|
||||||
_("Unexpected end of file. Expected foreign content"),
|
"Unexpected end of file. Expected foreign content",
|
||||||
"non-void-element-with-trailing-solidus":
|
"non-void-element-with-trailing-solidus":
|
||||||
_("Trailing solidus not allowed on element %(name)s"),
|
"Trailing solidus not allowed on element %(name)s",
|
||||||
"unexpected-html-element-in-foreign-content":
|
"unexpected-html-element-in-foreign-content":
|
||||||
_("Element %(name)s not allowed in a non-html context"),
|
"Element %(name)s not allowed in a non-html context",
|
||||||
"unexpected-end-tag-before-html":
|
"unexpected-end-tag-before-html":
|
||||||
_("Unexpected end tag (%(name)s) before html."),
|
"Unexpected end tag (%(name)s) before html.",
|
||||||
"XXX-undefined-error":
|
"XXX-undefined-error":
|
||||||
_("Undefined error (this sucks and should be fixed)"),
|
"Undefined error (this sucks and should be fixed)",
|
||||||
}
|
}
|
||||||
|
|
||||||
namespaces = {
|
namespaces = {
|
||||||
@@ -298,7 +296,7 @@ namespaces = {
|
|||||||
"xmlns": "http://www.w3.org/2000/xmlns/"
|
"xmlns": "http://www.w3.org/2000/xmlns/"
|
||||||
}
|
}
|
||||||
|
|
||||||
scopingElements = frozenset((
|
scopingElements = frozenset([
|
||||||
(namespaces["html"], "applet"),
|
(namespaces["html"], "applet"),
|
||||||
(namespaces["html"], "caption"),
|
(namespaces["html"], "caption"),
|
||||||
(namespaces["html"], "html"),
|
(namespaces["html"], "html"),
|
||||||
@@ -316,9 +314,9 @@ scopingElements = frozenset((
|
|||||||
(namespaces["svg"], "foreignObject"),
|
(namespaces["svg"], "foreignObject"),
|
||||||
(namespaces["svg"], "desc"),
|
(namespaces["svg"], "desc"),
|
||||||
(namespaces["svg"], "title"),
|
(namespaces["svg"], "title"),
|
||||||
))
|
])
|
||||||
|
|
||||||
formattingElements = frozenset((
|
formattingElements = frozenset([
|
||||||
(namespaces["html"], "a"),
|
(namespaces["html"], "a"),
|
||||||
(namespaces["html"], "b"),
|
(namespaces["html"], "b"),
|
||||||
(namespaces["html"], "big"),
|
(namespaces["html"], "big"),
|
||||||
@@ -333,9 +331,9 @@ formattingElements = frozenset((
|
|||||||
(namespaces["html"], "strong"),
|
(namespaces["html"], "strong"),
|
||||||
(namespaces["html"], "tt"),
|
(namespaces["html"], "tt"),
|
||||||
(namespaces["html"], "u")
|
(namespaces["html"], "u")
|
||||||
))
|
])
|
||||||
|
|
||||||
specialElements = frozenset((
|
specialElements = frozenset([
|
||||||
(namespaces["html"], "address"),
|
(namespaces["html"], "address"),
|
||||||
(namespaces["html"], "applet"),
|
(namespaces["html"], "applet"),
|
||||||
(namespaces["html"], "area"),
|
(namespaces["html"], "area"),
|
||||||
@@ -416,22 +414,22 @@ specialElements = frozenset((
|
|||||||
(namespaces["html"], "wbr"),
|
(namespaces["html"], "wbr"),
|
||||||
(namespaces["html"], "xmp"),
|
(namespaces["html"], "xmp"),
|
||||||
(namespaces["svg"], "foreignObject")
|
(namespaces["svg"], "foreignObject")
|
||||||
))
|
])
|
||||||
|
|
||||||
htmlIntegrationPointElements = frozenset((
|
htmlIntegrationPointElements = frozenset([
|
||||||
(namespaces["mathml"], "annotaion-xml"),
|
(namespaces["mathml"], "annotaion-xml"),
|
||||||
(namespaces["svg"], "foreignObject"),
|
(namespaces["svg"], "foreignObject"),
|
||||||
(namespaces["svg"], "desc"),
|
(namespaces["svg"], "desc"),
|
||||||
(namespaces["svg"], "title")
|
(namespaces["svg"], "title")
|
||||||
))
|
])
|
||||||
|
|
||||||
mathmlTextIntegrationPointElements = frozenset((
|
mathmlTextIntegrationPointElements = frozenset([
|
||||||
(namespaces["mathml"], "mi"),
|
(namespaces["mathml"], "mi"),
|
||||||
(namespaces["mathml"], "mo"),
|
(namespaces["mathml"], "mo"),
|
||||||
(namespaces["mathml"], "mn"),
|
(namespaces["mathml"], "mn"),
|
||||||
(namespaces["mathml"], "ms"),
|
(namespaces["mathml"], "ms"),
|
||||||
(namespaces["mathml"], "mtext")
|
(namespaces["mathml"], "mtext")
|
||||||
))
|
])
|
||||||
|
|
||||||
adjustForeignAttributes = {
|
adjustForeignAttributes = {
|
||||||
"xlink:actuate": ("xlink", "actuate", namespaces["xlink"]),
|
"xlink:actuate": ("xlink", "actuate", namespaces["xlink"]),
|
||||||
@@ -451,21 +449,21 @@ adjustForeignAttributes = {
|
|||||||
unadjustForeignAttributes = dict([((ns, local), qname) for qname, (prefix, local, ns) in
|
unadjustForeignAttributes = dict([((ns, local), qname) for qname, (prefix, local, ns) in
|
||||||
adjustForeignAttributes.items()])
|
adjustForeignAttributes.items()])
|
||||||
|
|
||||||
spaceCharacters = frozenset((
|
spaceCharacters = frozenset([
|
||||||
"\t",
|
"\t",
|
||||||
"\n",
|
"\n",
|
||||||
"\u000C",
|
"\u000C",
|
||||||
" ",
|
" ",
|
||||||
"\r"
|
"\r"
|
||||||
))
|
])
|
||||||
|
|
||||||
tableInsertModeElements = frozenset((
|
tableInsertModeElements = frozenset([
|
||||||
"table",
|
"table",
|
||||||
"tbody",
|
"tbody",
|
||||||
"tfoot",
|
"tfoot",
|
||||||
"thead",
|
"thead",
|
||||||
"tr"
|
"tr"
|
||||||
))
|
])
|
||||||
|
|
||||||
asciiLowercase = frozenset(string.ascii_lowercase)
|
asciiLowercase = frozenset(string.ascii_lowercase)
|
||||||
asciiUppercase = frozenset(string.ascii_uppercase)
|
asciiUppercase = frozenset(string.ascii_uppercase)
|
||||||
@@ -486,7 +484,7 @@ headingElements = (
|
|||||||
"h6"
|
"h6"
|
||||||
)
|
)
|
||||||
|
|
||||||
voidElements = frozenset((
|
voidElements = frozenset([
|
||||||
"base",
|
"base",
|
||||||
"command",
|
"command",
|
||||||
"event-source",
|
"event-source",
|
||||||
@@ -502,11 +500,11 @@ voidElements = frozenset((
|
|||||||
"input",
|
"input",
|
||||||
"source",
|
"source",
|
||||||
"track"
|
"track"
|
||||||
))
|
])
|
||||||
|
|
||||||
cdataElements = frozenset(('title', 'textarea'))
|
cdataElements = frozenset(['title', 'textarea'])
|
||||||
|
|
||||||
rcdataElements = frozenset((
|
rcdataElements = frozenset([
|
||||||
'style',
|
'style',
|
||||||
'script',
|
'script',
|
||||||
'xmp',
|
'xmp',
|
||||||
@@ -514,27 +512,27 @@ rcdataElements = frozenset((
|
|||||||
'noembed',
|
'noembed',
|
||||||
'noframes',
|
'noframes',
|
||||||
'noscript'
|
'noscript'
|
||||||
))
|
])
|
||||||
|
|
||||||
booleanAttributes = {
|
booleanAttributes = {
|
||||||
"": frozenset(("irrelevant",)),
|
"": frozenset(["irrelevant"]),
|
||||||
"style": frozenset(("scoped",)),
|
"style": frozenset(["scoped"]),
|
||||||
"img": frozenset(("ismap",)),
|
"img": frozenset(["ismap"]),
|
||||||
"audio": frozenset(("autoplay", "controls")),
|
"audio": frozenset(["autoplay", "controls"]),
|
||||||
"video": frozenset(("autoplay", "controls")),
|
"video": frozenset(["autoplay", "controls"]),
|
||||||
"script": frozenset(("defer", "async")),
|
"script": frozenset(["defer", "async"]),
|
||||||
"details": frozenset(("open",)),
|
"details": frozenset(["open"]),
|
||||||
"datagrid": frozenset(("multiple", "disabled")),
|
"datagrid": frozenset(["multiple", "disabled"]),
|
||||||
"command": frozenset(("hidden", "disabled", "checked", "default")),
|
"command": frozenset(["hidden", "disabled", "checked", "default"]),
|
||||||
"hr": frozenset(("noshade")),
|
"hr": frozenset(["noshade"]),
|
||||||
"menu": frozenset(("autosubmit",)),
|
"menu": frozenset(["autosubmit"]),
|
||||||
"fieldset": frozenset(("disabled", "readonly")),
|
"fieldset": frozenset(["disabled", "readonly"]),
|
||||||
"option": frozenset(("disabled", "readonly", "selected")),
|
"option": frozenset(["disabled", "readonly", "selected"]),
|
||||||
"optgroup": frozenset(("disabled", "readonly")),
|
"optgroup": frozenset(["disabled", "readonly"]),
|
||||||
"button": frozenset(("disabled", "autofocus")),
|
"button": frozenset(["disabled", "autofocus"]),
|
||||||
"input": frozenset(("disabled", "readonly", "required", "autofocus", "checked", "ismap")),
|
"input": frozenset(["disabled", "readonly", "required", "autofocus", "checked", "ismap"]),
|
||||||
"select": frozenset(("disabled", "readonly", "autofocus", "multiple")),
|
"select": frozenset(["disabled", "readonly", "autofocus", "multiple"]),
|
||||||
"output": frozenset(("disabled", "readonly")),
|
"output": frozenset(["disabled", "readonly"]),
|
||||||
}
|
}
|
||||||
|
|
||||||
# entitiesWindows1252 has to be _ordered_ and needs to have an index. It
|
# entitiesWindows1252 has to be _ordered_ and needs to have an index. It
|
||||||
@@ -574,7 +572,7 @@ entitiesWindows1252 = (
|
|||||||
376 # 0x9F 0x0178 LATIN CAPITAL LETTER Y WITH DIAERESIS
|
376 # 0x9F 0x0178 LATIN CAPITAL LETTER Y WITH DIAERESIS
|
||||||
)
|
)
|
||||||
|
|
||||||
xmlEntities = frozenset(('lt;', 'gt;', 'amp;', 'apos;', 'quot;'))
|
xmlEntities = frozenset(['lt;', 'gt;', 'amp;', 'apos;', 'quot;'])
|
||||||
|
|
||||||
entities = {
|
entities = {
|
||||||
"AElig": "\xc6",
|
"AElig": "\xc6",
|
||||||
@@ -3088,8 +3086,8 @@ tokenTypes = {
|
|||||||
"ParseError": 7
|
"ParseError": 7
|
||||||
}
|
}
|
||||||
|
|
||||||
tagTokenTypes = frozenset((tokenTypes["StartTag"], tokenTypes["EndTag"],
|
tagTokenTypes = frozenset([tokenTypes["StartTag"], tokenTypes["EndTag"],
|
||||||
tokenTypes["EmptyTag"]))
|
tokenTypes["EmptyTag"]])
|
||||||
|
|
||||||
|
|
||||||
prefixes = dict([(v, k) for k, v in namespaces.items()])
|
prefixes = dict([(v, k) for k, v in namespaces.items()])
|
||||||
|
|||||||
@@ -1,8 +1,5 @@
|
|||||||
from __future__ import absolute_import, division, unicode_literals
|
from __future__ import absolute_import, division, unicode_literals
|
||||||
|
|
||||||
from gettext import gettext
|
|
||||||
_ = gettext
|
|
||||||
|
|
||||||
from . import _base
|
from . import _base
|
||||||
from ..constants import cdataElements, rcdataElements, voidElements
|
from ..constants import cdataElements, rcdataElements, voidElements
|
||||||
|
|
||||||
@@ -23,24 +20,24 @@ class Filter(_base.Filter):
|
|||||||
if type in ("StartTag", "EmptyTag"):
|
if type in ("StartTag", "EmptyTag"):
|
||||||
name = token["name"]
|
name = token["name"]
|
||||||
if contentModelFlag != "PCDATA":
|
if contentModelFlag != "PCDATA":
|
||||||
raise LintError(_("StartTag not in PCDATA content model flag: %(tag)s") % {"tag": name})
|
raise LintError("StartTag not in PCDATA content model flag: %(tag)s" % {"tag": name})
|
||||||
if not isinstance(name, str):
|
if not isinstance(name, str):
|
||||||
raise LintError(_("Tag name is not a string: %(tag)r") % {"tag": name})
|
raise LintError("Tag name is not a string: %(tag)r" % {"tag": name})
|
||||||
if not name:
|
if not name:
|
||||||
raise LintError(_("Empty tag name"))
|
raise LintError("Empty tag name")
|
||||||
if type == "StartTag" and name in voidElements:
|
if type == "StartTag" and name in voidElements:
|
||||||
raise LintError(_("Void element reported as StartTag token: %(tag)s") % {"tag": name})
|
raise LintError("Void element reported as StartTag token: %(tag)s" % {"tag": name})
|
||||||
elif type == "EmptyTag" and name not in voidElements:
|
elif type == "EmptyTag" and name not in voidElements:
|
||||||
raise LintError(_("Non-void element reported as EmptyTag token: %(tag)s") % {"tag": token["name"]})
|
raise LintError("Non-void element reported as EmptyTag token: %(tag)s" % {"tag": token["name"]})
|
||||||
if type == "StartTag":
|
if type == "StartTag":
|
||||||
open_elements.append(name)
|
open_elements.append(name)
|
||||||
for name, value in token["data"]:
|
for name, value in token["data"]:
|
||||||
if not isinstance(name, str):
|
if not isinstance(name, str):
|
||||||
raise LintError(_("Attribute name is not a string: %(name)r") % {"name": name})
|
raise LintError("Attribute name is not a string: %(name)r" % {"name": name})
|
||||||
if not name:
|
if not name:
|
||||||
raise LintError(_("Empty attribute name"))
|
raise LintError("Empty attribute name")
|
||||||
if not isinstance(value, str):
|
if not isinstance(value, str):
|
||||||
raise LintError(_("Attribute value is not a string: %(value)r") % {"value": value})
|
raise LintError("Attribute value is not a string: %(value)r" % {"value": value})
|
||||||
if name in cdataElements:
|
if name in cdataElements:
|
||||||
contentModelFlag = "CDATA"
|
contentModelFlag = "CDATA"
|
||||||
elif name in rcdataElements:
|
elif name in rcdataElements:
|
||||||
@@ -51,43 +48,43 @@ class Filter(_base.Filter):
|
|||||||
elif type == "EndTag":
|
elif type == "EndTag":
|
||||||
name = token["name"]
|
name = token["name"]
|
||||||
if not isinstance(name, str):
|
if not isinstance(name, str):
|
||||||
raise LintError(_("Tag name is not a string: %(tag)r") % {"tag": name})
|
raise LintError("Tag name is not a string: %(tag)r" % {"tag": name})
|
||||||
if not name:
|
if not name:
|
||||||
raise LintError(_("Empty tag name"))
|
raise LintError("Empty tag name")
|
||||||
if name in voidElements:
|
if name in voidElements:
|
||||||
raise LintError(_("Void element reported as EndTag token: %(tag)s") % {"tag": name})
|
raise LintError("Void element reported as EndTag token: %(tag)s" % {"tag": name})
|
||||||
start_name = open_elements.pop()
|
start_name = open_elements.pop()
|
||||||
if start_name != name:
|
if start_name != name:
|
||||||
raise LintError(_("EndTag (%(end)s) does not match StartTag (%(start)s)") % {"end": name, "start": start_name})
|
raise LintError("EndTag (%(end)s) does not match StartTag (%(start)s)" % {"end": name, "start": start_name})
|
||||||
contentModelFlag = "PCDATA"
|
contentModelFlag = "PCDATA"
|
||||||
|
|
||||||
elif type == "Comment":
|
elif type == "Comment":
|
||||||
if contentModelFlag != "PCDATA":
|
if contentModelFlag != "PCDATA":
|
||||||
raise LintError(_("Comment not in PCDATA content model flag"))
|
raise LintError("Comment not in PCDATA content model flag")
|
||||||
|
|
||||||
elif type in ("Characters", "SpaceCharacters"):
|
elif type in ("Characters", "SpaceCharacters"):
|
||||||
data = token["data"]
|
data = token["data"]
|
||||||
if not isinstance(data, str):
|
if not isinstance(data, str):
|
||||||
raise LintError(_("Attribute name is not a string: %(name)r") % {"name": data})
|
raise LintError("Attribute name is not a string: %(name)r" % {"name": data})
|
||||||
if not data:
|
if not data:
|
||||||
raise LintError(_("%(type)s token with empty data") % {"type": type})
|
raise LintError("%(type)s token with empty data" % {"type": type})
|
||||||
if type == "SpaceCharacters":
|
if type == "SpaceCharacters":
|
||||||
data = data.strip(spaceCharacters)
|
data = data.strip(spaceCharacters)
|
||||||
if data:
|
if data:
|
||||||
raise LintError(_("Non-space character(s) found in SpaceCharacters token: %(token)r") % {"token": data})
|
raise LintError("Non-space character(s) found in SpaceCharacters token: %(token)r" % {"token": data})
|
||||||
|
|
||||||
elif type == "Doctype":
|
elif type == "Doctype":
|
||||||
name = token["name"]
|
name = token["name"]
|
||||||
if contentModelFlag != "PCDATA":
|
if contentModelFlag != "PCDATA":
|
||||||
raise LintError(_("Doctype not in PCDATA content model flag: %(name)s") % {"name": name})
|
raise LintError("Doctype not in PCDATA content model flag: %(name)s" % {"name": name})
|
||||||
if not isinstance(name, str):
|
if not isinstance(name, str):
|
||||||
raise LintError(_("Tag name is not a string: %(tag)r") % {"tag": name})
|
raise LintError("Tag name is not a string: %(tag)r" % {"tag": name})
|
||||||
# XXX: what to do with token["data"] ?
|
# XXX: what to do with token["data"] ?
|
||||||
|
|
||||||
elif type in ("ParseError", "SerializeError"):
|
elif type in ("ParseError", "SerializeError"):
|
||||||
pass
|
pass
|
||||||
|
|
||||||
else:
|
else:
|
||||||
raise LintError(_("Unknown token type: %(type)s") % {"type": type})
|
raise LintError("Unknown token type: %(type)s" % {"type": type})
|
||||||
|
|
||||||
yield token
|
yield token
|
||||||
|
|||||||
@@ -18,6 +18,7 @@ from .constants import cdataElements, rcdataElements
|
|||||||
from .constants import tokenTypes, ReparseException, namespaces
|
from .constants import tokenTypes, ReparseException, namespaces
|
||||||
from .constants import htmlIntegrationPointElements, mathmlTextIntegrationPointElements
|
from .constants import htmlIntegrationPointElements, mathmlTextIntegrationPointElements
|
||||||
from .constants import adjustForeignAttributes as adjustForeignAttributesMap
|
from .constants import adjustForeignAttributes as adjustForeignAttributesMap
|
||||||
|
from .constants import E
|
||||||
|
|
||||||
|
|
||||||
def parse(doc, treebuilder="etree", encoding=None,
|
def parse(doc, treebuilder="etree", encoding=None,
|
||||||
@@ -129,6 +130,17 @@ class HTMLParser(object):
|
|||||||
|
|
||||||
self.framesetOK = True
|
self.framesetOK = True
|
||||||
|
|
||||||
|
@property
|
||||||
|
def documentEncoding(self):
|
||||||
|
"""The name of the character encoding
|
||||||
|
that was used to decode the input stream,
|
||||||
|
or :obj:`None` if that is not determined yet.
|
||||||
|
|
||||||
|
"""
|
||||||
|
if not hasattr(self, 'tokenizer'):
|
||||||
|
return None
|
||||||
|
return self.tokenizer.stream.charEncoding[0]
|
||||||
|
|
||||||
def isHTMLIntegrationPoint(self, element):
|
def isHTMLIntegrationPoint(self, element):
|
||||||
if (element.name == "annotation-xml" and
|
if (element.name == "annotation-xml" and
|
||||||
element.namespace == namespaces["mathml"]):
|
element.namespace == namespaces["mathml"]):
|
||||||
@@ -245,7 +257,7 @@ class HTMLParser(object):
|
|||||||
# XXX The idea is to make errorcode mandatory.
|
# XXX The idea is to make errorcode mandatory.
|
||||||
self.errors.append((self.tokenizer.stream.position(), errorcode, datavars))
|
self.errors.append((self.tokenizer.stream.position(), errorcode, datavars))
|
||||||
if self.strict:
|
if self.strict:
|
||||||
raise ParseError
|
raise ParseError(E[errorcode] % datavars)
|
||||||
|
|
||||||
def normalizeToken(self, token):
|
def normalizeToken(self, token):
|
||||||
""" HTML5 specific normalizations to the token stream """
|
""" HTML5 specific normalizations to the token stream """
|
||||||
@@ -868,7 +880,7 @@ def getPhases(debug):
|
|||||||
self.startTagHandler = utils.MethodDispatcher([
|
self.startTagHandler = utils.MethodDispatcher([
|
||||||
("html", self.startTagHtml),
|
("html", self.startTagHtml),
|
||||||
(("base", "basefont", "bgsound", "command", "link", "meta",
|
(("base", "basefont", "bgsound", "command", "link", "meta",
|
||||||
"noframes", "script", "style", "title"),
|
"script", "style", "title"),
|
||||||
self.startTagProcessInHead),
|
self.startTagProcessInHead),
|
||||||
("body", self.startTagBody),
|
("body", self.startTagBody),
|
||||||
("frameset", self.startTagFrameset),
|
("frameset", self.startTagFrameset),
|
||||||
@@ -1205,8 +1217,7 @@ def getPhases(debug):
|
|||||||
attributes["name"] = "isindex"
|
attributes["name"] = "isindex"
|
||||||
self.processStartTag(impliedTagToken("input", "StartTag",
|
self.processStartTag(impliedTagToken("input", "StartTag",
|
||||||
attributes=attributes,
|
attributes=attributes,
|
||||||
selfClosing=
|
selfClosing=token["selfClosing"]))
|
||||||
token["selfClosing"]))
|
|
||||||
self.processEndTag(impliedTagToken("label"))
|
self.processEndTag(impliedTagToken("label"))
|
||||||
self.processStartTag(impliedTagToken("hr", "StartTag"))
|
self.processStartTag(impliedTagToken("hr", "StartTag"))
|
||||||
self.processEndTag(impliedTagToken("form"))
|
self.processEndTag(impliedTagToken("form"))
|
||||||
|
|||||||
@@ -28,7 +28,18 @@ asciiLettersBytes = frozenset([item.encode("ascii") for item in asciiLetters])
|
|||||||
asciiUppercaseBytes = frozenset([item.encode("ascii") for item in asciiUppercase])
|
asciiUppercaseBytes = frozenset([item.encode("ascii") for item in asciiUppercase])
|
||||||
spacesAngleBrackets = spaceCharactersBytes | frozenset([b">", b"<"])
|
spacesAngleBrackets = spaceCharactersBytes | frozenset([b">", b"<"])
|
||||||
|
|
||||||
invalid_unicode_re = re.compile("[\u0001-\u0008\u000B\u000E-\u001F\u007F-\u009F\uD800-\uDFFF\uFDD0-\uFDEF\uFFFE\uFFFF\U0001FFFE\U0001FFFF\U0002FFFE\U0002FFFF\U0003FFFE\U0003FFFF\U0004FFFE\U0004FFFF\U0005FFFE\U0005FFFF\U0006FFFE\U0006FFFF\U0007FFFE\U0007FFFF\U0008FFFE\U0008FFFF\U0009FFFE\U0009FFFF\U000AFFFE\U000AFFFF\U000BFFFE\U000BFFFF\U000CFFFE\U000CFFFF\U000DFFFE\U000DFFFF\U000EFFFE\U000EFFFF\U000FFFFE\U000FFFFF\U0010FFFE\U0010FFFF]")
|
|
||||||
|
invalid_unicode_no_surrogate = "[\u0001-\u0008\u000B\u000E-\u001F\u007F-\u009F\uFDD0-\uFDEF\uFFFE\uFFFF\U0001FFFE\U0001FFFF\U0002FFFE\U0002FFFF\U0003FFFE\U0003FFFF\U0004FFFE\U0004FFFF\U0005FFFE\U0005FFFF\U0006FFFE\U0006FFFF\U0007FFFE\U0007FFFF\U0008FFFE\U0008FFFF\U0009FFFE\U0009FFFF\U000AFFFE\U000AFFFF\U000BFFFE\U000BFFFF\U000CFFFE\U000CFFFF\U000DFFFE\U000DFFFF\U000EFFFE\U000EFFFF\U000FFFFE\U000FFFFF\U0010FFFE\U0010FFFF]"
|
||||||
|
|
||||||
|
if utils.supports_lone_surrogates:
|
||||||
|
# Use one extra step of indirection and create surrogates with
|
||||||
|
# unichr. Not using this indirection would introduce an illegal
|
||||||
|
# unicode literal on platforms not supporting such lone
|
||||||
|
# surrogates.
|
||||||
|
invalid_unicode_re = re.compile(invalid_unicode_no_surrogate +
|
||||||
|
eval('"\\uD800-\\uDFFF"'))
|
||||||
|
else:
|
||||||
|
invalid_unicode_re = re.compile(invalid_unicode_no_surrogate)
|
||||||
|
|
||||||
non_bmp_invalid_codepoints = set([0x1FFFE, 0x1FFFF, 0x2FFFE, 0x2FFFF, 0x3FFFE,
|
non_bmp_invalid_codepoints = set([0x1FFFE, 0x1FFFF, 0x2FFFE, 0x2FFFF, 0x3FFFE,
|
||||||
0x3FFFF, 0x4FFFE, 0x4FFFF, 0x5FFFE, 0x5FFFF,
|
0x3FFFF, 0x4FFFE, 0x4FFFF, 0x5FFFE, 0x5FFFF,
|
||||||
@@ -164,13 +175,18 @@ class HTMLUnicodeInputStream(object):
|
|||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
# Craziness
|
if not utils.supports_lone_surrogates:
|
||||||
if len("\U0010FFFF") == 1:
|
# Such platforms will have already checked for such
|
||||||
|
# surrogate errors, so no need to do this checking.
|
||||||
|
self.reportCharacterErrors = None
|
||||||
|
self.replaceCharactersRegexp = None
|
||||||
|
elif len("\U0010FFFF") == 1:
|
||||||
self.reportCharacterErrors = self.characterErrorsUCS4
|
self.reportCharacterErrors = self.characterErrorsUCS4
|
||||||
self.replaceCharactersRegexp = re.compile("[\uD800-\uDFFF]")
|
self.replaceCharactersRegexp = re.compile(eval('"[\\uD800-\\uDFFF]"'))
|
||||||
else:
|
else:
|
||||||
self.reportCharacterErrors = self.characterErrorsUCS2
|
self.reportCharacterErrors = self.characterErrorsUCS2
|
||||||
self.replaceCharactersRegexp = re.compile("([\uD800-\uDBFF](?![\uDC00-\uDFFF])|(?<![\uD800-\uDBFF])[\uDC00-\uDFFF])")
|
self.replaceCharactersRegexp = re.compile(
|
||||||
|
eval('"([\\uD800-\\uDBFF](?![\\uDC00-\\uDFFF])|(?<![\\uD800-\\uDBFF])[\\uDC00-\\uDFFF])"'))
|
||||||
|
|
||||||
# List of where new lines occur
|
# List of where new lines occur
|
||||||
self.newLines = [0]
|
self.newLines = [0]
|
||||||
@@ -265,11 +281,12 @@ class HTMLUnicodeInputStream(object):
|
|||||||
self._bufferedCharacter = data[-1]
|
self._bufferedCharacter = data[-1]
|
||||||
data = data[:-1]
|
data = data[:-1]
|
||||||
|
|
||||||
self.reportCharacterErrors(data)
|
if self.reportCharacterErrors:
|
||||||
|
self.reportCharacterErrors(data)
|
||||||
|
|
||||||
# Replace invalid characters
|
# Replace invalid characters
|
||||||
# Note U+0000 is dealt with in the tokenizer
|
# Note U+0000 is dealt with in the tokenizer
|
||||||
data = self.replaceCharactersRegexp.sub("\ufffd", data)
|
data = self.replaceCharactersRegexp.sub("\ufffd", data)
|
||||||
|
|
||||||
data = data.replace("\r\n", "\n")
|
data = data.replace("\r\n", "\n")
|
||||||
data = data.replace("\r", "\n")
|
data = data.replace("\r", "\n")
|
||||||
|
|||||||
@@ -2,11 +2,26 @@ from __future__ import absolute_import, division, unicode_literals
|
|||||||
|
|
||||||
import re
|
import re
|
||||||
from xml.sax.saxutils import escape, unescape
|
from xml.sax.saxutils import escape, unescape
|
||||||
|
from six.moves import urllib_parse as urlparse
|
||||||
|
|
||||||
from .tokenizer import HTMLTokenizer
|
from .tokenizer import HTMLTokenizer
|
||||||
from .constants import tokenTypes
|
from .constants import tokenTypes
|
||||||
|
|
||||||
|
|
||||||
|
content_type_rgx = re.compile(r'''
|
||||||
|
^
|
||||||
|
# Match a content type <application>/<type>
|
||||||
|
(?P<content_type>[-a-zA-Z0-9.]+/[-a-zA-Z0-9.]+)
|
||||||
|
# Match any character set and encoding
|
||||||
|
(?:(?:;charset=(?:[-a-zA-Z0-9]+)(?:;(?:base64))?)
|
||||||
|
|(?:;(?:base64))?(?:;charset=(?:[-a-zA-Z0-9]+))?)
|
||||||
|
# Assume the rest is data
|
||||||
|
,.*
|
||||||
|
$
|
||||||
|
''',
|
||||||
|
re.VERBOSE)
|
||||||
|
|
||||||
|
|
||||||
class HTMLSanitizerMixin(object):
|
class HTMLSanitizerMixin(object):
|
||||||
""" sanitization of XHTML+MathML+SVG and of inline style attributes."""
|
""" sanitization of XHTML+MathML+SVG and of inline style attributes."""
|
||||||
|
|
||||||
@@ -100,8 +115,8 @@ class HTMLSanitizerMixin(object):
|
|||||||
'xml:base', 'xml:lang', 'xml:space', 'xmlns', 'xmlns:xlink', 'y',
|
'xml:base', 'xml:lang', 'xml:space', 'xmlns', 'xmlns:xlink', 'y',
|
||||||
'y1', 'y2', 'zoomAndPan']
|
'y1', 'y2', 'zoomAndPan']
|
||||||
|
|
||||||
attr_val_is_uri = ['href', 'src', 'cite', 'action', 'longdesc', 'poster',
|
attr_val_is_uri = ['href', 'src', 'cite', 'action', 'longdesc', 'poster', 'background', 'datasrc',
|
||||||
'xlink:href', 'xml:base']
|
'dynsrc', 'lowsrc', 'ping', 'poster', 'xlink:href', 'xml:base']
|
||||||
|
|
||||||
svg_attr_val_allows_ref = ['clip-path', 'color-profile', 'cursor', 'fill',
|
svg_attr_val_allows_ref = ['clip-path', 'color-profile', 'cursor', 'fill',
|
||||||
'filter', 'marker', 'marker-start', 'marker-mid', 'marker-end',
|
'filter', 'marker', 'marker-start', 'marker-mid', 'marker-end',
|
||||||
@@ -138,7 +153,9 @@ class HTMLSanitizerMixin(object):
|
|||||||
acceptable_protocols = ['ed2k', 'ftp', 'http', 'https', 'irc',
|
acceptable_protocols = ['ed2k', 'ftp', 'http', 'https', 'irc',
|
||||||
'mailto', 'news', 'gopher', 'nntp', 'telnet', 'webcal',
|
'mailto', 'news', 'gopher', 'nntp', 'telnet', 'webcal',
|
||||||
'xmpp', 'callto', 'feed', 'urn', 'aim', 'rsync', 'tag',
|
'xmpp', 'callto', 'feed', 'urn', 'aim', 'rsync', 'tag',
|
||||||
'ssh', 'sftp', 'rtsp', 'afs']
|
'ssh', 'sftp', 'rtsp', 'afs', 'data']
|
||||||
|
|
||||||
|
acceptable_content_types = ['image/png', 'image/jpeg', 'image/gif', 'image/webp', 'image/bmp', 'text/plain']
|
||||||
|
|
||||||
# subclasses may define their own versions of these constants
|
# subclasses may define their own versions of these constants
|
||||||
allowed_elements = acceptable_elements + mathml_elements + svg_elements
|
allowed_elements = acceptable_elements + mathml_elements + svg_elements
|
||||||
@@ -147,6 +164,7 @@ class HTMLSanitizerMixin(object):
|
|||||||
allowed_css_keywords = acceptable_css_keywords
|
allowed_css_keywords = acceptable_css_keywords
|
||||||
allowed_svg_properties = acceptable_svg_properties
|
allowed_svg_properties = acceptable_svg_properties
|
||||||
allowed_protocols = acceptable_protocols
|
allowed_protocols = acceptable_protocols
|
||||||
|
allowed_content_types = acceptable_content_types
|
||||||
|
|
||||||
# Sanitize the +html+, escaping all elements not in ALLOWED_ELEMENTS, and
|
# Sanitize the +html+, escaping all elements not in ALLOWED_ELEMENTS, and
|
||||||
# stripping out all # attributes not in ALLOWED_ATTRIBUTES. Style
|
# stripping out all # attributes not in ALLOWED_ATTRIBUTES. Style
|
||||||
@@ -189,10 +207,17 @@ class HTMLSanitizerMixin(object):
|
|||||||
unescape(attrs[attr])).lower()
|
unescape(attrs[attr])).lower()
|
||||||
# remove replacement characters from unescaped characters
|
# remove replacement characters from unescaped characters
|
||||||
val_unescaped = val_unescaped.replace("\ufffd", "")
|
val_unescaped = val_unescaped.replace("\ufffd", "")
|
||||||
if (re.match("^[a-z0-9][-+.a-z0-9]*:", val_unescaped) and
|
uri = urlparse.urlparse(val_unescaped)
|
||||||
(val_unescaped.split(':')[0] not in
|
if uri and uri.scheme:
|
||||||
self.allowed_protocols)):
|
if uri.scheme not in self.allowed_protocols:
|
||||||
del attrs[attr]
|
del attrs[attr]
|
||||||
|
if uri.scheme == 'data':
|
||||||
|
m = content_type_rgx.match(uri.path)
|
||||||
|
if not m:
|
||||||
|
del attrs[attr]
|
||||||
|
elif m.group('content_type') not in self.allowed_content_types:
|
||||||
|
del attrs[attr]
|
||||||
|
|
||||||
for attr in self.svg_attr_val_allows_ref:
|
for attr in self.svg_attr_val_allows_ref:
|
||||||
if attr in attrs:
|
if attr in attrs:
|
||||||
attrs[attr] = re.sub(r'url\s*\(\s*[^#\s][^)]+?\)',
|
attrs[attr] = re.sub(r'url\s*\(\s*[^#\s][^)]+?\)',
|
||||||
@@ -245,7 +270,7 @@ class HTMLSanitizerMixin(object):
|
|||||||
elif prop.split('-')[0].lower() in ['background', 'border', 'margin',
|
elif prop.split('-')[0].lower() in ['background', 'border', 'margin',
|
||||||
'padding']:
|
'padding']:
|
||||||
for keyword in value.split():
|
for keyword in value.split():
|
||||||
if not keyword in self.acceptable_css_keywords and \
|
if keyword not in self.acceptable_css_keywords and \
|
||||||
not re.match("^(#[0-9a-f]+|rgb\(\d+%?,\d*%?,?\d*%?\)?|\d{0,2}\.?\d{0,2}(cm|em|ex|in|mm|pc|pt|px|%|,|\))?)$", keyword):
|
not re.match("^(#[0-9a-f]+|rgb\(\d+%?,\d*%?,?\d*%?\)?|\d{0,2}\.?\d{0,2}(cm|em|ex|in|mm|pc|pt|px|%|,|\))?)$", keyword):
|
||||||
break
|
break
|
||||||
else:
|
else:
|
||||||
|
|||||||
@@ -1,9 +1,6 @@
|
|||||||
from __future__ import absolute_import, division, unicode_literals
|
from __future__ import absolute_import, division, unicode_literals
|
||||||
from six import text_type
|
from six import text_type
|
||||||
|
|
||||||
import gettext
|
|
||||||
_ = gettext.gettext
|
|
||||||
|
|
||||||
try:
|
try:
|
||||||
from functools import reduce
|
from functools import reduce
|
||||||
except ImportError:
|
except ImportError:
|
||||||
@@ -35,7 +32,7 @@ else:
|
|||||||
v = utils.surrogatePairToCodepoint(v)
|
v = utils.surrogatePairToCodepoint(v)
|
||||||
else:
|
else:
|
||||||
v = ord(v)
|
v = ord(v)
|
||||||
if not v in encode_entity_map or k.islower():
|
if v not in encode_entity_map or k.islower():
|
||||||
# prefer < over < and similarly for &, >, etc.
|
# prefer < over < and similarly for &, >, etc.
|
||||||
encode_entity_map[v] = k
|
encode_entity_map[v] = k
|
||||||
|
|
||||||
@@ -208,7 +205,7 @@ class HTMLSerializer(object):
|
|||||||
if token["systemId"]:
|
if token["systemId"]:
|
||||||
if token["systemId"].find('"') >= 0:
|
if token["systemId"].find('"') >= 0:
|
||||||
if token["systemId"].find("'") >= 0:
|
if token["systemId"].find("'") >= 0:
|
||||||
self.serializeError(_("System identifer contains both single and double quote characters"))
|
self.serializeError("System identifer contains both single and double quote characters")
|
||||||
quote_char = "'"
|
quote_char = "'"
|
||||||
else:
|
else:
|
||||||
quote_char = '"'
|
quote_char = '"'
|
||||||
@@ -220,7 +217,7 @@ class HTMLSerializer(object):
|
|||||||
elif type in ("Characters", "SpaceCharacters"):
|
elif type in ("Characters", "SpaceCharacters"):
|
||||||
if type == "SpaceCharacters" or in_cdata:
|
if type == "SpaceCharacters" or in_cdata:
|
||||||
if in_cdata and token["data"].find("</") >= 0:
|
if in_cdata and token["data"].find("</") >= 0:
|
||||||
self.serializeError(_("Unexpected </ in CDATA"))
|
self.serializeError("Unexpected </ in CDATA")
|
||||||
yield self.encode(token["data"])
|
yield self.encode(token["data"])
|
||||||
else:
|
else:
|
||||||
yield self.encode(escape(token["data"]))
|
yield self.encode(escape(token["data"]))
|
||||||
@@ -231,7 +228,7 @@ class HTMLSerializer(object):
|
|||||||
if name in rcdataElements and not self.escape_rcdata:
|
if name in rcdataElements and not self.escape_rcdata:
|
||||||
in_cdata = True
|
in_cdata = True
|
||||||
elif in_cdata:
|
elif in_cdata:
|
||||||
self.serializeError(_("Unexpected child element of a CDATA element"))
|
self.serializeError("Unexpected child element of a CDATA element")
|
||||||
for (attr_namespace, attr_name), attr_value in token["data"].items():
|
for (attr_namespace, attr_name), attr_value in token["data"].items():
|
||||||
# TODO: Add namespace support here
|
# TODO: Add namespace support here
|
||||||
k = attr_name
|
k = attr_name
|
||||||
@@ -279,20 +276,20 @@ class HTMLSerializer(object):
|
|||||||
if name in rcdataElements:
|
if name in rcdataElements:
|
||||||
in_cdata = False
|
in_cdata = False
|
||||||
elif in_cdata:
|
elif in_cdata:
|
||||||
self.serializeError(_("Unexpected child element of a CDATA element"))
|
self.serializeError("Unexpected child element of a CDATA element")
|
||||||
yield self.encodeStrict("</%s>" % name)
|
yield self.encodeStrict("</%s>" % name)
|
||||||
|
|
||||||
elif type == "Comment":
|
elif type == "Comment":
|
||||||
data = token["data"]
|
data = token["data"]
|
||||||
if data.find("--") >= 0:
|
if data.find("--") >= 0:
|
||||||
self.serializeError(_("Comment contains --"))
|
self.serializeError("Comment contains --")
|
||||||
yield self.encodeStrict("<!--%s-->" % token["data"])
|
yield self.encodeStrict("<!--%s-->" % token["data"])
|
||||||
|
|
||||||
elif type == "Entity":
|
elif type == "Entity":
|
||||||
name = token["name"]
|
name = token["name"]
|
||||||
key = name + ";"
|
key = name + ";"
|
||||||
if not key in entities:
|
if key not in entities:
|
||||||
self.serializeError(_("Entity %s not recognized" % name))
|
self.serializeError("Entity %s not recognized" % name)
|
||||||
if self.resolve_entities and key not in xmlEntities:
|
if self.resolve_entities and key not in xmlEntities:
|
||||||
data = entities[key]
|
data = entities[key]
|
||||||
else:
|
else:
|
||||||
|
|||||||
@@ -158,7 +158,7 @@ def getDomBuilder(DomImplementation):
|
|||||||
else:
|
else:
|
||||||
# HACK: allow text nodes as children of the document node
|
# HACK: allow text nodes as children of the document node
|
||||||
if hasattr(self.dom, '_child_node_types'):
|
if hasattr(self.dom, '_child_node_types'):
|
||||||
if not Node.TEXT_NODE in self.dom._child_node_types:
|
if Node.TEXT_NODE not in self.dom._child_node_types:
|
||||||
self.dom._child_node_types = list(self.dom._child_node_types)
|
self.dom._child_node_types = list(self.dom._child_node_types)
|
||||||
self.dom._child_node_types.append(Node.TEXT_NODE)
|
self.dom._child_node_types.append(Node.TEXT_NODE)
|
||||||
self.dom.appendChild(self.dom.createTextNode(data))
|
self.dom.appendChild(self.dom.createTextNode(data))
|
||||||
|
|||||||
@@ -10,8 +10,12 @@ returning an iterator generating tokens.
|
|||||||
|
|
||||||
from __future__ import absolute_import, division, unicode_literals
|
from __future__ import absolute_import, division, unicode_literals
|
||||||
|
|
||||||
|
__all__ = ["getTreeWalker", "pprint", "dom", "etree", "genshistream", "lxmletree",
|
||||||
|
"pulldom"]
|
||||||
|
|
||||||
import sys
|
import sys
|
||||||
|
|
||||||
|
from .. import constants
|
||||||
from ..utils import default_etree
|
from ..utils import default_etree
|
||||||
|
|
||||||
treeWalkerCache = {}
|
treeWalkerCache = {}
|
||||||
@@ -55,3 +59,89 @@ def getTreeWalker(treeType, implementation=None, **kwargs):
|
|||||||
# XXX: NEVER cache here, caching is done in the etree submodule
|
# XXX: NEVER cache here, caching is done in the etree submodule
|
||||||
return etree.getETreeModule(implementation, **kwargs).TreeWalker
|
return etree.getETreeModule(implementation, **kwargs).TreeWalker
|
||||||
return treeWalkerCache.get(treeType)
|
return treeWalkerCache.get(treeType)
|
||||||
|
|
||||||
|
|
||||||
|
def concatenateCharacterTokens(tokens):
|
||||||
|
pendingCharacters = []
|
||||||
|
for token in tokens:
|
||||||
|
type = token["type"]
|
||||||
|
if type in ("Characters", "SpaceCharacters"):
|
||||||
|
pendingCharacters.append(token["data"])
|
||||||
|
else:
|
||||||
|
if pendingCharacters:
|
||||||
|
yield {"type": "Characters", "data": "".join(pendingCharacters)}
|
||||||
|
pendingCharacters = []
|
||||||
|
yield token
|
||||||
|
if pendingCharacters:
|
||||||
|
yield {"type": "Characters", "data": "".join(pendingCharacters)}
|
||||||
|
|
||||||
|
|
||||||
|
def pprint(walker):
|
||||||
|
"""Pretty printer for tree walkers"""
|
||||||
|
output = []
|
||||||
|
indent = 0
|
||||||
|
for token in concatenateCharacterTokens(walker):
|
||||||
|
type = token["type"]
|
||||||
|
if type in ("StartTag", "EmptyTag"):
|
||||||
|
# tag name
|
||||||
|
if token["namespace"] and token["namespace"] != constants.namespaces["html"]:
|
||||||
|
if token["namespace"] in constants.prefixes:
|
||||||
|
ns = constants.prefixes[token["namespace"]]
|
||||||
|
else:
|
||||||
|
ns = token["namespace"]
|
||||||
|
name = "%s %s" % (ns, token["name"])
|
||||||
|
else:
|
||||||
|
name = token["name"]
|
||||||
|
output.append("%s<%s>" % (" " * indent, name))
|
||||||
|
indent += 2
|
||||||
|
# attributes (sorted for consistent ordering)
|
||||||
|
attrs = token["data"]
|
||||||
|
for (namespace, localname), value in sorted(attrs.items()):
|
||||||
|
if namespace:
|
||||||
|
if namespace in constants.prefixes:
|
||||||
|
ns = constants.prefixes[namespace]
|
||||||
|
else:
|
||||||
|
ns = namespace
|
||||||
|
name = "%s %s" % (ns, localname)
|
||||||
|
else:
|
||||||
|
name = localname
|
||||||
|
output.append("%s%s=\"%s\"" % (" " * indent, name, value))
|
||||||
|
# self-closing
|
||||||
|
if type == "EmptyTag":
|
||||||
|
indent -= 2
|
||||||
|
|
||||||
|
elif type == "EndTag":
|
||||||
|
indent -= 2
|
||||||
|
|
||||||
|
elif type == "Comment":
|
||||||
|
output.append("%s<!-- %s -->" % (" " * indent, token["data"]))
|
||||||
|
|
||||||
|
elif type == "Doctype":
|
||||||
|
if token["name"]:
|
||||||
|
if token["publicId"]:
|
||||||
|
output.append("""%s<!DOCTYPE %s "%s" "%s">""" %
|
||||||
|
(" " * indent,
|
||||||
|
token["name"],
|
||||||
|
token["publicId"],
|
||||||
|
token["systemId"] if token["systemId"] else ""))
|
||||||
|
elif token["systemId"]:
|
||||||
|
output.append("""%s<!DOCTYPE %s "" "%s">""" %
|
||||||
|
(" " * indent,
|
||||||
|
token["name"],
|
||||||
|
token["systemId"]))
|
||||||
|
else:
|
||||||
|
output.append("%s<!DOCTYPE %s>" % (" " * indent,
|
||||||
|
token["name"]))
|
||||||
|
else:
|
||||||
|
output.append("%s<!DOCTYPE >" % (" " * indent,))
|
||||||
|
|
||||||
|
elif type == "Characters":
|
||||||
|
output.append("%s\"%s\"" % (" " * indent, token["data"]))
|
||||||
|
|
||||||
|
elif type == "SpaceCharacters":
|
||||||
|
assert False, "concatenateCharacterTokens should have got rid of all Space tokens"
|
||||||
|
|
||||||
|
else:
|
||||||
|
raise ValueError("Unknown token type, %s" % type)
|
||||||
|
|
||||||
|
return "\n".join(output)
|
||||||
|
|||||||
@@ -1,8 +1,8 @@
|
|||||||
from __future__ import absolute_import, division, unicode_literals
|
from __future__ import absolute_import, division, unicode_literals
|
||||||
from six import text_type, string_types
|
from six import text_type, string_types
|
||||||
|
|
||||||
import gettext
|
__all__ = ["DOCUMENT", "DOCTYPE", "TEXT", "ELEMENT", "COMMENT", "ENTITY", "UNKNOWN",
|
||||||
_ = gettext.gettext
|
"TreeWalker", "NonRecursiveTreeWalker"]
|
||||||
|
|
||||||
from xml.dom import Node
|
from xml.dom import Node
|
||||||
|
|
||||||
@@ -58,7 +58,7 @@ class TreeWalker(object):
|
|||||||
"namespace": to_text(namespace),
|
"namespace": to_text(namespace),
|
||||||
"data": attrs}
|
"data": attrs}
|
||||||
if hasChildren:
|
if hasChildren:
|
||||||
yield self.error(_("Void element has children"))
|
yield self.error("Void element has children")
|
||||||
|
|
||||||
def startTag(self, namespace, name, attrs):
|
def startTag(self, namespace, name, attrs):
|
||||||
assert namespace is None or isinstance(namespace, string_types), type(namespace)
|
assert namespace is None or isinstance(namespace, string_types), type(namespace)
|
||||||
@@ -122,7 +122,7 @@ class TreeWalker(object):
|
|||||||
return {"type": "Entity", "name": text_type(name)}
|
return {"type": "Entity", "name": text_type(name)}
|
||||||
|
|
||||||
def unknown(self, nodeType):
|
def unknown(self, nodeType):
|
||||||
return self.error(_("Unknown node type: ") + nodeType)
|
return self.error("Unknown node type: " + nodeType)
|
||||||
|
|
||||||
|
|
||||||
class NonRecursiveTreeWalker(TreeWalker):
|
class NonRecursiveTreeWalker(TreeWalker):
|
||||||
|
|||||||
@@ -2,9 +2,6 @@ from __future__ import absolute_import, division, unicode_literals
|
|||||||
|
|
||||||
from xml.dom import Node
|
from xml.dom import Node
|
||||||
|
|
||||||
import gettext
|
|
||||||
_ = gettext.gettext
|
|
||||||
|
|
||||||
from . import _base
|
from . import _base
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -7,12 +7,10 @@ except ImportError:
|
|||||||
from ordereddict import OrderedDict
|
from ordereddict import OrderedDict
|
||||||
except ImportError:
|
except ImportError:
|
||||||
OrderedDict = dict
|
OrderedDict = dict
|
||||||
import gettext
|
|
||||||
_ = gettext.gettext
|
|
||||||
|
|
||||||
import re
|
import re
|
||||||
|
|
||||||
from six import text_type
|
from six import string_types
|
||||||
|
|
||||||
from . import _base
|
from . import _base
|
||||||
from ..utils import moduleFactoryFactory
|
from ..utils import moduleFactoryFactory
|
||||||
@@ -60,7 +58,7 @@ def getETreeBuilder(ElementTreeImplementation):
|
|||||||
return _base.COMMENT, node.text
|
return _base.COMMENT, node.text
|
||||||
|
|
||||||
else:
|
else:
|
||||||
assert type(node.tag) == text_type, type(node.tag)
|
assert isinstance(node.tag, string_types), type(node.tag)
|
||||||
# This is assumed to be an ordinary element
|
# This is assumed to be an ordinary element
|
||||||
match = tag_regexp.match(node.tag)
|
match = tag_regexp.match(node.tag)
|
||||||
if match:
|
if match:
|
||||||
|
|||||||
@@ -4,9 +4,6 @@ from six import text_type
|
|||||||
from lxml import etree
|
from lxml import etree
|
||||||
from ..treebuilders.etree import tag_regexp
|
from ..treebuilders.etree import tag_regexp
|
||||||
|
|
||||||
from gettext import gettext
|
|
||||||
_ = gettext
|
|
||||||
|
|
||||||
from . import _base
|
from . import _base
|
||||||
|
|
||||||
from .. import ihatexml
|
from .. import ihatexml
|
||||||
@@ -130,7 +127,7 @@ class TreeWalker(_base.NonRecursiveTreeWalker):
|
|||||||
def getNodeDetails(self, node):
|
def getNodeDetails(self, node):
|
||||||
if isinstance(node, tuple): # Text node
|
if isinstance(node, tuple): # Text node
|
||||||
node, key = node
|
node, key = node
|
||||||
assert key in ("text", "tail"), _("Text nodes are text or tail, found %s") % key
|
assert key in ("text", "tail"), "Text nodes are text or tail, found %s" % key
|
||||||
return _base.TEXT, ensure_str(getattr(node, key))
|
return _base.TEXT, ensure_str(getattr(node, key))
|
||||||
|
|
||||||
elif isinstance(node, Root):
|
elif isinstance(node, Root):
|
||||||
@@ -169,7 +166,7 @@ class TreeWalker(_base.NonRecursiveTreeWalker):
|
|||||||
attrs, len(node) > 0 or node.text)
|
attrs, len(node) > 0 or node.text)
|
||||||
|
|
||||||
def getFirstChild(self, node):
|
def getFirstChild(self, node):
|
||||||
assert not isinstance(node, tuple), _("Text nodes have no children")
|
assert not isinstance(node, tuple), "Text nodes have no children"
|
||||||
|
|
||||||
assert len(node) or node.text, "Node has no children"
|
assert len(node) or node.text, "Node has no children"
|
||||||
if node.text:
|
if node.text:
|
||||||
@@ -180,7 +177,7 @@ class TreeWalker(_base.NonRecursiveTreeWalker):
|
|||||||
def getNextSibling(self, node):
|
def getNextSibling(self, node):
|
||||||
if isinstance(node, tuple): # Text node
|
if isinstance(node, tuple): # Text node
|
||||||
node, key = node
|
node, key = node
|
||||||
assert key in ("text", "tail"), _("Text nodes are text or tail, found %s") % key
|
assert key in ("text", "tail"), "Text nodes are text or tail, found %s" % key
|
||||||
if key == "text":
|
if key == "text":
|
||||||
# XXX: we cannot use a "bool(node) and node[0] or None" construct here
|
# XXX: we cannot use a "bool(node) and node[0] or None" construct here
|
||||||
# because node[0] might evaluate to False if it has no child element
|
# because node[0] might evaluate to False if it has no child element
|
||||||
@@ -196,7 +193,7 @@ class TreeWalker(_base.NonRecursiveTreeWalker):
|
|||||||
def getParentNode(self, node):
|
def getParentNode(self, node):
|
||||||
if isinstance(node, tuple): # Text node
|
if isinstance(node, tuple): # Text node
|
||||||
node, key = node
|
node, key = node
|
||||||
assert key in ("text", "tail"), _("Text nodes are text or tail, found %s") % key
|
assert key in ("text", "tail"), "Text nodes are text or tail, found %s" % key
|
||||||
if key == "text":
|
if key == "text":
|
||||||
return node
|
return node
|
||||||
# else: fallback to "normal" processing
|
# else: fallback to "normal" processing
|
||||||
|
|||||||
+22
-1
@@ -2,6 +2,8 @@ from __future__ import absolute_import, division, unicode_literals
|
|||||||
|
|
||||||
from types import ModuleType
|
from types import ModuleType
|
||||||
|
|
||||||
|
from six import text_type
|
||||||
|
|
||||||
try:
|
try:
|
||||||
import xml.etree.cElementTree as default_etree
|
import xml.etree.cElementTree as default_etree
|
||||||
except ImportError:
|
except ImportError:
|
||||||
@@ -9,7 +11,26 @@ except ImportError:
|
|||||||
|
|
||||||
|
|
||||||
__all__ = ["default_etree", "MethodDispatcher", "isSurrogatePair",
|
__all__ = ["default_etree", "MethodDispatcher", "isSurrogatePair",
|
||||||
"surrogatePairToCodepoint", "moduleFactoryFactory"]
|
"surrogatePairToCodepoint", "moduleFactoryFactory",
|
||||||
|
"supports_lone_surrogates"]
|
||||||
|
|
||||||
|
|
||||||
|
# Platforms not supporting lone surrogates (\uD800-\uDFFF) should be
|
||||||
|
# caught by the below test. In general this would be any platform
|
||||||
|
# using UTF-16 as its encoding of unicode strings, such as
|
||||||
|
# Jython. This is because UTF-16 itself is based on the use of such
|
||||||
|
# surrogates, and there is no mechanism to further escape such
|
||||||
|
# escapes.
|
||||||
|
try:
|
||||||
|
_x = eval('"\\uD800"')
|
||||||
|
if not isinstance(_x, text_type):
|
||||||
|
# We need this with u"" because of http://bugs.jython.org/issue2039
|
||||||
|
_x = eval('u"\\uD800"')
|
||||||
|
assert isinstance(_x, text_type)
|
||||||
|
except:
|
||||||
|
supports_lone_surrogates = False
|
||||||
|
else:
|
||||||
|
supports_lone_surrogates = True
|
||||||
|
|
||||||
|
|
||||||
class MethodDispatcher(dict):
|
class MethodDispatcher(dict):
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
#!/usr/bin/python
|
#!/usr/bin/python
|
||||||
|
|
||||||
from pynma import PyNMA
|
from .pynma import PyNMA
|
||||||
|
|
||||||
|
|||||||
Regular → Executable
+39
-28
@@ -1,12 +1,20 @@
|
|||||||
#!/usr/bin/python
|
#!/usr/bin/python
|
||||||
|
|
||||||
from xml.dom.minidom import parseString
|
from xml.dom.minidom import parseString
|
||||||
from httplib import HTTPSConnection
|
|
||||||
from urllib import urlencode
|
|
||||||
|
|
||||||
__version__ = "0.1"
|
try:
|
||||||
|
from http.client import HTTPSConnection
|
||||||
|
except ImportError:
|
||||||
|
from httplib import HTTPSConnection
|
||||||
|
|
||||||
API_SERVER = 'nma.usk.bz'
|
try:
|
||||||
|
from urllib.parse import urlencode
|
||||||
|
except ImportError:
|
||||||
|
from urllib import urlencode
|
||||||
|
|
||||||
|
__version__ = "1.0"
|
||||||
|
|
||||||
|
API_SERVER = 'www.notifymyandroid.com'
|
||||||
ADD_PATH = '/publicapi/notify'
|
ADD_PATH = '/publicapi/notify'
|
||||||
|
|
||||||
USER_AGENT="PyNMA/v%s"%__version__
|
USER_AGENT="PyNMA/v%s"%__version__
|
||||||
@@ -18,14 +26,14 @@ def uniq_preserve(seq): # Dave Kirby
|
|||||||
|
|
||||||
def uniq(seq):
|
def uniq(seq):
|
||||||
# Not order preserving
|
# Not order preserving
|
||||||
return {}.fromkeys(seq).keys()
|
return list({}.fromkeys(seq).keys())
|
||||||
|
|
||||||
class PyNMA(object):
|
class PyNMA(object):
|
||||||
"""PyNMA(apikey=[], developerkey=None)
|
"""PyNMA(apikey=[], developerkey=None)
|
||||||
takes 2 optional arguments:
|
takes 2 optional arguments:
|
||||||
- (opt) apykey: might me a string containing 1 key or an array of keys
|
- (opt) apykey: might me a string containing 1 key or an array of keys
|
||||||
- (opt) developerkey: where you can store your developer key
|
- (opt) developerkey: where you can store your developer key
|
||||||
"""
|
"""
|
||||||
|
|
||||||
def __init__(self, apikey=[], developerkey=None):
|
def __init__(self, apikey=[], developerkey=None):
|
||||||
self._developerkey = None
|
self._developerkey = None
|
||||||
@@ -60,19 +68,20 @@ class PyNMA(object):
|
|||||||
if type(developerkey) == str and len(developerkey) == 48:
|
if type(developerkey) == str and len(developerkey) == 48:
|
||||||
self._developerkey = developerkey
|
self._developerkey = developerkey
|
||||||
|
|
||||||
def push(self, application="", event="", description="", url="", priority=0, batch_mode=False):
|
def push(self, application="", event="", description="", url="", contenttype=None, priority=0, batch_mode=False, html=False):
|
||||||
"""Pushes a message on the registered API keys.
|
"""Pushes a message on the registered API keys.
|
||||||
takes 5 arguments:
|
takes 5 arguments:
|
||||||
- (req) application: application name [256]
|
- (req) application: application name [256]
|
||||||
- (req) event: event name [1000]
|
- (req) event: event name [1000]
|
||||||
- (req) description: description [10000]
|
- (req) description: description [10000]
|
||||||
- (opt) url: url [512]
|
- (opt) url: url [512]
|
||||||
- (opt) priority: from -2 (lowest) to 2 (highest) (def:0)
|
- (opt) contenttype: Content Type (act: None (plain text) or text/html)
|
||||||
- (opt) batch_mode: call API 5 by 5 (def:False)
|
- (opt) priority: from -2 (lowest) to 2 (highest) (def:0)
|
||||||
|
- (opt) batch_mode: push to all keys at once (def:False)
|
||||||
Warning: using batch_mode will return error only if all API keys are bad
|
- (opt) html: shortcut for contenttype=text/html
|
||||||
cf: http://nma.usk.bz/api.php
|
Warning: using batch_mode will return error only if all API keys are bad
|
||||||
"""
|
cf: http://nma.usk.bz/api.php
|
||||||
|
"""
|
||||||
datas = {
|
datas = {
|
||||||
'application': application[:256].encode('utf8'),
|
'application': application[:256].encode('utf8'),
|
||||||
'event': event[:1024].encode('utf8'),
|
'event': event[:1024].encode('utf8'),
|
||||||
@@ -83,6 +92,9 @@ class PyNMA(object):
|
|||||||
if url:
|
if url:
|
||||||
datas['url'] = url[:512]
|
datas['url'] = url[:512]
|
||||||
|
|
||||||
|
if contenttype == "text/html" or html == True: # Currently only accepted content type
|
||||||
|
datas['content-type'] = "text/html"
|
||||||
|
|
||||||
if self._developerkey:
|
if self._developerkey:
|
||||||
datas['developerkey'] = self._developerkey
|
datas['developerkey'] = self._developerkey
|
||||||
|
|
||||||
@@ -94,10 +106,9 @@ class PyNMA(object):
|
|||||||
res = self.callapi('POST', ADD_PATH, datas)
|
res = self.callapi('POST', ADD_PATH, datas)
|
||||||
results[key] = res
|
results[key] = res
|
||||||
else:
|
else:
|
||||||
for i in range(0, len(self._apikey), 5):
|
datas['apikey'] = ",".join(self._apikey)
|
||||||
datas['apikey'] = ",".join(self._apikey[i:i+5])
|
res = self.callapi('POST', ADD_PATH, datas)
|
||||||
res = self.callapi('POST', ADD_PATH, datas)
|
results[datas['apikey']] = res
|
||||||
results[datas['apikey']] = res
|
|
||||||
return results
|
return results
|
||||||
|
|
||||||
def callapi(self, method, path, args):
|
def callapi(self, method, path, args):
|
||||||
@@ -110,7 +121,7 @@ class PyNMA(object):
|
|||||||
|
|
||||||
try:
|
try:
|
||||||
res = self._parse_reponse(resp.read())
|
res = self._parse_reponse(resp.read())
|
||||||
except Exception, e:
|
except Exception as e:
|
||||||
res = {'type': "pynmaerror",
|
res = {'type': "pynmaerror",
|
||||||
'code': 600,
|
'code': 600,
|
||||||
'message': str(e)
|
'message': str(e)
|
||||||
@@ -124,12 +135,12 @@ class PyNMA(object):
|
|||||||
for elem in root.childNodes:
|
for elem in root.childNodes:
|
||||||
if elem.nodeType == elem.TEXT_NODE: continue
|
if elem.nodeType == elem.TEXT_NODE: continue
|
||||||
if elem.tagName == 'success':
|
if elem.tagName == 'success':
|
||||||
res = dict(elem.attributes.items())
|
res = dict(list(elem.attributes.items()))
|
||||||
res['message'] = ""
|
res['message'] = ""
|
||||||
res['type'] = elem.tagName
|
res['type'] = elem.tagName
|
||||||
return res
|
return res
|
||||||
if elem.tagName == 'error':
|
if elem.tagName == 'error':
|
||||||
res = dict(elem.attributes.items())
|
res = dict(list(elem.attributes.items()))
|
||||||
res['message'] = elem.firstChild.nodeValue
|
res['message'] = elem.firstChild.nodeValue
|
||||||
res['type'] = elem.tagName
|
res['type'] = elem.tagName
|
||||||
return res
|
return res
|
||||||
|
|||||||
Reference in New Issue
Block a user