Compare commits

..
69 Commits
Author SHA1 Message Date
rembo10 6720575bd9 v0.5.9 2015-09-05 16:01:52 -07:00
rembo10 b1ec750d1d Updated changelog for v0.5.9 2015-09-05 16:00:55 -07:00
Ade d9c78da997 blackhole convert magnet
Seems only torcache working, comment others and add user agent
2015-09-05 11:42:35 +12:00
Ade 1fa26e763d kat flac 403 fix 2015-09-05 00:03:37 +12:00
Ade bb71b3d3d9 Strike provider 2015-09-04 21:09:22 +12:00
Ade 65e8e56e4a Show artist name rather than guid when deleting 2015-09-04 18:45:32 +12:00
Ade ee415d0e50 Fixes #2347 2015-09-04 18:32:15 +12:00
Ade 9db88a6c4f Removing nabz broken 2015-09-03 18:13:23 +12:00
rembo10 e5151a0a19 #2343 Added option to stop post processing if no good metadata match is found 2015-09-02 12:59:17 -07:00
rembo10 570234717a Changing the vip mirror to url to prepare for server upgrade 2015-09-02 12:11:25 -07:00
Ade 3f60d1f8de torznab config issues 2015-09-02 21:16:22 +12:00
AdeHub 65d100c814 Merge pull request #2338 from zone117x/develop
Implemented Jackett (Custom Torznab) option for Torrent Search Providers
2015-08-24 21:13:50 +12:00
unknown 61bdab8621 Implemented Jackett (Custom Torznab) option for Torrent Search Providers 2015-08-23 11:41:23 -06:00
Ade 8c94c9ecbf last.fm fix
failing if mbid not found
2015-08-21 19:23:56 +12:00
Ade 7fd895ecb5 Potential fix for #2316 2015-08-21 18:50:45 +12:00
Ade 213bc07646 Keep Original Folder fix 2015-08-07 17:43:22 +12:00
Ade 1f081c181c rutracker 2nd login
try logging in again after 10 secs if 1st attempt fails
2015-08-05 21:27:43 +12:00
Ade db1d31900e rutracker small fix up 2015-08-03 18:39:51 +12:00
Ade 2219c79ac3 potential kat fix 2015-08-02 13:52:41 +12:00
Ade d2782179aa rutracker revision
- Now uses requests with more logging
- Update to latest BeautifulSoup and html5lib libs
2015-08-02 12:18:25 +12:00
rembo10 d90a31afc7 Fix for #2282 2015-07-23 18:09:38 +01:00
rembo10 1c236fe0f3 Real fix for pushover 2015-07-23 11:25:06 +01:00
rembo10 24165fb5f3 v0.5.8 2015-07-13 15:22:03 -07:00
rembo10 71da960386 Updated changelog for v0.5.8 2015-07-13 15:21:44 -07:00
rembo10 a16c2d0a26 Fix for #2279: What.cd not honoring custom search string 2015-07-13 15:09:01 -07:00
rembo10 53de44a225 typo when excluding similar artist update on an artist scan 2015-07-08 23:15:38 -07:00
rembo10 e32cff60e5 Took out some unneccessary functions if we're just scanning an artist 2015-07-08 22:26:55 -07:00
rembo10 0ebba8d17c Fix for #2271: Scanning an empty directory with DETECT_BITRATE turned on 2015-07-08 21:58:41 -07:00
rembo10 464db79c04 Some cleanup to the scanArtist function 2015-07-08 11:52:00 -07:00
rembo10 65019d8323 Fix for #2260 & #2129 & 2175 & #2163 : Better artist directory scanning 2015-07-08 02:41:17 -07:00
rembo10 f131d316ca Set localhost as default 2015-07-08 00:45:33 -07:00
rembo10 e6663c96b5 Fixed some pushbullet config text 2015-07-07 23:56:17 -07:00
rembo10 5f06d11fe9 Fixed pushbullet notifications 2015-07-07 23:47:38 -07:00
rembo10 6db2836662 Moved plex notifications to requests lib for pre-v12 plex 2015-07-07 22:54:34 -07:00
rembo10 cb209f8d98 Fix for #2264: Metacritic parsing failing with htmlparser 2015-07-06 13:05:51 -07:00
rembo10 5634e17a2e Fix for #2155, local variable response referenced before assignment utorrent 2015-07-06 01:28:48 -07:00
rembo10 7c990edeaf Better fix for #2183 2015-07-06 00:34:16 -07:00
rembo10 9149f62760 Fix for #2173: Uncaught exception in get_age 2015-07-06 00:21:26 -07:00
rembo10 872ec6b11c Fix for #2089 2015-07-06 00:09:43 -07:00
rembo10 064a21a07c Better fix for #2183 2015-07-06 00:02:09 -07:00
rembo10 57c3c7b2e0 Fixed songkick description to prevent spilling over on default fonts 2015-07-05 21:44:37 -07:00
rembo10 6dc75d7a44 Cleaned up the config page a little bit. Took out {option} enabled buttons and just put them in with the option. Alphabetized the notifiers. Fixed checkbox on chrome 2015-07-05 21:40:52 -07:00
rembo10 5f3b7e05c5 Fixed some appearance stuff on the config page (if fonts weren't installed) 2015-07-05 15:04:18 -07:00
rembo10 61d3d8ad00 Fix for history page showing nzbs snatched from hp indexer as unknown 2015-07-05 12:05:04 -07:00
rembo10 e36440f834 Fix for XSS bug when searching 2015-07-04 23:53:03 -07:00
rembo10 4029543d51 Added option to wait until an album's release date before searching 2015-07-04 20:43:06 -07:00
rembo10 8aee9ac91f Fixed plex client url in config 2015-07-04 19:36:39 -07:00
rembo10 6e5e15f06f Fixed plex notifications for version 12 2015-07-04 19:27:27 -07:00
rembo10 a72bd4cb7c Fixed up the appearance of the config page (was kind of broken on Firefox) 2015-07-04 17:26:47 -07:00
rembo10 d062ab553c Merge branch 'master' of https://github.com/texke/headphones into develop 2015-07-04 12:26:05 -07:00
rembo10 a41edde12a Merge branch 'wfriesen-logtypo' into develop 2015-07-04 12:23:57 -07:00
rembo10 7608927b4f Merge branch 'logtypo' of https://github.com/wfriesen/headphones into wfriesen-logtypo 2015-07-04 12:23:29 -07:00
rembo10 24d7647c08 Updated notifymyandroid library 2015-07-04 12:13:40 -07:00
rembo10 a14369cb7d Added option to only include 'official' extras, rather than everything (bootlegs, promos, etc) 2015-07-04 00:47:02 -07:00
rembo10 3e81c4f3b6 Fix for #2183. Check to make sure the response has headers before trying to parse them 2015-07-03 21:01:46 -07:00
rembo10 84a6c19677 Release v0.5.7 2015-07-01 02:18:01 -07:00
rembo10 624a1d5509 Updated changelog for 0.5.7 release 2015-07-01 02:17:51 -07:00
rembo10 8b6d35cfe2 Updated changelog to reflect metacritic changes 2015-07-01 02:08:13 -07:00
rembo10 867e502448 Cache metacritic scores in the artists table so we don't lose them if we can't connect for whatever reason 2015-07-01 02:06:27 -07:00
rembo10 74275b969e Added a useragent to the metacritic request 2015-07-01 00:56:48 -07:00
William Friesen 275dde0630 Fixed typo in postprocessor message 2015-06-28 10:56:48 +10:00
Bas Stottelaar 46b470f30b Merge pull request #2253 from Hellowlol/patch-1
Add getLogs and clearLogs to api
2015-06-26 08:04:05 +02:00
Hellowlol fbfec95a8a Add getLogs and clearLogs to api 2015-06-25 22:53:39 +02:00
rembo10 1e75a18f55 Initial commit to add plex tokens. Plex notifications need to be updated to requests, and the get request to the server might need to be cleaned up - but want to make sure it works first 2015-06-19 17:08:54 -07:00
Jens Rogier a98849bf30 Added Musicbrainz link to "alternate release" window 2015-06-18 17:01:01 +02:00
rembo10 1b998e9d01 Updated pushover to use requests lib 2015-06-12 00:08:35 -07:00
rembo10 ebd6d3f3f9 Added option to keep folders when force post-processing under Manage. Also, added config option to keep original folder globally 2015-06-09 00:31:50 -07:00
rembo10 1204afadbb Convert preferred bitrate to vbr preset for what.cd searching 2015-06-08 21:56:10 -07:00
rembo10 cd08b72df7 Added ability to test pushover 2015-06-08 21:55:14 -07:00
49 changed files with 2511 additions and 1604 deletions
+46
View File
@@ -1,5 +1,51 @@
# Changelog # Changelog
## v0.5.9
Released 05 September 2015
Highlights:
* Added: Providers Strike, Jackett, custom Torznabs
* Added: Option to stop post-processing if no good match found (#2343)
* Fixed: Blackhole -> Magnet, limit to torcache
* Fixed: Kat 403 flac error
* Fixed: Last.fm errors
* Fixed: Pushover notifications
* Improved: Rutracker logging, switched to requests lib
The full list of commits can be found [here](https://github.com/rembo10/headphones/compare/v0.5.6...v0.5.7).
## v0.5.8
Released 13 July 2015
Highlights:
* Added: Option to only include official extras
* Added: Option to wait until album release date before searching
* Fixed: NotifyMyAndroid notifications
* Fixed: Plex Notifications
* Fixed: Metacritic parsing
* Fixed: Pushbullet notifications
* Fixed: What.cd not honoring custom search term (#2279)
* Improved: XSS Search bug
* Improved: Config page layout
* Improved: Set localhost as default
* Improved: Better single artist scanning
The full list of commits can be found [here](https://github.com/rembo10/headphones/compare/v0.5.6...v0.5.7).
## v0.5.7
Released 01 July 2015
Highlights:
* Improved: Moved pushover to use requests lib
* Improved: Plex tokens with Plex Home
* Improved: Added getLogs & clearLogs to api
* Improved: Cache MetaCritic scores. Added user-agent header to fix 403 errors
* Improved: Specify whether to delete folders when force post-processing
* Improved: Convert target bitrate to vbr preset for what.cd searching
* Improved: Switched Pushover to requests lib
The full list of commits can be found [here](https://github.com/rembo10/headphones/compare/v0.5.6...v0.5.7).
## v0.5.6 ## v0.5.6
Released 08 June 2015 Released 08 June 2015
+2 -1
View File
@@ -35,6 +35,7 @@
%for alternate_album in alternate_albums: %for alternate_album in alternate_albums:
<% <%
track_count = len(myDB.select("SELECT * from alltracks WHERE ReleaseID=?", [alternate_album['ReleaseID']])) track_count = len(myDB.select("SELECT * from alltracks WHERE ReleaseID=?", [alternate_album['ReleaseID']]))
mb_link = "http://musicbrainz.org/release/" + alternate_album['ReleaseID']
have_track_count = len(myDB.select("SELECT * from alltracks WHERE ReleaseID=? AND Location IS NOT NULL", [alternate_album['ReleaseID']])) have_track_count = len(myDB.select("SELECT * from alltracks WHERE ReleaseID=? AND Location IS NOT NULL", [alternate_album['ReleaseID']]))
if alternate_album['AlbumID'] == alternate_album['ReleaseID']: if alternate_album['AlbumID'] == alternate_album['ReleaseID']:
alternate_album_name = "Headphones Default Release (" + str(alternate_album['ReleaseDate']) + ") [" + str(have_track_count) + "/" + str(track_count) + " tracks]" alternate_album_name = "Headphones Default Release (" + str(alternate_album['ReleaseDate']) + ") [" + str(have_track_count) + "/" + str(track_count) + " tracks]"
@@ -42,7 +43,7 @@
alternate_album_name = alternate_album['AlbumTitle'] + " (" + alternate_album['ReleaseCountry'] + ", " + str(alternate_album['ReleaseDate']) + ", " + alternate_album['ReleaseFormat'] + ") [" + str(have_track_count) + "/" + str(track_count) + " tracks]" alternate_album_name = alternate_album['AlbumTitle'] + " (" + alternate_album['ReleaseCountry'] + ", " + str(alternate_album['ReleaseDate']) + ", " + alternate_album['ReleaseFormat'] + ") [" + str(have_track_count) + "/" + str(track_count) + " tracks]"
%> %>
<a href="#" onclick="doAjaxCall('switchAlbum?AlbumID=${album['AlbumID']}&ReleaseID=${alternate_album['ReleaseID']}', $(this), 'table');" data-success="Switched release to: ${alternate_album_name}">${alternate_album_name}</a><br> <a href="#" onclick="doAjaxCall('switchAlbum?AlbumID=${album['AlbumID']}&ReleaseID=${alternate_album['ReleaseID']}', $(this), 'table');" data-success="Switched release to: ${alternate_album_name}">${alternate_album_name}</a><a href="${mb_link}" target="_blank">MB</a><br>
%endfor %endfor
%endif %endif
</div> </div>
File diff suppressed because it is too large Load Diff
+20 -1
View File
@@ -325,7 +325,7 @@ form fieldset small.heading {
margin-top: -15px; margin-top: -15px;
} }
form .row { form .row {
font-family: Helvetica, Arial; font-family: Helvetica, Arial, sans-serif;
margin-bottom: 10px; margin-bottom: 10px;
} }
form .row label { form .row label {
@@ -408,6 +408,18 @@ form .checkbox small {
form .indent input { form .indent input {
margin-left: 15px; margin-left: 15px;
} }
form .suboptions {
margin-left: 15px;
}
.option{
font-size: 15px;
font-weight: 600;
vertical-align: middle;
}
input.bigcheck[type="checkbox"] {
width: 16px;
height: 16px;
}
ul, ul,
ol { ol {
margin-left: 2em; margin-left: 2em;
@@ -1420,6 +1432,13 @@ div#artistheader h2 a {
line-height: 10px; line-height: 10px;
height: 10px; height: 10px;
} }
.nopad {
margin-bottom: 0px !important;
margin-top: 0px !important;
padding-top: 0px !important;
padding-bottom: 0px !important;
padding: 0px 0px !important;
}
#album_table th#albumname, #album_table th#albumname,
#upcoming_table th#artistname, #upcoming_table th#artistname,
#wanted_table th#artistname { #wanted_table th#artistname {
+2
View File
@@ -52,6 +52,8 @@
fileid = 'torrent' fileid = 'torrent'
if item['URL'].find('rutracker') != -1: if item['URL'].find('rutracker') != -1:
fileid = 'torrent' fileid = 'torrent'
if item['URL'].find('codeshy') != -1:
fileid = 'nzb'
folder = 'Folder: ' + item['FolderName'] folder = 'Folder: ' + item['FolderName']
+67 -20
View File
@@ -137,27 +137,26 @@
No empty artists found. No empty artists found.
%endif %endif
</div> </div>
<a href="#" onclick="doAjaxCall('forcePostProcess',$(this))" data-success="Post-Processor is being loaded" data-error="Error during Post-Processing"><i class="fa fa-wrench fa-fw"></i> Force Post-Process Albums in Download Folder</a> <div id="post_process">
<a href="#" class="btnOpenDialog"><i class="fa fa-wrench fa-fw"></i> Force Post-Process Albums in Download Folder</a>
</div>
</div> </div>
</fieldset> </fieldset>
<form action="forcePostProcess" method="GET"> <fieldset>
<fieldset> <div class="row" id="post_process_alternate">
<div class="row"> <label>Force Post-Process Albums in Alternate Folder</label>
<label>Force Post-Process Albums in Alternate Folder</label> <input type="text" value="" name="dir" id="dir" size="50" />
<input type="text" value="" name="dir" id="dir" size="50" /><input type="submit" /> <input type="button" class="btnOpenDialog" value="Submit" />
</div> </div>
</fieldset> </fieldset>
</form> <fieldset>
<form action="forcePostProcess" method="GET"> <div class="row" id="post_process_single">
<fieldset> <label>Post-Process Single Folder</label>
<div class="row"> <input type="text" value="" name="album_dir" id="album_dir" size="50" />
<label>Post-Process Single Folder</label> <input type="button" class="btnOpenDialog" value="Submit" />
<input type="text" value="" name="album_dir" id="album_dir" size="50" /><input type="submit" /> </div>
</div> </fieldset>
</fieldset>
</form>
</div> </div>
@@ -177,8 +176,7 @@
</fieldset> </fieldset>
</div> </div>
<div id="dialog-confirm"></div>
</div> </div>
</%def> </%def>
@@ -188,6 +186,55 @@
$('#autoadd').append('<input type="hidden" name="scan" value=1 />'); $('#autoadd').append('<input type="hidden" name="scan" value=1 />');
}; };
function fnOpenNormalDialog(id) {
if (id === "post_process"){
var url = "forcePostProcess?"
}
if (id === "post_process_alternate"){
var dir = $('#dir').val();
var url = "forcePostProcess?dir=" + dir + "&"
}
if (id === "post_process_single"){
var dir = $('#album_dir').val();
var url = "forcePostProcess?album_dir=" + dir + "&"
}
var t = $('<a data-success="Post-Processor is being loaded" data-error="Error during Post-Processing">');
$("#dialog-confirm").html("Do you want to keep the original folder(s) after post-processing? If you click no, the folders still might be kept depending on your global settings");
// Define the Dialog and its properties.
$("#dialog-confirm").dialog({
resizable: false,
modal: true,
title: "Keep original folder(s)?",
height: 170,
width: 400,
buttons: {
"Yes": function () {
$(this).dialog('close');
doAjaxCall(url + "keep_original_folder=True", t);
},
"No": function () {
$(this).dialog('close');
doAjaxCall(url + "keep_original_folder=False", t);
}
}
});
}
$('.btnOpenDialog').click(function(e){
e.preventDefault();
var parentId = $(this).closest('div').prop('id');
fnOpenNormalDialog(parentId);
});
function callback(value) {
if (value) {
alert("Confirmed");
} else {
alert("Rejected");
}
}
function initThisPage() { function initThisPage() {
$('#manage_albums').click(function() { $('#manage_albums').click(function() {
$('#dialog').dialog(); $('#dialog').dialog();
+6 -1
View File
@@ -353,7 +353,7 @@ def dbcheck():
conn = sqlite3.connect(DB_FILE) conn = sqlite3.connect(DB_FILE)
c = conn.cursor() c = conn.cursor()
c.execute( c.execute(
'CREATE TABLE IF NOT EXISTS artists (ArtistID TEXT UNIQUE, ArtistName TEXT, ArtistSortName TEXT, DateAdded TEXT, Status TEXT, IncludeExtras INTEGER, LatestAlbum TEXT, ReleaseDate TEXT, AlbumID TEXT, HaveTracks INTEGER, TotalTracks INTEGER, LastUpdated TEXT, ArtworkURL TEXT, ThumbURL TEXT, Extras TEXT, Type TEXT)') 'CREATE TABLE IF NOT EXISTS artists (ArtistID TEXT UNIQUE, ArtistName TEXT, ArtistSortName TEXT, DateAdded TEXT, Status TEXT, IncludeExtras INTEGER, LatestAlbum TEXT, ReleaseDate TEXT, AlbumID TEXT, HaveTracks INTEGER, TotalTracks INTEGER, LastUpdated TEXT, ArtworkURL TEXT, ThumbURL TEXT, Extras TEXT, Type TEXT, MetaCritic TEXT)')
# ReleaseFormat here means CD,Digital,Vinyl, etc. If using the default # ReleaseFormat here means CD,Digital,Vinyl, etc. If using the default
# Headphones hybrid release, ReleaseID will equal AlbumID (AlbumID is # Headphones hybrid release, ReleaseID will equal AlbumID (AlbumID is
# releasegroup id) # releasegroup id)
@@ -599,6 +599,11 @@ def dbcheck():
except sqlite3.OperationalError: except sqlite3.OperationalError:
c.execute('ALTER TABLE artists ADD COLUMN Type TEXT DEFAULT NULL') c.execute('ALTER TABLE artists ADD COLUMN Type TEXT DEFAULT NULL')
try:
c.execute('SELECT MetaCritic from artists')
except sqlite3.OperationalError:
c.execute('ALTER TABLE artists ADD COLUMN MetaCritic TEXT DEFAULT NULL')
conn.commit() conn.commit()
c.close() c.close()
+9 -3
View File
@@ -21,8 +21,8 @@ import json
cmd_list = ['getIndex', 'getArtist', 'getAlbum', 'getUpcoming', 'getWanted', 'getSnatched', 'getSimilar', 'getHistory', 'getLogs', cmd_list = ['getIndex', 'getArtist', 'getAlbum', 'getUpcoming', 'getWanted', 'getSnatched', 'getSimilar', 'getHistory', 'getLogs',
'findArtist', 'findAlbum', 'addArtist', 'delArtist', 'pauseArtist', 'resumeArtist', 'refreshArtist', 'findArtist', 'findAlbum', 'addArtist', 'delArtist', 'pauseArtist', 'resumeArtist', 'refreshArtist',
'addAlbum', 'queueAlbum', 'unqueueAlbum', 'forceSearch', 'forceProcess', 'forceActiveArtistsUpdate', 'addAlbum', 'queueAlbum', 'unqueueAlbum', 'forceSearch', 'forceProcess', 'forceActiveArtistsUpdate',
'getVersion', 'checkGithub','shutdown', 'restart', 'update', 'getArtistArt', 'getAlbumArt', 'getVersion', 'checkGithub', 'shutdown', 'restart', 'update', 'getArtistArt', 'getAlbumArt',
'getArtistInfo', 'getAlbumInfo', 'getArtistThumb', 'getAlbumThumb', 'getArtistInfo', 'getAlbumInfo', 'getArtistThumb', 'getAlbumThumb', 'clearLogs',
'choose_specific_download', 'download_specific_release'] 'choose_specific_download', 'download_specific_release']
@@ -176,7 +176,13 @@ class Api(object):
return return
def _getLogs(self, **kwargs): def _getLogs(self, **kwargs):
pass self.data = headphones.LOG_LIST
return
def _clearLogs(self, **kwargs):
headphones.LOG_LIST = []
self.data = 'Cleared log'
return
def _findArtist(self, **kwargs): def _findArtist(self, **kwargs):
if 'name' not in kwargs: if 'name' not in kwargs:
+32 -1
View File
@@ -53,6 +53,7 @@ _CONFIG_DEFINITIONS = {
'DELETE_LOSSLESS_FILES': (int, 'General', 1), 'DELETE_LOSSLESS_FILES': (int, 'General', 1),
'DESTINATION_DIR': (str, 'General', ''), 'DESTINATION_DIR': (str, 'General', ''),
'DETECT_BITRATE': (int, 'General', 0), 'DETECT_BITRATE': (int, 'General', 0),
'DO_NOT_PROCESS_UNMATCHED': (int, 'General', 0),
'DOWNLOAD_DIR': (str, 'General', ''), 'DOWNLOAD_DIR': (str, 'General', ''),
'DOWNLOAD_SCAN_INTERVAL': (int, 'General', 5), 'DOWNLOAD_SCAN_INTERVAL': (int, 'General', 5),
'DOWNLOAD_TORRENT_DIR': (str, 'General', ''), 'DOWNLOAD_TORRENT_DIR': (str, 'General', ''),
@@ -81,6 +82,7 @@ _CONFIG_DEFINITIONS = {
'ENCODER_PATH': (str, 'General', ''), 'ENCODER_PATH': (str, 'General', ''),
'EXTRAS': (str, 'General', ''), 'EXTRAS': (str, 'General', ''),
'EXTRA_NEWZNABS': (list, 'Newznab', ''), 'EXTRA_NEWZNABS': (list, 'Newznab', ''),
'EXTRA_TORZNABS': (list, 'Torznab', ''),
'FILE_FORMAT': (str, 'General', 'Track Artist - Album [Year] - Title'), 'FILE_FORMAT': (str, 'General', 'Track Artist - Album [Year] - Title'),
'FILE_PERMISSIONS': (str, 'General', '0644'), 'FILE_PERMISSIONS': (str, 'General', '0644'),
'FILE_UNDERSCORES': (int, 'General', 0), 'FILE_UNDERSCORES': (int, 'General', 0),
@@ -99,7 +101,7 @@ _CONFIG_DEFINITIONS = {
'HPUSER': (str, 'General', ''), 'HPUSER': (str, 'General', ''),
'HTTPS_CERT': (str, 'General', ''), 'HTTPS_CERT': (str, 'General', ''),
'HTTPS_KEY': (str, 'General', ''), 'HTTPS_KEY': (str, 'General', ''),
'HTTP_HOST': (str, 'General', '0.0.0.0'), 'HTTP_HOST': (str, 'General', 'localhost'),
'HTTP_PASSWORD': (str, 'General', ''), 'HTTP_PASSWORD': (str, 'General', ''),
'HTTP_PORT': (int, 'General', 8181), 'HTTP_PORT': (int, 'General', 8181),
'HTTP_PROXY': (int, 'General', 0), 'HTTP_PROXY': (int, 'General', 0),
@@ -154,6 +156,7 @@ _CONFIG_DEFINITIONS = {
'NZBSORG_HASH': (str, 'NZBsorg', ''), 'NZBSORG_HASH': (str, 'NZBsorg', ''),
'NZBSORG_UID': (str, 'NZBsorg', ''), 'NZBSORG_UID': (str, 'NZBsorg', ''),
'NZB_DOWNLOADER': (int, 'General', 0), 'NZB_DOWNLOADER': (int, 'General', 0),
'OFFICIAL_RELEASES_ONLY': (int, 'General', 0),
'OMGWTFNZBS': (int, 'omgwtfnzbs', 0), 'OMGWTFNZBS': (int, 'omgwtfnzbs', 0),
'OMGWTFNZBS_APIKEY': (str, 'omgwtfnzbs', ''), 'OMGWTFNZBS_APIKEY': (str, 'omgwtfnzbs', ''),
'OMGWTFNZBS_UID': (str, 'omgwtfnzbs', ''), 'OMGWTFNZBS_UID': (str, 'omgwtfnzbs', ''),
@@ -175,6 +178,7 @@ _CONFIG_DEFINITIONS = {
'PLEX_SERVER_HOST': (str, 'Plex', ''), 'PLEX_SERVER_HOST': (str, 'Plex', ''),
'PLEX_UPDATE': (int, 'Plex', 0), 'PLEX_UPDATE': (int, 'Plex', 0),
'PLEX_USERNAME': (str, 'Plex', ''), 'PLEX_USERNAME': (str, 'Plex', ''),
'PLEX_TOKEN': (str, 'Plex', ''),
'PREFERRED_BITRATE': (str, 'General', ''), 'PREFERRED_BITRATE': (str, 'General', ''),
'PREFERRED_BITRATE_ALLOW_LOSSLESS': (int, 'General', 0), 'PREFERRED_BITRATE_ALLOW_LOSSLESS': (int, 'General', 0),
'PREFERRED_BITRATE_HIGH_BUFFER': (int, 'General', 0), 'PREFERRED_BITRATE_HIGH_BUFFER': (int, 'General', 0),
@@ -200,6 +204,7 @@ _CONFIG_DEFINITIONS = {
'PUSHOVER_PRIORITY': (int, 'Pushover', 0), 'PUSHOVER_PRIORITY': (int, 'Pushover', 0),
'RENAME_FILES': (int, 'General', 0), 'RENAME_FILES': (int, 'General', 0),
'REPLACE_EXISTING_FOLDERS': (int, 'General', 0), 'REPLACE_EXISTING_FOLDERS': (int, 'General', 0),
'KEEP_ORIGINAL_FOLDER': (int, 'General', 0),
'REQUIRED_WORDS': (str, 'General', ''), 'REQUIRED_WORDS': (str, 'General', ''),
'RUTRACKER': (int, 'Rutracker', 0), 'RUTRACKER': (int, 'Rutracker', 0),
'RUTRACKER_PASSWORD': (str, 'Rutracker', ''), 'RUTRACKER_PASSWORD': (str, 'Rutracker', ''),
@@ -216,6 +221,8 @@ _CONFIG_DEFINITIONS = {
'SONGKICK_ENABLED': (int, 'Songkick', 1), 'SONGKICK_ENABLED': (int, 'Songkick', 1),
'SONGKICK_FILTER_ENABLED': (int, 'Songkick', 0), 'SONGKICK_FILTER_ENABLED': (int, 'Songkick', 0),
'SONGKICK_LOCATION': (str, 'Songkick', ''), 'SONGKICK_LOCATION': (str, 'Songkick', ''),
'STRIKE': (int, 'Strike', 0),
'STRIKE_RATIO': (str, 'Strike', ''),
'SUBSONIC_ENABLED': (int, 'Subsonic', 0), 'SUBSONIC_ENABLED': (int, 'Subsonic', 0),
'SUBSONIC_HOST': (str, 'Subsonic', ''), 'SUBSONIC_HOST': (str, 'Subsonic', ''),
'SUBSONIC_PASSWORD': (str, 'Subsonic', ''), 'SUBSONIC_PASSWORD': (str, 'Subsonic', ''),
@@ -224,6 +231,10 @@ _CONFIG_DEFINITIONS = {
'TORRENTBLACKHOLE_DIR': (str, 'General', ''), 'TORRENTBLACKHOLE_DIR': (str, 'General', ''),
'TORRENT_DOWNLOADER': (int, 'General', 0), 'TORRENT_DOWNLOADER': (int, 'General', 0),
'TORRENT_REMOVAL_INTERVAL': (int, 'General', 720), 'TORRENT_REMOVAL_INTERVAL': (int, 'General', 720),
'TORZNAB': (int, 'Torznab', 0),
'TORZNAB_APIKEY': (str, 'Torznab', ''),
'TORZNAB_ENABLED': (int, 'Torznab', 1),
'TORZNAB_HOST': (str, 'Torznab', ''),
'TRANSMISSION_HOST': (str, 'Transmission', ''), 'TRANSMISSION_HOST': (str, 'Transmission', ''),
'TRANSMISSION_PASSWORD': (str, 'Transmission', ''), 'TRANSMISSION_PASSWORD': (str, 'Transmission', ''),
'TRANSMISSION_USERNAME': (str, 'Transmission', ''), 'TRANSMISSION_USERNAME': (str, 'Transmission', ''),
@@ -239,6 +250,7 @@ _CONFIG_DEFINITIONS = {
'UTORRENT_PASSWORD': (str, 'uTorrent', ''), 'UTORRENT_PASSWORD': (str, 'uTorrent', ''),
'UTORRENT_USERNAME': (str, 'uTorrent', ''), 'UTORRENT_USERNAME': (str, 'uTorrent', ''),
'VERIFY_SSL_CERT': (bool_int, 'Advanced', 1), 'VERIFY_SSL_CERT': (bool_int, 'Advanced', 1),
'WAIT_UNTIL_RELEASE_DATE' : (int, 'General', 0),
'WAFFLES': (int, 'Waffles', 0), 'WAFFLES': (int, 'Waffles', 0),
'WAFFLES_PASSKEY': (str, 'Waffles', ''), 'WAFFLES_PASSKEY': (str, 'Waffles', ''),
'WAFFLES_RATIO': (str, 'Waffles', ''), 'WAFFLES_RATIO': (str, 'Waffles', ''),
@@ -347,6 +359,25 @@ class Config(object):
extra_newznabs.append(item) extra_newznabs.append(item)
self.EXTRA_NEWZNABS = extra_newznabs self.EXTRA_NEWZNABS = extra_newznabs
def get_extra_torznabs(self):
""" Return the extra torznab tuples """
extra_torznabs = list(
itertools.izip(*[itertools.islice(self.EXTRA_TORZNABS, i, None, 3)
for i in range(3)])
)
return extra_torznabs
def clear_extra_torznabs(self):
""" Forget about the configured extra torznabs """
self.EXTRA_TORZNABS = []
def add_extra_torznab(self, torznab):
""" Add a new extra torznab """
extra_torznabs = self.EXTRA_TORZNABS
for item in torznab:
extra_torznabs.append(item)
self.EXTRA_TORZNABS = extra_torznabs
def __getattr__(self, name): def __getattr__(self, name):
""" """
Returns something from the ini unless it is a real property Returns something from the ini unless it is a real property
+1 -1
View File
@@ -149,7 +149,7 @@ def get_age(date):
try: try:
days_old = int(split_date[0]) * 365 + int(split_date[1]) * 30 + int(split_date[2]) days_old = int(split_date[0]) * 365 + int(split_date[1]) * 30 + int(split_date[2])
except IndexError: except (IndexError,ValueError):
days_old = False days_old = False
return days_old return days_old
+1 -1
View File
@@ -496,7 +496,7 @@ def addArtisttoDB(artistid, extrasonly=False, forcefull=False, type="artist"):
cache.getThumb(ArtistID=artistid) cache.getThumb(ArtistID=artistid)
logger.info(u"Fetching Metacritic reviews for: %s" % artist['artist_name']) logger.info(u"Fetching Metacritic reviews for: %s" % artist['artist_name'])
metacritic.update(artist['artist_name'], artist['releasegroups']) metacritic.update(artistid, artist['artist_name'], artist['releasegroups'])
if errors: if errors:
logger.info("[%s] Finished updating artist: %s but with errors, so not marking it as updated in the database" % (artist['artist_name'], artist['artist_name'])) logger.info("[%s] Finished updating artist: %s but with errors, so not marking it as updated in the database" % (artist['artist_name'], artist['artist_name']))
+2 -2
View File
@@ -79,7 +79,7 @@ def getSimilar():
try: try:
artist_mbid = artist["mbid"] artist_mbid = artist["mbid"]
artist_name = artist["name"] artist_name = artist["name"]
except TypeError: except KeyError:
continue continue
if not any(artist_mbid in x for x in results): if not any(artist_mbid in x for x in results):
@@ -116,7 +116,7 @@ def getArtists():
return return
logger.info("Fetching artists from Last.FM for username: %s", headphones.CONFIG.LASTFM_USERNAME) logger.info("Fetching artists from Last.FM for username: %s", headphones.CONFIG.LASTFM_USERNAME)
data = request_lastfm("library.getartists", limit=10000, user=headphones.CONFIG.LASTFM_USERNAME) data = request_lastfm("library.getartists", limit=1000, user=headphones.CONFIG.LASTFM_USERNAME)
if data and "artists" in data: if data and "artists" in data:
artistlist = [] artistlist = []
+8 -6
View File
@@ -23,7 +23,7 @@ from headphones import db, logger, helpers, importer, lastfm
# You can scan a single directory and append it to the current library by # You can scan a single directory and append it to the current library by
# specifying append=True, ArtistID and ArtistName. # specifying append=True, ArtistID and ArtistName.
def libraryScan(dir=None, append=False, ArtistID=None, ArtistName=None, def libraryScan(dir=None, append=False, ArtistID=None, ArtistName=None,
cron=False): cron=False, artistScan=False):
if cron and not headphones.CONFIG.LIBRARYSCAN: if cron and not headphones.CONFIG.LIBRARYSCAN:
return return
@@ -36,7 +36,7 @@ def libraryScan(dir=None, append=False, ArtistID=None, ArtistName=None,
# If we're appending a dir, it's coming from the post processor which is # If we're appending a dir, it's coming from the post processor which is
# already bytestring # already bytestring
if not append: if not append or artistScan:
dir = dir.encode(headphones.SYS_ENCODING) dir = dir.encode(headphones.SYS_ENCODING)
if not os.path.isdir(dir): if not os.path.isdir(dir):
@@ -53,7 +53,7 @@ def libraryScan(dir=None, append=False, ArtistID=None, ArtistName=None,
tracks = myDB.select('SELECT Location from alltracks WHERE Location IS NOT NULL UNION SELECT Location from tracks WHERE Location IS NOT NULL') tracks = myDB.select('SELECT Location from alltracks WHERE Location IS NOT NULL UNION SELECT Location from tracks WHERE Location IS NOT NULL')
for track in tracks: for track in tracks:
encoded_track_string = track['Location'].encode(headphones.SYS_ENCODING) encoded_track_string = track['Location'].encode(headphones.SYS_ENCODING, 'replace')
if not os.path.isfile(encoded_track_string): if not os.path.isfile(encoded_track_string):
myDB.action('UPDATE tracks SET Location=?, BitRate=?, Format=? WHERE Location=?', [None, None, None, track['Location']]) myDB.action('UPDATE tracks SET Location=?, BitRate=?, Format=? WHERE Location=?', [None, None, None, track['Location']])
myDB.action('UPDATE alltracks SET Location=?, BitRate=?, Format=? WHERE Location=?', [None, None, None, track['Location']]) myDB.action('UPDATE alltracks SET Location=?, BitRate=?, Format=? WHERE Location=?', [None, None, None, track['Location']])
@@ -287,7 +287,7 @@ def libraryScan(dir=None, append=False, ArtistID=None, ArtistName=None,
logger.info('Completed matching tracks from directory: %s' % dir.decode(headphones.SYS_ENCODING, 'replace')) logger.info('Completed matching tracks from directory: %s' % dir.decode(headphones.SYS_ENCODING, 'replace'))
if not append: if not append or artistScan:
logger.info('Updating scanned artist track counts') logger.info('Updating scanned artist track counts')
# Clean up the new artist list # Clean up the new artist list
@@ -334,7 +334,7 @@ def libraryScan(dir=None, append=False, ArtistID=None, ArtistName=None,
for artist in artist_list: for artist in artist_list:
myDB.action('INSERT OR IGNORE INTO newartists VALUES (?)', [artist]) myDB.action('INSERT OR IGNORE INTO newartists VALUES (?)', [artist])
if headphones.CONFIG.DETECT_BITRATE: if headphones.CONFIG.DETECT_BITRATE and bitrates:
headphones.CONFIG.PREFERRED_BITRATE = sum(bitrates) / len(bitrates) / 1000 headphones.CONFIG.PREFERRED_BITRATE = sum(bitrates) / len(bitrates) / 1000
else: else:
@@ -346,6 +346,8 @@ def libraryScan(dir=None, append=False, ArtistID=None, ArtistName=None,
if not append: if not append:
update_album_status() update_album_status()
if not append and not artistScan:
lastfm.getSimilar() lastfm.getSimilar()
logger.info('Library scan complete') logger.info('Library scan complete')
@@ -374,7 +376,7 @@ def update_album_status(AlbumID=None):
album_completion = 0 album_completion = 0
logger.info('Album %s does not have any tracks in database' % album['AlbumTitle']) logger.info('Album %s does not have any tracks in database' % album['AlbumTitle'])
if album_completion >= headphones.CONFIG.ALBUM_COMPLETION_PCT and album['Status'] == 'Skipped': if album_completion >= headphones.CONFIG.ALBUM_COMPLETION_PCT:
new_album_status = "Downloaded" new_album_status = "Downloaded"
# I don't think we want to change Downloaded->Skipped..... # I don't think we want to change Downloaded->Skipped.....
+7 -5
View File
@@ -51,7 +51,7 @@ def startmb():
mbpass = headphones.CONFIG.CUSTOMPASS mbpass = headphones.CONFIG.CUSTOMPASS
sleepytime = int(headphones.CONFIG.CUSTOMSLEEP) sleepytime = int(headphones.CONFIG.CUSTOMSLEEP)
elif headphones.CONFIG.MIRROR == "headphones": elif headphones.CONFIG.MIRROR == "headphones":
mbhost = "144.76.94.239" mbhost = "codeshy.com"
mbport = 8181 mbport = 8181
mbuser = headphones.CONFIG.HPUSER mbuser = headphones.CONFIG.HPUSER
mbpass = headphones.CONFIG.HPPASS mbpass = headphones.CONFIG.HPPASS
@@ -466,6 +466,11 @@ def get_new_releases(rgid, includeExtras=False, forcefull=False):
myDB = db.DBConnection() myDB = db.DBConnection()
results = [] results = []
release_status = "official"
if includeExtras and not headphones.CONFIG.OFFICIAL_RELEASES_ONLY:
release_status = []
try: try:
limit = 100 limit = 100
newResults = None newResults = None
@@ -474,6 +479,7 @@ def get_new_releases(rgid, includeExtras=False, forcefull=False):
newResults = musicbrainzngs.browse_releases( newResults = musicbrainzngs.browse_releases(
release_group=rgid, release_group=rgid,
includes=['artist-credits', 'labels', 'recordings', 'release-groups', 'media'], includes=['artist-credits', 'labels', 'recordings', 'release-groups', 'media'],
release_status = release_status,
limit=limit, limit=limit,
offset=len(results)) offset=len(results))
if 'release-list' not in newResults: if 'release-list' not in newResults:
@@ -513,10 +519,6 @@ def get_new_releases(rgid, includeExtras=False, forcefull=False):
num_new_releases = 0 num_new_releases = 0
for releasedata in results: for releasedata in results:
#releasedata.get will return None if it doesn't have a status
#all official releases should have the Official status included
if not includeExtras and releasedata.get('status') != 'Official':
continue
release = {} release = {}
rel_id_check = releasedata['id'] rel_id_check = releasedata['id']
+42 -12
View File
@@ -14,11 +14,13 @@
# along with Headphones. If not, see <http://www.gnu.org/licenses/>. # along with Headphones. If not, see <http://www.gnu.org/licenses/>.
import re import re
import json
import headphones import headphones
from headphones import db, helpers, logger, request from headphones import db, helpers, logger, request
from headphones.common import USER_AGENT
def update(artist_name,release_groups): def update(artistid, artist_name,release_groups):
""" Pretty simple and crude function to find the artist page on metacritic, """ Pretty simple and crude function to find the artist page on metacritic,
then parse that page to get critic & user scores for albums""" then parse that page to get critic & user scores for albums"""
@@ -31,28 +33,56 @@ def update(artist_name,release_groups):
mc_artist_name = mc_artist_name.replace(" ","-") mc_artist_name = mc_artist_name.replace(" ","-")
headers = {'User-Agent': 'Mozilla/5.0 (Windows NT 6.3; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/41.0.2243.2 Safari/537.36'}
url = "http://www.metacritic.com/person/" + mc_artist_name + "?filter-options=music&sort_options=date&num_items=100" url = "http://www.metacritic.com/person/" + mc_artist_name + "?filter-options=music&sort_options=date&num_items=100"
res = request.request_soup(url, parser='html.parser') res = request.request_soup(url, headers=headers)
rows = None
try: try:
rows = res.tbody.find_all('tr') table = res.find("table", class_="credits person_credits")
rows = table.tbody.find_all('tr')
except: except:
logger.info("Unable to get metacritic scores for: %s" % artist_name) logger.info("Unable to get metacritic scores for: %s" % artist_name)
return
myDB = db.DBConnection() myDB = db.DBConnection()
artist = myDB.action('SELECT * FROM artists WHERE ArtistID=?', [artistid]).fetchone()
for row in rows: score_list = []
title = row.a.string
# If we couldn't get anything from MetaCritic for whatever reason,
# let's try to load scores from the db
if not rows:
if artist['MetaCritic']:
score_list = json.loads(artist['MetaCritic'])
else:
return
# If we did get scores, let's update the db with them
else:
for row in rows:
title = row.a.string
scores = row.find_all("span")
critic_score = scores[0].string
user_score = scores[1].string
score_dict = {'title':title,'critic_score':critic_score,'user_score':user_score}
score_list.append(score_dict)
# Save scores to the database
controlValueDict = {"ArtistID": artistid}
newValueDict = {'MetaCritic':json.dumps(score_list)}
myDB.upsert("artists", newValueDict, controlValueDict)
for score in score_list:
title = score['title']
# Iterate through the release groups we got passed to see if we can find
# a match
for rg in release_groups: for rg in release_groups:
if rg['title'].lower() == title.lower(): if rg['title'].lower() == title.lower():
scores = row.find_all("span") critic_score = score['critic_score']
critic_score = scores[0].string user_score = score['user_score']
user_score = scores[1].string
controlValueDict = {"AlbumID": rg['id']} controlValueDict = {"AlbumID": rg['id']}
newValueDict = {'CriticScore':critic_score,'UserScore':user_score} newValueDict = {'CriticScore':critic_score,'UserScore':user_score}
myDB.upsert("albums", newValueDict, controlValueDict) myDB.upsert("albums", newValueDict, controlValueDict)
+65 -93
View File
@@ -316,34 +316,32 @@ class Plex(object):
self.client_hosts = headphones.CONFIG.PLEX_CLIENT_HOST self.client_hosts = headphones.CONFIG.PLEX_CLIENT_HOST
self.username = headphones.CONFIG.PLEX_USERNAME self.username = headphones.CONFIG.PLEX_USERNAME
self.password = headphones.CONFIG.PLEX_PASSWORD self.password = headphones.CONFIG.PLEX_PASSWORD
self.token = headphones.CONFIG.PLEX_TOKEN
def _sendhttp(self, host, command): def _sendhttp(self, host, command):
username = self.username url = host + '/xbmcCmds/xbmcHttp/?' + command
password = self.password
url_command = urllib.urlencode(command) if self.password:
response = request.request_response(url, auth=(self.username, self.password))
url = host + '/xbmcCmds/xbmcHttp/?' + url_command else:
response = request.request_response(url)
req = urllib2.Request(url)
if password:
base64string = base64.encodestring('%s:%s' % (username, password)).replace('\n', '')
req.add_header("Authorization", "Basic %s" % base64string)
logger.info('Plex url: %s' % url)
try:
handle = urllib2.urlopen(req)
except Exception as e:
logger.warn('Error opening Plex url: %s' % e)
return
response = handle.read().decode(headphones.SYS_ENCODING)
return response return response
def _sendjson(self, host, method, params={}):
data = [{'id': 0, 'jsonrpc': '2.0', 'method': method, 'params': params}]
headers = {'Content-Type': 'application/json'}
url = host + '/jsonrpc'
if self.password:
response = request.request_json(url, method="post", data=json.dumps(data), headers=headers, auth=(self.username, self.password))
else:
response = request.request_json(url, method="post", data=json.dumps(data), headers=headers)
if response:
return response[0]['result']
def update(self): def update(self):
# From what I read you can't update the music library on a per directory or per path basis # From what I read you can't update the music library on a per directory or per path basis
@@ -354,13 +352,15 @@ class Plex(object):
for host in hosts: for host in hosts:
logger.info('Sending library update command to Plex Media Server@ ' + host) logger.info('Sending library update command to Plex Media Server@ ' + host)
url = "%s/library/sections" % host url = "%s/library/sections" % host
try: if self.token:
xml_sections = minidom.parse(urllib.urlopen(url)) params = {'X-Plex-Token': self.token}
except IOError, e: else:
logger.warn("Error while trying to contact Plex Media Server: %s" % e) params = False
return False
r = request.request_minidom(url, params=params)
sections = r.getElementsByTagName('Directory')
sections = xml_sections.getElementsByTagName('Directory')
if not sections: if not sections:
logger.info(u"Plex Media Server not running on: " + host) logger.info(u"Plex Media Server not running on: " + host)
return False return False
@@ -368,11 +368,7 @@ class Plex(object):
for s in sections: for s in sections:
if s.getAttribute('type') == "artist": if s.getAttribute('type') == "artist":
url = "%s/library/sections/%s/refresh" % (host, s.getAttribute('key')) url = "%s/library/sections/%s/refresh" % (host, s.getAttribute('key'))
try: request.request_response(url, params=params)
urllib.urlopen(url)
except Exception as e:
logger.warn("Error updating library section for Plex Media Server: %s" % e)
return False
def notify(self, artist, album, albumartpath): def notify(self, artist, album, albumartpath):
@@ -383,17 +379,24 @@ class Plex(object):
time = "3000" # in ms time = "3000" # in ms
for host in hosts: for host in hosts:
logger.info('Sending notification command to Plex Media Server @ ' + host) logger.info('Sending notification command to Plex client @ ' + host)
try: try:
notification = header + "," + message + "," + time + "," + albumartpath version = self._sendjson(host, 'Application.GetProperties', {'properties': ['version']})['version']['major']
notifycommand = {'command': 'ExecBuiltIn', 'parameter': 'Notification(' + notification + ')'}
request = self._sendhttp(host, notifycommand) if version < 12: #Eden
notification = header + "," + message + "," + time + "," + albumartpath
notifycommand = {'command': 'ExecBuiltIn', 'parameter': 'Notification(' + notification + ')'}
request = self._sendhttp(host, notifycommand)
else: #Frodo
params = {'title': header, 'message': message, 'displaytime': int(time), 'image': albumartpath}
request = self._sendjson(host, 'GUI.ShowNotification', params)
if not request: if not request:
raise Exception raise Exception
except: except Exception:
logger.warn('Error sending notification request to Plex Media Server') logger.error('Error sending notification request to Plex client @ ' + host)
class NMA(object): class NMA(object):
@@ -440,52 +443,30 @@ class PUSHBULLET(object):
self.apikey = headphones.CONFIG.PUSHBULLET_APIKEY self.apikey = headphones.CONFIG.PUSHBULLET_APIKEY
self.deviceid = headphones.CONFIG.PUSHBULLET_DEVICEID self.deviceid = headphones.CONFIG.PUSHBULLET_DEVICEID
def conf(self, options): def notify(self, message, status):
return cherrypy.config['config'].get('PUSHBULLET', options)
def notify(self, message, event):
if not headphones.CONFIG.PUSHBULLET_ENABLED: if not headphones.CONFIG.PUSHBULLET_ENABLED:
return return
http_handler = HTTPSConnection("api.pushbullet.com") url = "https://api.pushbullet.com/v2/pushes"
data = {'type': "note", data = {'type': "note",
'title': "Headphones", 'title': "Headphones",
'body': message.encode("utf-8")} 'body': message + ': ' + status}
http_handler.request("POST", if self.deviceid:
"/v2/pushes", data['device_iden'] = self.deviceid
headers={'Content-type': "application/json",
'Authorization': 'Basic %s' % base64.b64encode(headphones.CONFIG.PUSHBULLET_APIKEY + ":")},
body=json.dumps(data))
response = http_handler.getresponse()
request_status = response.status
logger.debug(u"PushBullet response status: %r" % request_status)
logger.debug(u"PushBullet response headers: %r" % response.getheaders())
logger.debug(u"PushBullet response body: %r" % response.read())
if request_status == 200: headers={'Content-type': "application/json",
logger.info(u"PushBullet notifications sent.") 'Authorization': 'Bearer ' + headphones.CONFIG.PUSHBULLET_APIKEY}
return True
elif request_status >= 400 and request_status < 500: response = request.request_json(url, method="post", headers=headers, data=json.dumps(data))
logger.info(u"PushBullet request failed: %s" % response.reason)
return False if response:
logger.info(u"PushBullet notifications sent.")
return True
else: else:
logger.info(u"PushBullet notification failed serverside.") logger.info(u"PushBullet notification failed.")
return False return False
def updateLibrary(self):
#For uniformity reasons not removed
return
def test(self, apikey, deviceid):
self.enabled = True
self.apikey = apikey
self.deviceid = deviceid
self.notify('Main Screen Activate', 'Test Message')
class PUSHALOT(object): class PUSHALOT(object):
@@ -583,7 +564,7 @@ class PUSHOVER(object):
if not headphones.CONFIG.PUSHOVER_ENABLED: if not headphones.CONFIG.PUSHOVER_ENABLED:
return return
http_handler = HTTPSConnection("api.pushover.net") url = "https://api.pushover.net/1/messages.json"
data = {'token': self.application_token, data = {'token': self.application_token,
'user': headphones.CONFIG.PUSHOVER_KEYS, 'user': headphones.CONFIG.PUSHOVER_KEYS,
@@ -591,25 +572,16 @@ class PUSHOVER(object):
'message': message.encode("utf-8"), 'message': message.encode("utf-8"),
'priority': headphones.CONFIG.PUSHOVER_PRIORITY} 'priority': headphones.CONFIG.PUSHOVER_PRIORITY}
http_handler.request("POST", headers = {'Content-type': "application/x-www-form-urlencoded"}
"/1/messages.json",
headers={'Content-type': "application/x-www-form-urlencoded"},
body=urlencode(data))
response = http_handler.getresponse()
request_status = response.status
logger.debug(u"Pushover response status: %r" % request_status)
logger.debug(u"Pushover response headers: %r" % response.getheaders())
logger.debug(u"Pushover response body: %r" % response.read())
if request_status == 200: response = request.request_response(url, method="POST", headers=headers, data=data)
logger.info(u"Pushover notifications sent.")
return True if response:
elif request_status >= 400 and request_status < 500: logger.info(u"Pushover notifications sent.")
logger.info(u"Pushover request failed: %s" % response.reason) return True
return False
else: else:
logger.info(u"Pushover notification failed.") logger.error(u"Pushover notification failed.")
return False return False
def updateLibrary(self): def updateLibrary(self):
#For uniformity reasons not removed #For uniformity reasons not removed
+25 -22
View File
@@ -59,7 +59,7 @@ def checkFolder():
logger.debug("Checking download folder finished.") logger.debug("Checking download folder finished.")
def verify(albumid, albumpath, Kind=None, forced=False): def verify(albumid, albumpath, Kind=None, forced=False, keep_original_folder=False):
myDB = db.DBConnection() myDB = db.DBConnection()
release = myDB.action('SELECT * from albums WHERE AlbumID=?', [albumid]).fetchone() release = myDB.action('SELECT * from albums WHERE AlbumID=?', [albumid]).fetchone()
@@ -216,7 +216,7 @@ def verify(albumid, albumpath, Kind=None, forced=False):
logger.debug('Matching metadata album: %s with album name: %s' % (metaalbum, dbalbum)) logger.debug('Matching metadata album: %s with album name: %s' % (metaalbum, dbalbum))
if metaartist == dbartist and metaalbum == dbalbum: if metaartist == dbartist and metaalbum == dbalbum:
doPostProcessing(albumid, albumpath, release, tracks, downloaded_track_list, Kind) doPostProcessing(albumid, albumpath, release, tracks, downloaded_track_list, Kind, keep_original_folder)
return return
# test #2: filenames # test #2: filenames
@@ -234,7 +234,7 @@ def verify(albumid, albumpath, Kind=None, forced=False):
logger.debug('Checking if track title: %s is in file name: %s' % (dbtrack, filetrack)) logger.debug('Checking if track title: %s is in file name: %s' % (dbtrack, filetrack))
if dbtrack in filetrack: if dbtrack in filetrack:
doPostProcessing(albumid, albumpath, release, tracks, downloaded_track_list, Kind) doPostProcessing(albumid, albumpath, release, tracks, downloaded_track_list, Kind, keep_original_folder)
return return
# test #3: number of songs and duration # test #3: number of songs and duration
@@ -266,7 +266,7 @@ def verify(albumid, albumpath, Kind=None, forced=False):
logger.debug('Database track duration: %i' % db_track_duration) logger.debug('Database track duration: %i' % db_track_duration)
delta = abs(downloaded_track_duration - db_track_duration) delta = abs(downloaded_track_duration - db_track_duration)
if delta < 240: if delta < 240:
doPostProcessing(albumid, albumpath, release, tracks, downloaded_track_list, Kind) doPostProcessing(albumid, albumpath, release, tracks, downloaded_track_list, Kind, keep_original_folder)
return return
logger.warn(u'Could not identify album: %s. It may not be the intended album.' % albumpath.decode(headphones.SYS_ENCODING, 'replace')) logger.warn(u'Could not identify album: %s. It may not be the intended album.' % albumpath.decode(headphones.SYS_ENCODING, 'replace'))
@@ -276,11 +276,11 @@ def verify(albumid, albumpath, Kind=None, forced=False):
renameUnprocessedFolder(albumpath, tag="Unprocessed") renameUnprocessedFolder(albumpath, tag="Unprocessed")
def doPostProcessing(albumid, albumpath, release, tracks, downloaded_track_list, Kind=None): def doPostProcessing(albumid, albumpath, release, tracks, downloaded_track_list, Kind=None, keep_original_folder=False):
logger.info('Starting post-processing for: %s - %s' % (release['ArtistName'], release['AlbumTitle'])) logger.info('Starting post-processing for: %s - %s' % (release['ArtistName'], release['AlbumTitle']))
# Check to see if we're preserving the torrent dir # Check to see if we're preserving the torrent dir
if headphones.CONFIG.KEEP_TORRENT_FILES and Kind == "torrent" and 'headphones-modified' not in albumpath: if (headphones.CONFIG.KEEP_TORRENT_FILES and Kind == "torrent" and 'headphones-modified' not in albumpath) or headphones.CONFIG.KEEP_ORIGINAL_FOLDER or keep_original_folder:
new_folder = os.path.join(albumpath, 'headphones-modified'.encode(headphones.SYS_ENCODING, 'replace')) new_folder = os.path.join(albumpath, 'headphones-modified'.encode(headphones.SYS_ENCODING, 'replace'))
logger.info("Copying files to 'headphones-modified' subfolder to preserve downloaded files for seeding") logger.info("Copying files to 'headphones-modified' subfolder to preserve downloaded files for seeding")
try: try:
@@ -372,7 +372,9 @@ def doPostProcessing(albumid, albumpath, release, tracks, downloaded_track_list,
addAlbumArt(artwork, albumpath, release) addAlbumArt(artwork, albumpath, release)
if headphones.CONFIG.CORRECT_METADATA: if headphones.CONFIG.CORRECT_METADATA:
correctMetadata(albumid, release, downloaded_track_list) correctedMetadata = correctMetadata(albumid, release, downloaded_track_list)
if not correctedMetadata and headphones.CONFIG.DO_NOT_PROCESS_UNMATCHED:
return
if headphones.CONFIG.EMBED_LYRICS: if headphones.CONFIG.EMBED_LYRICS:
embedLyrics(downloaded_track_list) embedLyrics(downloaded_track_list)
@@ -475,7 +477,7 @@ def doPostProcessing(albumid, albumpath, release, tracks, downloaded_track_list,
if headphones.CONFIG.PUSHBULLET_ENABLED: if headphones.CONFIG.PUSHBULLET_ENABLED:
logger.info(u"PushBullet request") logger.info(u"PushBullet request")
pushbullet = notifiers.PUSHBULLET() pushbullet = notifiers.PUSHBULLET()
pushbullet.notify(pushmessage, "Download and Postprocessing completed") pushbullet.notify(pushmessage, statusmessage)
if headphones.CONFIG.TWITTER_ENABLED: if headphones.CONFIG.TWITTER_ENABLED:
logger.info(u"Sending Twitter notification") logger.info(u"Sending Twitter notification")
@@ -595,7 +597,6 @@ def renameNFO(albumpath):
except Exception as e: except Exception as e:
logger.error(u'Could not rename file: %s. Error: %s' % (os.path.join(r, file).decode(headphones.SYS_ENCODING, 'replace'), e)) logger.error(u'Could not rename file: %s. Error: %s' % (os.path.join(r, file).decode(headphones.SYS_ENCODING, 'replace'), e))
def moveFiles(albumpath, release, tracks): def moveFiles(albumpath, release, tracks):
logger.info("Moving files: %s" % albumpath) logger.info("Moving files: %s" % albumpath)
try: try:
@@ -858,16 +859,16 @@ def correctMetadata(albumid, release, downloaded_track_list):
cur_artist, cur_album, candidates, rec = autotag.tag_album(items, search_artist=helpers.latinToAscii(release['ArtistName']), search_album=helpers.latinToAscii(release['AlbumTitle'])) cur_artist, cur_album, candidates, rec = autotag.tag_album(items, search_artist=helpers.latinToAscii(release['ArtistName']), search_album=helpers.latinToAscii(release['AlbumTitle']))
except Exception as e: except Exception as e:
logger.error('Error getting recommendation: %s. Not writing metadata', e) logger.error('Error getting recommendation: %s. Not writing metadata', e)
return return False
if str(rec) == 'Recommendation.none': if str(rec) == 'Recommendation.none':
logger.warn('No accurate album match found for %s, %s - not writing metadata', release['ArtistName'], release['AlbumTitle']) logger.warn('No accurate album match found for %s, %s - not writing metadata', release['ArtistName'], release['AlbumTitle'])
return return False
if candidates: if candidates:
dist, info, mapping, extra_items, extra_tracks = candidates[0] dist, info, mapping, extra_items, extra_tracks = candidates[0]
else: else:
logger.warn('No accurate album match found for %s, %s - not writing metadata', release['ArtistName'], release['AlbumTitle']) logger.warn('No accurate album match found for %s, %s - not writing metadata', release['ArtistName'], release['AlbumTitle'])
return return False
logger.info('Beets recommendation for tagging items: %s' % rec) logger.info('Beets recommendation for tagging items: %s' % rec)
@@ -889,7 +890,9 @@ def correctMetadata(albumid, release, downloaded_track_list):
logger.info("Successfully applied metadata to: %s", item.path.decode(headphones.SYS_ENCODING, 'replace')) logger.info("Successfully applied metadata to: %s", item.path.decode(headphones.SYS_ENCODING, 'replace'))
except Exception as e: except Exception as e:
logger.warn("Error writing metadata to '%s': %s", item.path.decode(headphones.SYS_ENCODING, 'replace'), str(e)) logger.warn("Error writing metadata to '%s': %s", item.path.decode(headphones.SYS_ENCODING, 'replace'), str(e))
return False
return True
def embedLyrics(downloaded_track_list): def embedLyrics(downloaded_track_list):
logger.info('Adding lyrics') logger.info('Adding lyrics')
@@ -1060,7 +1063,7 @@ def renameUnprocessedFolder(path, tag):
return return
def forcePostProcess(dir=None, expand_subfolders=True, album_dir=None): def forcePostProcess(dir=None, expand_subfolders=True, album_dir=None, keep_original_folder=False):
logger.info('Force checking download folder for completed downloads') logger.info('Force checking download folder for completed downloads')
@@ -1136,7 +1139,7 @@ def forcePostProcess(dir=None, expand_subfolders=True, album_dir=None):
continue continue
else: else:
logger.info('Found a match in the database: %s. Verifying to make sure it is the correct album', snatched['Title']) logger.info('Found a match in the database: %s. Verifying to make sure it is the correct album', snatched['Title'])
verify(snatched['AlbumID'], folder, snatched['Kind']) verify(snatched['AlbumID'], folder, snatched['Kind'], keep_original_folder=keep_original_folder)
continue continue
# Attempt 2: strip release group id from filename # Attempt 2: strip release group id from filename
@@ -1153,10 +1156,10 @@ def forcePostProcess(dir=None, expand_subfolders=True, album_dir=None):
release = myDB.action('SELECT ArtistName, AlbumTitle, AlbumID from albums WHERE AlbumID=?', [rgid]).fetchone() release = myDB.action('SELECT ArtistName, AlbumTitle, AlbumID from albums WHERE AlbumID=?', [rgid]).fetchone()
if release: if release:
logger.info('Found a match in the database: %s - %s. Verifying to make sure it is the correct album', release['ArtistName'], release['AlbumTitle']) logger.info('Found a match in the database: %s - %s. Verifying to make sure it is the correct album', release['ArtistName'], release['AlbumTitle'])
verify(release['AlbumID'], folder, forced=True) verify(release['AlbumID'], folder, forced=True, keep_original_folder=keep_original_folder)
continue continue
else: else:
logger.info('Found a (possibly) valid Musicbrainz realse group id in album folder name.') logger.info('Found a (possibly) valid Musicbrainz release group id in album folder name.')
verify(rgid, folder, forced=True) verify(rgid, folder, forced=True)
continue continue
@@ -1172,7 +1175,7 @@ def forcePostProcess(dir=None, expand_subfolders=True, album_dir=None):
release = myDB.action('SELECT AlbumID, ArtistName, AlbumTitle from albums WHERE ArtistName LIKE ? and AlbumTitle LIKE ?', [name, album]).fetchone() release = myDB.action('SELECT AlbumID, ArtistName, AlbumTitle from albums WHERE ArtistName LIKE ? and AlbumTitle LIKE ?', [name, album]).fetchone()
if release: if release:
logger.info('Found a match in the database: %s - %s. Verifying to make sure it is the correct album', release['ArtistName'], release['AlbumTitle']) logger.info('Found a match in the database: %s - %s. Verifying to make sure it is the correct album', release['ArtistName'], release['AlbumTitle'])
verify(release['AlbumID'], folder) verify(release['AlbumID'], folder, keep_original_folder=keep_original_folder)
continue continue
else: else:
logger.info('Querying MusicBrainz for the release group id for: %s - %s', name, album) logger.info('Querying MusicBrainz for the release group id for: %s - %s', name, album)
@@ -1183,7 +1186,7 @@ def forcePostProcess(dir=None, expand_subfolders=True, album_dir=None):
rgid = None rgid = None
if rgid: if rgid:
verify(rgid, folder) verify(rgid, folder, keep_original_folder=keep_original_folder)
continue continue
else: else:
logger.info('No match found on MusicBrainz for: %s - %s', name, album) logger.info('No match found on MusicBrainz for: %s - %s', name, album)
@@ -1207,7 +1210,7 @@ def forcePostProcess(dir=None, expand_subfolders=True, album_dir=None):
release = myDB.action('SELECT AlbumID, ArtistName, AlbumTitle from albums WHERE ArtistName LIKE ? and AlbumTitle LIKE ?', [name, album]).fetchone() release = myDB.action('SELECT AlbumID, ArtistName, AlbumTitle from albums WHERE ArtistName LIKE ? and AlbumTitle LIKE ?', [name, album]).fetchone()
if release: if release:
logger.info('Found a match in the database: %s - %s. Verifying to make sure it is the correct album', release['ArtistName'], release['AlbumTitle']) logger.info('Found a match in the database: %s - %s. Verifying to make sure it is the correct album', release['ArtistName'], release['AlbumTitle'])
verify(release['AlbumID'], folder) verify(release['AlbumID'], folder, keep_original_folder=keep_original_folder)
continue continue
else: else:
logger.info('Querying MusicBrainz for the release group id for: %s - %s', name, album) logger.info('Querying MusicBrainz for the release group id for: %s - %s', name, album)
@@ -1218,7 +1221,7 @@ def forcePostProcess(dir=None, expand_subfolders=True, album_dir=None):
rgid = None rgid = None
if rgid: if rgid:
verify(rgid, folder) verify(rgid, folder, keep_original_folder=keep_original_folder)
continue continue
else: else:
logger.info('No match found on MusicBrainz for: %s - %s', name, album) logger.info('No match found on MusicBrainz for: %s - %s', name, album)
@@ -1231,7 +1234,7 @@ def forcePostProcess(dir=None, expand_subfolders=True, album_dir=None):
release = myDB.action('SELECT AlbumID, ArtistName, AlbumTitle from albums WHERE AlbumTitle LIKE ?', [folder_basename]).fetchone() release = myDB.action('SELECT AlbumID, ArtistName, AlbumTitle from albums WHERE AlbumTitle LIKE ?', [folder_basename]).fetchone()
if release: if release:
logger.info('Found a match in the database: %s - %s. Verifying to make sure it is the correct album', release['ArtistName'], release['AlbumTitle']) logger.info('Found a match in the database: %s - %s. Verifying to make sure it is the correct album', release['ArtistName'], release['AlbumTitle'])
verify(release['AlbumID'], folder) verify(release['AlbumID'], folder, keep_original_folder=keep_original_folder)
continue continue
else: else:
logger.info('Querying MusicBrainz for the release group id for: %s', folder_basename) logger.info('Querying MusicBrainz for the release group id for: %s', folder_basename)
@@ -1242,7 +1245,7 @@ def forcePostProcess(dir=None, expand_subfolders=True, album_dir=None):
rgid = None rgid = None
if rgid: if rgid:
verify(rgid, folder) verify(rgid, folder, keep_original_folder=keep_original_folder)
continue continue
else: else:
logger.info('No match found on MusicBrainz for: %s - %s', name, album) logger.info('No match found on MusicBrainz for: %s - %s', name, album)
+1 -1
View File
@@ -206,7 +206,7 @@ def server_message(response):
message = None message = None
# First attempt is to 'read' the response as HTML # First attempt is to 'read' the response as HTML
if "text/html" in response.headers.get("content-type"): if response.headers.get("content-type") and "text/html" in response.headers.get("content-type"):
try: try:
soup = BeautifulSoup(response.content, "html5lib") soup = BeautifulSoup(response.content, "html5lib")
except Exception: except Exception:
+224
View File
@@ -0,0 +1,224 @@
#!/usr/bin/env python
import urllib
import requests as requests
from urlparse import urlparse
from bs4 import BeautifulSoup
import os
import time
import re
import headphones
from headphones import logger
class Rutracker(object):
def __init__(self):
self.session = requests.session()
self.timeout = 60
self.loggedin = False
self.maxsize = 0
self.search_referer = 'http://rutracker.org/forum/tracker.php'
def logged_in(self):
return self.loggedin
def still_logged_in(self, html):
if not html or "action=\"http://login.rutracker.org/forum/login.php\">" in html:
return False
else:
return True
def login(self):
"""
Logs in user
"""
loginpage = 'http://login.rutracker.org/forum/login.php'
post_params = {
'login_username': headphones.CONFIG.RUTRACKER_USER,
'login_password': headphones.CONFIG.RUTRACKER_PASSWORD,
'login': b'\xc2\xf5\xee\xe4' # '%C2%F5%EE%E4'
}
logger.info("Attempting to log in to rutracker...")
try:
r = self.session.post(loginpage, data=post_params, timeout=self.timeout)
# try again
if 'bb_data' not in r.cookies.keys():
time.sleep(10)
r = self.session.post(loginpage, data=post_params, timeout=self.timeout)
if r.status_code != 200:
logger.error("rutracker login returned status code %s" % r.status_code)
self.loggedin = False
else:
if 'bb_data' in r.cookies.keys():
self.loggedin = True
logger.info("Successfully logged in to rutracker")
else:
logger.error("Could not login to rutracker, credentials maybe incorrect, site is down or too many attempts. Try again later")
self.loggedin = False
return self.loggedin
except Exception as e:
logger.error("Unknown error logging in to rutracker: %s" % e)
self.loggedin = False
return self.loggedin
def searchurl(self, artist, album, year, format):
"""
Return the search url
"""
# Build search url
searchterm = ''
if artist != 'Various Artists':
searchterm = artist
searchterm = searchterm + ' '
searchterm = searchterm + album
searchterm = searchterm + ' '
searchterm = searchterm + year
if format == 'lossless':
format = '+lossless'
self.maxsize = 10000000000
elif format == 'lossless+mp3':
format = '+lossless||mp3||aac'
self.maxsize = 10000000000
else:
format = '+mp3||aac'
self.maxsize = 300000000
# sort by size, descending.
sort = '&o=7&s=2'
searchurl = "%s?nm=%s%s%s" % (self.search_referer, urllib.quote(searchterm), format, sort)
logger.info("Searching rutracker using term: %s", searchterm)
return searchurl
def search(self, searchurl):
"""
Parse the search results and return valid torrent list
"""
try:
headers = {'Referer': self.search_referer}
r = self.session.get(url=searchurl, headers=headers, timeout=self.timeout)
soup = BeautifulSoup(r.content, 'html5lib')
# Debug
#logger.debug (soup.prettify())
# Check if still logged in
if not self.still_logged_in(soup):
self.login()
r = self.session.get(url=searchurl, timeout=self.timeout)
soup = BeautifulSoup(r.content, 'html5lib')
if not self.still_logged_in(soup):
logger.error("Error getting rutracker data")
return None
# Process
rulist = []
i = soup.find('table', id='tor-tbl')
if not i:
logger.info("No valid results found from rutracker")
return None
minimumseeders = int(headphones.CONFIG.NUMBEROFSEEDERS) - 1
for item in zip(i.find_all(class_='hl-tags'),i.find_all(class_='dl-stub'),i.find_all(class_='seedmed')):
title = item[0].get_text()
url = item[1].get('href')
size_formatted = item[1].get_text()[:-2]
seeds = item[2].get_text()
size_parts = size_formatted.split()
size = float(size_parts[0])
if size_parts[1] == 'KB':
size *= 1024
if size_parts[1] == 'MB':
size *= 1024 ** 2
if size_parts[1] == 'GB':
size *= 1024 ** 3
if size_parts[1] == 'TB':
size *= 1024 ** 4
if size < self.maxsize and minimumseeders < int(seeds):
logger.info('Found %s. Size: %s' % (title, size_formatted))
#Torrent topic page
torrent_id = dict([part.split('=') for part in urlparse(url)[4].split('&')])['t']
topicurl = 'http://rutracker.org/forum/viewtopic.php?t=' + torrent_id
rulist.append((title, size, topicurl, 'rutracker.org', 'torrent', True))
else:
logger.info("%s is larger than the maxsize or has too little seeders for this category, skipping. (Size: %i bytes, Seeders: %i)" % (title, size, int(seeds)))
if not rulist:
logger.info("No valid results found from rutracker")
return rulist
except Exception as e:
logger.error("An unknown error occurred in the rutracker parser: %s" % e)
return None
def get_torrent_data(self, url):
"""
return the .torrent data
"""
torrent_id = dict([part.split('=') for part in urlparse(url)[4].split('&')])['t']
downloadurl = 'http://dl.rutracker.org/forum/dl.php?t=' + torrent_id
cookie = {'bb_dl': torrent_id}
try:
headers = {'Referer': url}
r = self.session.get(url=downloadurl, cookies=cookie, headers=headers, timeout=self.timeout)
return r.content
except Exception as e:
logger.error('Error getting torrent: %s', e)
return False
#TODO get this working in utorrent.py
def utorrent_add_file(self, data):
host = headphones.CONFIG.UTORRENT_HOST
if not host.startswith('http'):
host = 'http://' + host
if host.endswith('/'):
host = host[:-1]
if host.endswith('/gui'):
host = host[:-4]
base_url = host
url = base_url + '/gui/'
self.session.auth = (headphones.CONFIG.UTORRENT_USERNAME, headphones.CONFIG.UTORRENT_PASSWORD)
try:
r = self.session.get(url + 'token.html')
except Exception as e:
logger.error('Error getting token: %s', e)
return
if r.status_code == 401:
logger.debug('Error reaching utorrent')
return
regex = re.search(r'.+>([^<]+)</div></html>', r.text)
if regex is None:
logger.debug('Error reading token')
return
self.session.params = {'token': regex.group(1)}
files = {'torrent_file': ("", data)}
try:
self.session.post(url, params={'action': 'add-file'}, files=files)
except Exception as e:
logger.exception('Error adding file to utorrent %s', e)
+193 -82
View File
@@ -29,30 +29,27 @@ import string
import shutil import shutil
import random import random
import urllib import urllib
import datetime
import headphones import headphones
import subprocess import subprocess
import unicodedata import unicodedata
from headphones.common import USER_AGENT from headphones.common import USER_AGENT
from headphones import logger, db, helpers, classes, sab, nzbget, request from headphones import logger, db, helpers, classes, sab, nzbget, request
from headphones import utorrent, transmission, notifiers from headphones import utorrent, transmission, notifiers, rutracker
from bencode import bencode, bdecode from bencode import bencode, bdecode
import headphones.searcher_rutracker as rutrackersearch
# Magnet to torrent services, for Black hole. Stolen from CouchPotato. # Magnet to torrent services, for Black hole. Stolen from CouchPotato.
TORRENT_TO_MAGNET_SERVICES = [ TORRENT_TO_MAGNET_SERVICES = [
'https://zoink.it/torrent/%s.torrent', #'https://zoink.it/torrent/%s.torrent',
'http://torrage.com/torrent/%s.torrent', #'http://torrage.com/torrent/%s.torrent',
'https://torcache.net/torrent/%s.torrent', 'https://torcache.net/torrent/%s.torrent',
] ]
# Persistent What.cd API object # Persistent What.cd API object
gazelle = None gazelle = None
ruobj = None
# RUtracker search object
rutracker = rutrackersearch.Rutracker()
def fix_url(s, charset="utf-8"): def fix_url(s, charset="utf-8"):
@@ -167,6 +164,8 @@ def get_seed_ratio(provider):
seed_ratio = headphones.CONFIG.WAFFLES_RATIO seed_ratio = headphones.CONFIG.WAFFLES_RATIO
elif provider == 'Mininova': elif provider == 'Mininova':
seed_ratio = headphones.CONFIG.MININOVA_RATIO seed_ratio = headphones.CONFIG.MININOVA_RATIO
elif provider == 'Strike':
seed_ratio = headphones.CONFIG.STRIKE_RATIO
else: else:
seed_ratio = None seed_ratio = None
@@ -189,10 +188,22 @@ def searchforalbum(albumid=None, new=False, losslessOnly=False,
results = myDB.select('SELECT * from albums WHERE Status="Wanted" OR Status="Wanted Lossless"') results = myDB.select('SELECT * from albums WHERE Status="Wanted" OR Status="Wanted Lossless"')
for album in results: for album in results:
if not album['AlbumTitle'] or not album['ArtistName']: if not album['AlbumTitle'] or not album['ArtistName']:
logger.warn('Skipping release %s. No title available', album['AlbumID']) logger.warn('Skipping release %s. No title available', album['AlbumID'])
continue continue
if headphones.CONFIG.WAIT_UNTIL_RELEASE_DATE and album['ReleaseDate']:
try:
release_date = datetime.datetime.strptime(album['ReleaseDate'], "%Y-%m-%d")
except:
logger.warn("No valid date for: %s. Skipping automatic search" % album['AlbumTitle'])
continue
if release_date > datetime.datetime.today():
logger.info("Skipping: %s. Waiting for release date of: %s" % (album['AlbumTitle'], album['ReleaseDate']))
continue
new = True new = True
if album['Status'] == "Wanted Lossless": if album['Status'] == "Wanted Lossless":
@@ -219,7 +230,7 @@ def do_sorted_search(album, new, losslessOnly, choose_specific_download=False):
NZB_PROVIDERS = (headphones.CONFIG.HEADPHONES_INDEXER or headphones.CONFIG.NEWZNAB or headphones.CONFIG.NZBSORG or headphones.CONFIG.OMGWTFNZBS) NZB_PROVIDERS = (headphones.CONFIG.HEADPHONES_INDEXER or headphones.CONFIG.NEWZNAB or headphones.CONFIG.NZBSORG or headphones.CONFIG.OMGWTFNZBS)
NZB_DOWNLOADERS = (headphones.CONFIG.SAB_HOST or headphones.CONFIG.BLACKHOLE_DIR or headphones.CONFIG.NZBGET_HOST) NZB_DOWNLOADERS = (headphones.CONFIG.SAB_HOST or headphones.CONFIG.BLACKHOLE_DIR or headphones.CONFIG.NZBGET_HOST)
TORRENT_PROVIDERS = (headphones.CONFIG.KAT or headphones.CONFIG.PIRATEBAY or headphones.CONFIG.OLDPIRATEBAY or headphones.CONFIG.MININOVA or headphones.CONFIG.WAFFLES or headphones.CONFIG.RUTRACKER or headphones.CONFIG.WHATCD) TORRENT_PROVIDERS = (headphones.CONFIG.TORZNAB or headphones.CONFIG.KAT or headphones.CONFIG.PIRATEBAY or headphones.CONFIG.OLDPIRATEBAY or headphones.CONFIG.MININOVA or headphones.CONFIG.WAFFLES or headphones.CONFIG.RUTRACKER or headphones.CONFIG.WHATCD or headphones.CONFIG.STRIKE)
results = [] results = []
myDB = db.DBConnection() myDB = db.DBConnection()
@@ -780,10 +791,11 @@ def send_to_downloader(data, bestqual, album):
# Randomize list of services # Randomize list of services
services = TORRENT_TO_MAGNET_SERVICES[:] services = TORRENT_TO_MAGNET_SERVICES[:]
random.shuffle(services) random.shuffle(services)
headers = {'User-Agent': USER_AGENT}
for service in services: for service in services:
data = request.request_content(service % torrent_hash)
data = request.request_content(service % torrent_hash, headers=headers)
if data and "torcache" in data: if data and "torcache" in data:
if not torrent_to_file(download_path, data): if not torrent_to_file(download_path, data):
return return
@@ -805,15 +817,9 @@ def send_to_downloader(data, bestqual, album):
"to open or convert magnet links") "to open or convert magnet links")
return return
else: else:
if bestqual[3] == "rutracker.org":
download_path, _ = rutracker.get_torrent(bestqual[2],
headphones.CONFIG.TORRENTBLACKHOLE_DIR)
if not download_path: if not torrent_to_file(download_path, data):
return return
else:
if not torrent_to_file(download_path, data):
return
# Extract folder name from torrent # Extract folder name from torrent
folder_name = read_torrent_name(download_path, bestqual[0]) folder_name = read_torrent_name(download_path, bestqual[0])
@@ -823,13 +829,11 @@ def send_to_downloader(data, bestqual, album):
elif headphones.CONFIG.TORRENT_DOWNLOADER == 1: elif headphones.CONFIG.TORRENT_DOWNLOADER == 1:
logger.info("Sending torrent to Transmission") logger.info("Sending torrent to Transmission")
# rutracker needs cookies to be set, pass the .torrent file instead of url # Add torrent
if bestqual[3] == 'rutracker.org': if bestqual[3] == 'rutracker.org':
file_or_url, torrentid = rutracker.get_torrent(bestqual[2]) torrentid = transmission.addTorrent('', data)
else: else:
file_or_url = bestqual[2] torrentid = transmission.addTorrent(bestqual[2])
torrentid = transmission.addTorrent(file_or_url)
if not torrentid: if not torrentid:
logger.error("Error sending torrent to Transmission. Are you sure it's running?") logger.error("Error sending torrent to Transmission. Are you sure it's running?")
@@ -842,13 +846,6 @@ def send_to_downloader(data, bestqual, album):
logger.error('Torrent folder name could not be determined') logger.error('Torrent folder name could not be determined')
return return
# remove temp .torrent file created above
if bestqual[3] == 'rutracker.org':
try:
shutil.rmtree(os.path.split(file_or_url)[0])
except Exception as e:
logger.exception("Unhandled exception")
# Set Seed Ratio # Set Seed Ratio
seed_ratio = get_seed_ratio(bestqual[3]) seed_ratio = get_seed_ratio(bestqual[3])
if seed_ratio is not None: if seed_ratio is not None:
@@ -857,29 +854,29 @@ def send_to_downloader(data, bestqual, album):
else:# if headphones.CONFIG.TORRENT_DOWNLOADER == 2: else:# if headphones.CONFIG.TORRENT_DOWNLOADER == 2:
logger.info("Sending torrent to uTorrent") logger.info("Sending torrent to uTorrent")
# rutracker needs cookies to be set, pass the .torrent file instead of url # Add torrent
if bestqual[3] == 'rutracker.org': if bestqual[3] == 'rutracker.org':
file_or_url, torrentid = rutracker.get_torrent(bestqual[2]) ruobj.utorrent_add_file(data)
folder_name, cacheid = utorrent.dirTorrent(torrentid)
folder_name = os.path.basename(os.path.normpath(folder_name))
utorrent.labelTorrent(torrentid)
else: else:
file_or_url = bestqual[2] utorrent.addTorrent(bestqual[2])
torrentid = calculate_torrent_hash(file_or_url, data)
folder_name = utorrent.addTorrent(file_or_url, torrentid)
# Get hash
torrentid = calculate_torrent_hash(bestqual[2], data)
if not torrentid:
logger.error('Torrent id could not be determined')
return
# Get folder
folder_name = utorrent.getFolder(torrentid)
if folder_name: if folder_name:
logger.info('Torrent folder name: %s' % folder_name) logger.info('Torrent folder name: %s' % folder_name)
else: else:
logger.error('Torrent folder name could not be determined') logger.error('Torrent folder name could not be determined')
return return
# remove temp .torrent file created above # Set Label
if bestqual[3] == 'rutracker.org': if headphones.CONFIG.UTORRENT_LABEL:
try: utorrent.labelTorrent(torrentid)
shutil.rmtree(os.path.split(file_or_url)[0])
except Exception as e:
logger.exception("Unhandled exception")
# Set Seed Ratio # Set Seed Ratio
seed_ratio = get_seed_ratio(bestqual[3]) seed_ratio = get_seed_ratio(bestqual[3])
@@ -919,7 +916,7 @@ def send_to_downloader(data, bestqual, album):
if headphones.CONFIG.PUSHBULLET_ENABLED and headphones.CONFIG.PUSHBULLET_ONSNATCH: if headphones.CONFIG.PUSHBULLET_ENABLED and headphones.CONFIG.PUSHBULLET_ONSNATCH:
logger.info(u"Sending PushBullet notification") logger.info(u"Sending PushBullet notification")
pushbullet = notifiers.PUSHBULLET() pushbullet = notifiers.PUSHBULLET()
pushbullet.notify(name + " has been snatched!", "Download started") pushbullet.notify(name, "Download started")
if headphones.CONFIG.TWITTER_ENABLED and headphones.CONFIG.TWITTER_ONSNATCH: if headphones.CONFIG.TWITTER_ENABLED and headphones.CONFIG.TWITTER_ONSNATCH:
logger.info(u"Sending Twitter notification") logger.info(u"Sending Twitter notification")
twitter = notifiers.TwitterNotifier() twitter = notifiers.TwitterNotifier()
@@ -1028,12 +1025,7 @@ def verifyresult(title, artistterm, term, lossless):
def searchTorrent(album, new=False, losslessOnly=False, albumlength=None, choose_specific_download=False): def searchTorrent(album, new=False, losslessOnly=False, albumlength=None, choose_specific_download=False):
global gazelle # persistent what.cd api object to reduce number of login attempts global gazelle # persistent what.cd api object to reduce number of login attempts
global ruobj # and rutracker
# rutracker login
if headphones.CONFIG.RUTRACKER and album:
rulogin = rutracker.login(headphones.CONFIG.RUTRACKER_USER, headphones.CONFIG.RUTRACKER_PASSWORD)
if not rulogin:
logger.info(u'Could not login to rutracker, search results will exclude this provider')
albumid = album['AlbumID'] albumid = album['AlbumID']
reldate = album['ReleaseDate'] reldate = album['ReleaseDate']
@@ -1097,6 +1089,68 @@ def searchTorrent(album, new=False, losslessOnly=False, albumlength=None, choose
return proxy_url return proxy_url
if headphones.CONFIG.TORZNAB:
provider = "torznab"
torznab_hosts = []
if headphones.CONFIG.TORZNAB_HOST and headphones.CONFIG.TORZNAB_ENABLED:
torznab_hosts.append((headphones.CONFIG.TORZNAB_HOST, headphones.CONFIG.TORZNAB_APIKEY, headphones.CONFIG.TORZNAB_ENABLED))
for torznab_host in headphones.CONFIG.get_extra_torznabs():
if torznab_host[2] == '1' or torznab_host[2] == 1:
torznab_hosts.append(torznab_host)
if headphones.CONFIG.PREFERRED_QUALITY == 3 or losslessOnly:
categories = "3040"
elif headphones.CONFIG.PREFERRED_QUALITY == 1 or allow_lossless:
categories = "3040,3010"
else:
categories = "3010"
if album['Type'] == 'Other':
categories = "3030"
logger.info("Album type is audiobook/spokenword. Using audiobook category")
for torznab_host in torznab_hosts:
provider = torznab_host[0]
# Request results
logger.info('Parsing results from %s using search term: %s' % (torznab_host[0],term))
headers = {'User-Agent': USER_AGENT}
params = {
"t": "search",
"apikey": torznab_host[1],
"cat": categories,
"maxage": headphones.CONFIG.USENET_RETENTION,
"q": term
}
data = request.request_feed(
url=torznab_host[0] + '/api?',
params=params, headers=headers
)
# Process feed
if data:
if not len(data.entries):
logger.info(u"No results found from %s for %s", torznab_host[0], term)
else:
for item in data.entries:
try:
url = item.link
title = item.title
size = int(item.links[1]['length'])
if all(word.lower() in title.lower() for word in term.split()):
logger.info('Found %s. Size: %s' % (title, helpers.bytes_to_mb(size)))
resultlist.append((title, size, url, provider, 'torrent', True))
else:
logger.info('Skipping %s, not all search term words found' % title)
except Exception as e:
logger.exception("An unknown error occurred trying to parse the feed: %s" % e)
if headphones.CONFIG.KAT: if headphones.CONFIG.KAT:
provider = "Kick Ass Torrents" provider = "Kick Ass Torrents"
ka_term = term.replace("!", "") ka_term = term.replace("!", "")
@@ -1129,7 +1183,8 @@ def searchTorrent(album, new=False, losslessOnly=False, albumlength=None, choose
"field": "seeders", "field": "seeders",
"sorder": "desc" "sorder": "desc"
} }
data = request.request_json(url=providerurl, params=params) headers = {'User-Agent': USER_AGENT}
data = request.request_json(url=providerurl, params=params, headers=headers)
# Process feed # Process feed
if data: if data:
@@ -1145,7 +1200,7 @@ def searchTorrent(album, new=False, losslessOnly=False, albumlength=None, choose
size = int(item['size']) size = int(item['size'])
if format == "2": if format == "2":
torrent = request.request_content(url) torrent = request.request_content(url, headers=headers)
if not torrent or (int(torrent.find(".mp3")) > 0 and int(torrent.find(".flac")) < 1): if not torrent or (int(torrent.find(".mp3")) > 0 and int(torrent.find(".flac")) < 1):
rightformat = False rightformat = False
@@ -1226,45 +1281,38 @@ def searchTorrent(album, new=False, losslessOnly=False, albumlength=None, choose
logger.error(u"An error occurred while trying to parse the response from Waffles.fm: %s", e) logger.error(u"An error occurred while trying to parse the response from Waffles.fm: %s", e)
# rutracker.org # rutracker.org
if headphones.CONFIG.RUTRACKER and rulogin: if headphones.CONFIG.RUTRACKER:
provider = "rutracker.org" provider = "rutracker.org"
# Ignore if release date not specified, results too unpredictable # Ignore if release date not specified, results too unpredictable
if not year and not usersearchterm: if not year and not usersearchterm:
logger.info(u'Release date not specified, ignoring for rutracker.org') logger.info(u"Release date not specified, ignoring for rutracker.org")
else: else:
if headphones.CONFIG.PREFERRED_QUALITY == 3 or losslessOnly: if headphones.CONFIG.PREFERRED_QUALITY == 3 or losslessOnly:
format = 'lossless' format = 'lossless'
maxsize = 10000000000
elif headphones.CONFIG.PREFERRED_QUALITY == 1 or allow_lossless: elif headphones.CONFIG.PREFERRED_QUALITY == 1 or allow_lossless:
format = 'lossless+mp3' format = 'lossless+mp3'
maxsize = 10000000000
else: else:
format = 'mp3' format = 'mp3'
maxsize = 300000000
# build search url based on above # Login
if not usersearchterm: if not ruobj or not ruobj.logged_in():
searchURL = rutracker.searchurl(artistterm, albumterm, year, format) ruobj = rutracker.Rutracker()
else: if not ruobj.login():
searchURL = rutracker.searchurl(usersearchterm, ' ', ' ', format) ruobj = None
logger.info(u'Parsing results from <a href="%s">rutracker.org</a>' % searchURL) if ruobj and ruobj.logged_in():
# parse results and get best match # build search url
rulist = rutracker.search(searchURL, maxsize, minimumseeders, albumid) if not usersearchterm:
searchURL = ruobj.searchurl(artistterm, albumterm, year, format)
else:
searchURL = ruobj.searchurl(usersearchterm, ' ', ' ', format)
# add best match to overall results list # parse results
if rulist: rulist = ruobj.search(searchURL)
for ru in rulist: if rulist:
title = ru[0].decode('utf-8') resultlist.extend(rulist)
size = ru[1]
url = ru[2]
resultlist.append((title, size, url, provider, 'torrent', True))
logger.info('Found %s. Size: %s' % (title, helpers.bytes_to_mb(size)))
else:
logger.info(u"No valid results found from %s" % (provider))
if headphones.CONFIG.WHATCD: if headphones.CONFIG.WHATCD:
provider = "What.cd" provider = "What.cd"
@@ -1280,6 +1328,12 @@ def searchTorrent(album, new=False, losslessOnly=False, albumlength=None, choose
search_formats = [None] # should return all search_formats = [None] # should return all
bitrate = headphones.CONFIG.PREFERRED_BITRATE bitrate = headphones.CONFIG.PREFERRED_BITRATE
if bitrate: if bitrate:
if 225 <= int(bitrate) < 256:
bitrate = 'V0'
elif 200 <= int(bitrate) < 225:
bitrate = 'V1'
elif 175 <= int(bitrate) < 200:
bitrate = 'V2'
for encoding_string in gazelleencoding.ALL_ENCODINGS: for encoding_string in gazelleencoding.ALL_ENCODINGS:
if re.search(bitrate, encoding_string, flags=re.I): if re.search(bitrate, encoding_string, flags=re.I):
bitrate_string = encoding_string bitrate_string = encoding_string
@@ -1306,7 +1360,10 @@ def searchTorrent(album, new=False, losslessOnly=False, albumlength=None, choose
logger.info(u"Searching %s..." % provider) logger.info(u"Searching %s..." % provider)
all_torrents = [] all_torrents = []
for search_format in search_formats: for search_format in search_formats:
all_torrents.extend(gazelle.search_torrents(artistname=semi_clean_artist_term, if usersearchterm:
all_torrents.extend(gazelle.search_torrents(searchstr=usersearchterm, format=search_format, encoding=bitrate_string)['results'])
else:
all_torrents.extend(gazelle.search_torrents(artistname=semi_clean_artist_term,
groupname=semi_clean_album_term, groupname=semi_clean_album_term,
format=search_format, encoding=bitrate_string)['results']) format=search_format, encoding=bitrate_string)['results'])
@@ -1469,6 +1526,57 @@ def searchTorrent(album, new=False, losslessOnly=False, albumlength=None, choose
except Exception as e: except Exception as e:
logger.error(u"An unknown error occurred in the Old Pirate Bay parser: %s" % e) logger.error(u"An unknown error occurred in the Old Pirate Bay parser: %s" % e)
# Strike
if headphones.CONFIG.STRIKE:
provider = "Strike"
s_term = term.replace("!", "")
providerurl = fix_url("https://getstrike.net/api/v2/torrents/search/?phrase=")
providerurl = providerurl + s_term + "&category=Music"
if headphones.CONFIG.PREFERRED_QUALITY == 3 or losslessOnly:
format = "2"
providerurl = providerurl + "&subcategory=Lossless"
maxsize = 10000000000
elif headphones.CONFIG.PREFERRED_QUALITY == 1 or allow_lossless:
format = "10" # MP3 and FLAC
maxsize = 10000000000
else:
format = "8" # MP3 only
maxsize = 300000000
logger.info("Searching %s using term: %s" % (provider, s_term))
data = request.request_json(url=providerurl)
if not data or not data.get('torrents'):
logger.info("No results found on %s using search term: %s" % (provider, s_term))
else:
for item in data['torrents']:
try:
rightformat = True
title = item['torrent_title']
seeders = item['seeds']
url = item['magnet_uri']
size = int(item['size'])
subcategory = item['sub_category']
if format == 2:
if subcategory != "Lossless":
rightformat = False
if rightformat and size < maxsize and minimumseeders < int(seeders):
match = True
logger.info('Found %s. Size: %s' % (title, helpers.bytes_to_mb(size)))
else:
match = False
logger.info(
'%s is larger than the maxsize, the wrong format or has too little seeders for this category, skipping. (Size: %i bytes, Seeders: %d, Format: %s)',
title, size, int(seeders), rightformat)
resultlist.append((title, size, url, provider, 'torrent', match))
except Exception as e:
logger.exception("Unhandled exception in the Strike parser")
# Mininova # Mininova
if headphones.CONFIG.MININOVA: if headphones.CONFIG.MININOVA:
provider = "Mininova" provider = "Mininova"
@@ -1545,12 +1653,14 @@ def preprocess(resultlist):
for result in resultlist: for result in resultlist:
if result[4] == 'torrent': if result[4] == 'torrent':
# rutracker always needs the torrent data
if result[3] == 'rutracker.org':
return ruobj.get_torrent_data(result[2]), result
#Get out of here if we're using Transmission #Get out of here if we're using Transmission
if headphones.CONFIG.TORRENT_DOWNLOADER == 1: ## if not a magnet link still need the .torrent to generate hash... uTorrent support labeling if headphones.CONFIG.TORRENT_DOWNLOADER == 1: ## if not a magnet link still need the .torrent to generate hash... uTorrent support labeling
return True, result return True, result
# get outta here if rutracker
if result[3] == 'rutracker.org':
return True, result
# Get out of here if it's a magnet link # Get out of here if it's a magnet link
if result[2].lower().startswith("magnet:"): if result[2].lower().startswith("magnet:"):
return True, result return True, result
@@ -1559,7 +1669,8 @@ def preprocess(resultlist):
headers = {} headers = {}
if result[3] == 'Kick Ass Torrents': if result[3] == 'Kick Ass Torrents':
headers['Referer'] = 'http://kat.ph/' #headers['Referer'] = 'http://kat.ph/'
headers['User-Agent'] = USER_AGENT
elif result[3] == 'What.cd': elif result[3] == 'What.cd':
headers['User-Agent'] = 'Headphones' headers['User-Agent'] = 'Headphones'
elif result[3] == "The Pirate Bay" or result[3] == "Old Pirate Bay": elif result[3] == "The Pirate Bay" or result[3] == "Old Pirate Bay":
-349
View File
@@ -1,349 +0,0 @@
#!/usr/bin/env python
# coding=utf-8
# Headphones rutracker.org search
# Functions called from searcher.py
from bencode import bencode as bencode, bdecode
from urlparse import urlparse
from bs4 import BeautifulSoup
from tempfile import mkdtemp
from hashlib import sha1
import headphones
import requests
import cookielib
import urllib2
import urllib
import re
import os
from headphones import db, logger
class Rutracker():
logged_in = False
# Stores a number of login attempts to prevent recursion.
#login_counter = 0
def __init__(self):
self.cookiejar = cookielib.CookieJar()
self.opener = urllib2.build_opener(urllib2.HTTPCookieProcessor(self.cookiejar))
urllib2.install_opener(self.opener)
def login(self, login, password):
"""Implements tracker login procedure."""
self.logged_in = False
if login is None or password is None:
return False
#self.login_counter += 1
# No recursion wanted.
#if self.login_counter > 1:
# return False
params = urllib.urlencode({"login_username": login,
"login_password": password,
"login": "Вход"})
try:
self.opener.open("http://login.rutracker.org/forum/login.php", params)
except Exception:
pass
# Check if we're logged in
for cookie in self.cookiejar:
if cookie.name == 'bb_data':
self.logged_in = True
return self.logged_in
def searchurl(self, artist, album, year, format):
"""
Return the search url
"""
# Build search url
searchterm = ''
if artist != 'Various Artists':
searchterm = artist
searchterm = searchterm + ' '
searchterm = searchterm + album
searchterm = searchterm + ' '
searchterm = searchterm + year
providerurl = "http://rutracker.org/forum/tracker.php"
if format == 'lossless':
format = '+lossless'
elif format == 'lossless+mp3':
format = '+lossless||mp3||aac'
else:
format = '+mp3||aac'
# sort by size, descending.
sort = '&o=7&s=2'
searchurl = "%s?nm=%s%s%s" % (providerurl, urllib.quote(searchterm), format, sort)
return searchurl
def search(self, searchurl, maxsize, minseeders, albumid):
"""
Parse the search results and return valid torrent list
"""
titles = []
urls = []
seeders = []
sizes = []
torrentlist = []
rulist = []
try:
page = self.opener.open(searchurl, timeout=60)
soup = BeautifulSoup(page.read())
# Debug
#logger.debug (soup.prettify())
# Title
for link in soup.find_all('a', attrs={'class': 'med tLink hl-tags bold'}):
title = link.get_text()
titles.append(title)
# Download URL
for link in soup.find_all('a', attrs={'class': 'small tr-dl dl-stub'}):
url = link.get('href')
urls.append(url)
# Seeders
for link in soup.find_all('b', attrs={'class': 'seedmed'}):
seeder = link.get_text()
seeders.append(seeder)
# Size
for link in soup.find_all('td', attrs={'class': 'row4 small nowrap tor-size'}):
size = link.u.string
sizes.append(size)
except:
pass
# Combine lists
torrentlist = zip(titles, urls, seeders, sizes)
# return if nothing found
if not torrentlist:
return False
# don't bother checking track counts anymore, let searcher filter instead
# leave code in just in case
check_track_count = False
if check_track_count:
# get headphones track count for album, return if not found
myDB = db.DBConnection()
tracks = myDB.select('SELECT * from tracks WHERE AlbumID=?', [albumid])
hptrackcount = len(tracks)
if not hptrackcount:
logger.info('headphones track info not found, cannot compare to torrent')
return False
# Return all valid entries, ignored, required words now checked in searcher.py
#unwantedlist = ['promo', 'vinyl', '[lp]', 'songbook', 'tvrip', 'hdtv', 'dvd']
formatlist = ['ape', 'flac', 'ogg', 'm4a', 'aac', 'mp3', 'wav', 'aif']
deluxelist = ['deluxe', 'edition', 'japanese', 'exclusive']
for torrent in torrentlist:
returntitle = torrent[0].encode('utf-8')
url = torrent[1]
seeders = torrent[2]
size = torrent[3]
if int(size) <= maxsize and int(seeders) >= minseeders:
#Torrent topic page
torrent_id = dict([part.split('=') for part in urlparse(url)[4].split('&')])['t']
topicurl = 'http://rutracker.org/forum/viewtopic.php?t=' + torrent_id
# add to list
if not check_track_count:
valid = True
else:
# Check torrent info
self.cookiejar.set_cookie(cookielib.Cookie(version=0, name='bb_dl', value=torrent_id, port=None, port_specified=False, domain='.rutracker.org', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False))
# Debug
#for cookie in self.cookiejar:
# logger.debug ('Cookie: %s' % cookie)
try:
page = self.opener.open(url)
torrent = page.read()
if torrent:
decoded = bdecode(torrent)
metainfo = decoded['info']
page.close()
except Exception as e:
logger.error('Error getting torrent: %s' % e)
return False
# get torrent track count and check for cue
trackcount = 0
cuecount = 0
if 'files' in metainfo: # multi
for pathfile in metainfo['files']:
path = pathfile['path']
for file in path:
if any(file.lower().endswith('.' + x.lower()) for x in formatlist):
trackcount += 1
if '.cue' in file:
cuecount += 1
title = returntitle.lower()
logger.debug('torrent title: %s' % title)
logger.debug('headphones trackcount: %s' % hptrackcount)
logger.debug('rutracker trackcount: %s' % trackcount)
# If torrent track count less than headphones track count, and there's a cue, then attempt to get track count from log(s)
# This is for the case where we have a single .flac/.wav which can be split by cue
# Not great, but shouldn't be doing this too often
totallogcount = 0
if trackcount < hptrackcount and cuecount > 0 and cuecount < hptrackcount:
page = self.opener.open(topicurl, timeout=60)
soup = BeautifulSoup(page.read())
findtoc = soup.find_all(text='TOC of the extracted CD')
if not findtoc:
findtoc = soup.find_all(text='TOC извлечённого CD')
for toc in findtoc:
logcount = 0
for toccontent in toc.find_all_next(text=True):
cut_string = toccontent.split('|')
new_string = cut_string[0].lstrip().rstrip()
if new_string == '1' or new_string == '01':
logcount = 1
elif logcount > 0:
if new_string.isdigit():
logcount += 1
else:
break
totallogcount = totallogcount + logcount
if totallogcount > 0:
trackcount = totallogcount
logger.debug('rutracker logtrackcount: %s' % totallogcount)
# If torrent track count = hp track count then return torrent,
# if greater, check for deluxe/special/foreign editions
# if less, then allow if it's a single track with a cue
valid = False
if trackcount == hptrackcount:
valid = True
elif trackcount > hptrackcount:
if any(deluxe in title for deluxe in deluxelist):
valid = True
# Add to list
if valid:
rulist.append((returntitle, size, topicurl))
else:
if topicurl:
logger.info(u'<a href="%s">Torrent</a> found with %s tracks but the selected headphones release has %s tracks, skipping for rutracker.org' % (topicurl, trackcount, hptrackcount))
else:
logger.info('%s is larger than the maxsize or has too little seeders for this category, skipping. (Size: %i bytes, Seeders: %i)' % (returntitle, int(size), int(seeders)))
return rulist
def get_torrent(self, url, savelocation=None):
torrent_id = dict([part.split('=') for part in urlparse(url)[4].split('&')])['t']
self.cookiejar.set_cookie(cookielib.Cookie(version=0, name='bb_dl', value=torrent_id, port=None, port_specified=False, domain='.rutracker.org', domain_specified=False, domain_initial_dot=False, path='/', path_specified=True, secure=False, expires=None, discard=True, comment=None, comment_url=None, rest={'HttpOnly': None}, rfc2109=False))
downloadurl = 'http://dl.rutracker.org/forum/dl.php?t=' + torrent_id
torrent_name = torrent_id + '.torrent'
try:
prev = os.umask(headphones.UMASK)
page = self.opener.open(downloadurl)
torrent = page.read()
decoded = bdecode(torrent)
metainfo = decoded['info']
tor_hash = sha1(bencode(metainfo)).hexdigest()
if savelocation:
download_path = os.path.join(savelocation, torrent_name)
else:
tempdir = mkdtemp(suffix='_rutracker_torrents')
download_path = os.path.join(tempdir, torrent_name)
with open(download_path, 'wb') as f:
f.write(torrent)
os.umask(prev)
# Add file to utorrent
if headphones.CONFIG.TORRENT_DOWNLOADER == 2:
self.utorrent_add_file(download_path)
except Exception as e:
logger.error('Error getting torrent: %s', e)
return False
return download_path, tor_hash
#TODO get this working in utorrent.py
def utorrent_add_file(self, filename):
host = headphones.CONFIG.UTORRENT_HOST
if not host.startswith('http'):
host = 'http://' + host
if host.endswith('/'):
host = host[:-1]
if host.endswith('/gui'):
host = host[:-4]
base_url = host
username = headphones.CONFIG.UTORRENT_USERNAME
password = headphones.CONFIG.UTORRENT_PASSWORD
session = requests.Session()
url = base_url + '/gui/'
session.auth = (username, password)
try:
r = session.get(url + 'token.html')
except Exception:
logger.exception('Error getting token')
return
if r.status_code == '401':
logger.debug('Error reaching utorrent')
return
regex = re.search(r'.+>([^<]+)</div></html>', r.text)
if regex is None:
logger.debug('Error reading token')
return
session.params = {'token': regex.group(1)}
with open(filename, 'rb') as f:
try:
session.post(url, params={'action': 'add-file'},
files={'torrent_file': f})
except Exception:
logger.exception('Error adding file to utorrent')
return
+7 -4
View File
@@ -28,12 +28,15 @@ import headphones
# Store torrent id so we can check up on it # Store torrent id so we can check up on it
def addTorrent(link): def addTorrent(link, data=None):
method = 'torrent-add' method = 'torrent-add'
if link.endswith('.torrent'): if link.endswith('.torrent') or data:
with open(link, 'rb') as f: if data:
metainfo = str(base64.b64encode(f.read())) metainfo = str(base64.b64encode(data))
else:
with open(link, 'rb') as f:
metainfo = str(base64.b64encode(f.read()))
arguments = {'metainfo': metainfo, 'download-dir': headphones.CONFIG.DOWNLOAD_TORRENT_DIR} arguments = {'metainfo': metainfo, 'download-dir': headphones.CONFIG.DOWNLOAD_TORRENT_DIR}
else: else:
arguments = {'filename': link, 'download-dir': headphones.CONFIG.DOWNLOAD_TORRENT_DIR} arguments = {'filename': link, 'download-dir': headphones.CONFIG.DOWNLOAD_TORRENT_DIR}
+11 -6
View File
@@ -73,6 +73,7 @@ class utorrentclient(object):
except urllib2.HTTPError as err: except urllib2.HTTPError as err:
logger.debug('URL: ' + str(url)) logger.debug('URL: ' + str(url))
logger.debug('Error getting Token. uTorrent responded with error: ' + str(err)) logger.debug('Error getting Token. uTorrent responded with error: ' + str(err))
return
match = re.search(utorrentclient.TOKEN_REGEX, response.read()) match = re.search(utorrentclient.TOKEN_REGEX, response.read())
return match.group(1) return match.group(1)
@@ -147,6 +148,10 @@ class utorrentclient(object):
return self._action(params) return self._action(params)
def _action(self, params, body=None, content_type=None): def _action(self, params, body=None, content_type=None):
if not self.token:
return
url = self.base_url + '/gui/' + '?token=' + self.token + '&' + urllib.urlencode(params) url = self.base_url + '/gui/' + '?token=' + self.token + '&' + urllib.urlencode(params)
request = urllib2.Request(url) request = urllib2.Request(url)
@@ -215,7 +220,7 @@ def dirTorrent(hash, cacheid=None, return_name=None):
cacheid = torrentList['torrentc'] cacheid = torrentList['torrentc']
for torrent in torrents: for torrent in torrents:
if torrent[0].upper() == hash: if torrent[0].upper() == hash.upper():
if not return_name: if not return_name:
return torrent[26], cacheid return torrent[26], cacheid
else: else:
@@ -223,8 +228,12 @@ def dirTorrent(hash, cacheid=None, return_name=None):
return None, None return None, None
def addTorrent(link):
uTorrentClient = utorrentclient()
uTorrentClient.add_url(link)
def addTorrent(link, hash):
def getFolder(hash):
uTorrentClient = utorrentclient() uTorrentClient = utorrentclient()
# Get Active Directory from settings # Get Active Directory from settings
@@ -234,8 +243,6 @@ def addTorrent(link, hash):
logger.error('Could not get "Put new downloads in:" directory from uTorrent settings, please ensure it is set') logger.error('Could not get "Put new downloads in:" directory from uTorrent settings, please ensure it is set')
return None return None
uTorrentClient.add_url(link)
# Get Torrent Folder Name # Get Torrent Folder Name
torrent_folder, cacheid = dirTorrent(hash) torrent_folder, cacheid = dirTorrent(hash)
@@ -249,10 +256,8 @@ def addTorrent(link, hash):
if torrent_folder == active_dir or not torrent_folder: if torrent_folder == active_dir or not torrent_folder:
torrent_folder, cacheid = dirTorrent(hash, cacheid, return_name=True) torrent_folder, cacheid = dirTorrent(hash, cacheid, return_name=True)
labelTorrent(hash)
return torrent_folder return torrent_folder
else: else:
labelTorrent(hash)
if headphones.SYS_PLATFORM != "win32": if headphones.SYS_PLATFORM != "win32":
torrent_folder = torrent_folder.replace('\\', '/') torrent_folder = torrent_folder.replace('\\', '/')
return os.path.basename(os.path.normpath(torrent_folder)) return os.path.basename(os.path.normpath(torrent_folder))
+129 -15
View File
@@ -32,8 +32,10 @@ import random
import urllib import urllib
import json import json
import time import time
import cgi
import sys import sys
import os import os
import re
try: try:
# pylint:disable=E0611 # pylint:disable=E0611
@@ -149,7 +151,7 @@ class WebInterface(object):
searchresults = mb.findRelease(name, limit=100) searchresults = mb.findRelease(name, limit=100)
else: else:
searchresults = mb.findSeries(name, limit=100) searchresults = mb.findSeries(name, limit=100)
return serve_template(templatename="searchresults.html", title='Search Results for: "' + name + '"', searchresults=searchresults, name=name, type=type) return serve_template(templatename="searchresults.html", title='Search Results for: "' + cgi.escape(name) + '"', searchresults=searchresults, name=cgi.escape(name), type=type)
@cherrypy.expose @cherrypy.expose
def addArtist(self, artistid): def addArtist(self, artistid):
@@ -230,11 +232,11 @@ class WebInterface(object):
raise cherrypy.HTTPRedirect("artistPage?ArtistID=%s" % ArtistID) raise cherrypy.HTTPRedirect("artistPage?ArtistID=%s" % ArtistID)
def removeArtist(self, ArtistID): def removeArtist(self, ArtistID):
logger.info(u"Deleting all traces of artist: " + ArtistID)
myDB = db.DBConnection() myDB = db.DBConnection()
namecheck = myDB.select('SELECT ArtistName from artists where ArtistID=?', [ArtistID]) namecheck = myDB.select('SELECT ArtistName from artists where ArtistID=?', [ArtistID])
for name in namecheck: for name in namecheck:
artistname = name['ArtistName'] artistname = name['ArtistName']
logger.info(u"Deleting all traces of artist: " + artistname)
myDB.action('DELETE from artists WHERE ArtistID=?', [ArtistID]) myDB.action('DELETE from artists WHERE ArtistID=?', [ArtistID])
from headphones import cache from headphones import cache
@@ -265,14 +267,72 @@ class WebInterface(object):
@cherrypy.expose @cherrypy.expose
def scanArtist(self, ArtistID): def scanArtist(self, ArtistID):
logger.info(u"Scanning artist: %s", ArtistID)
myDB = db.DBConnection() myDB = db.DBConnection()
artistname = myDB.select('SELECT DISTINCT ArtistName FROM artists WHERE ArtistID=?', [ArtistID]) artist_name = myDB.select('SELECT DISTINCT ArtistName FROM artists WHERE ArtistID=?', [ArtistID])[0][0]
artistfolder = os.path.join(headphones.CONFIG.DESTINATION_DIR, artistname[0][0])
try: logger.info(u"Scanning artist: %s", artist_name)
threading.Thread(target=librarysync.libraryScan(dir=artistfolder)).start()
except Exception as e: full_folder_format = headphones.CONFIG.FOLDER_FORMAT
logger.error('Unable to complete the scan: %s', e) folder_format = re.findall(r'(.*?[Aa]rtist?)\.*', full_folder_format)[0]
acceptable_formats = ["$artist","$sortartist","$first/$artist","$first/$sortartist"]
if not folder_format.lower() in acceptable_formats:
logger.info("Can't determine the artist folder from the configured folder_format. Not scanning")
return
# Format the folder to match the settings
artist = artist_name.replace('/', '_')
if headphones.CONFIG.FILE_UNDERSCORES:
artist = artist.replace(' ', '_')
if artist.startswith('The '):
sortname = artist[4:] + ", The"
else:
sortname = artist
if sortname[0].isdigit():
firstchar = u'0-9'
else:
firstchar = sortname[0]
values = {'$Artist': artist,
'$SortArtist': sortname,
'$First': firstchar.upper(),
'$artist': artist.lower(),
'$sortartist': sortname.lower(),
'$first': firstchar.lower(),
}
folder = helpers.replace_all(folder_format.strip(), values, normalize=True)
folder = helpers.replace_illegal_chars(folder, type="folder")
folder = folder.replace('./', '_/').replace('/.', '/_')
if folder.endswith('.'):
folder = folder[:-1] + '_'
if folder.startswith('.'):
folder = '_' + folder[1:]
dirs = []
if headphones.CONFIG.MUSIC_DIR:
dirs.append(headphones.CONFIG.MUSIC_DIR)
if headphones.CONFIG.DESTINATION_DIR:
dirs.append(headphones.CONFIG.DESTINATION_DIR)
if headphones.CONFIG.LOSSLESS_DESTINATION_DIR:
dirs.append(headphones.CONFIG.LOSSLESS_DESTINATION_DIR)
dirs = set(dirs)
for dir in dirs:
artistfolder = os.path.join(dir, folder)
if not os.path.isdir(artistfolder):
logger.debug("Cannot find directory: " + artistfolder)
continue
threading.Thread(target=librarysync.libraryScan, kwargs={"dir":artistfolder, "artistScan":True, "ArtistID":ArtistID, "ArtistName":artist_name}).start()
raise cherrypy.HTTPRedirect("artistPage?ArtistID=%s" % ArtistID) raise cherrypy.HTTPRedirect("artistPage?ArtistID=%s" % ArtistID)
@cherrypy.expose @cherrypy.expose
@@ -740,9 +800,9 @@ class WebInterface(object):
raise cherrypy.HTTPRedirect("home") raise cherrypy.HTTPRedirect("home")
@cherrypy.expose @cherrypy.expose
def forcePostProcess(self, dir=None, album_dir=None): def forcePostProcess(self, dir=None, album_dir=None, keep_original_folder=False):
from headphones import postprocessor from headphones import postprocessor
threading.Thread(target=postprocessor.forcePostProcess, kwargs={'dir': dir, 'album_dir': album_dir}).start() threading.Thread(target=postprocessor.forcePostProcess, kwargs={'dir': dir, 'album_dir': album_dir, 'keep_original_folder':keep_original_folder == 'True'}).start()
raise cherrypy.HTTPRedirect("home") raise cherrypy.HTTPRedirect("home")
@cherrypy.expose @cherrypy.expose
@@ -1005,6 +1065,11 @@ class WebInterface(object):
"newznab_apikey": headphones.CONFIG.NEWZNAB_APIKEY, "newznab_apikey": headphones.CONFIG.NEWZNAB_APIKEY,
"newznab_enabled": checked(headphones.CONFIG.NEWZNAB_ENABLED), "newznab_enabled": checked(headphones.CONFIG.NEWZNAB_ENABLED),
"extra_newznabs": headphones.CONFIG.get_extra_newznabs(), "extra_newznabs": headphones.CONFIG.get_extra_newznabs(),
"use_torznab": checked(headphones.CONFIG.TORZNAB),
"torznab_host": headphones.CONFIG.TORZNAB_HOST,
"torznab_apikey": headphones.CONFIG.TORZNAB_APIKEY,
"torznab_enabled": checked(headphones.CONFIG.TORZNAB_ENABLED),
"extra_torznabs": headphones.CONFIG.get_extra_torznabs(),
"use_nzbsorg": checked(headphones.CONFIG.NZBSORG), "use_nzbsorg": checked(headphones.CONFIG.NZBSORG),
"nzbsorg_uid": headphones.CONFIG.NZBSORG_UID, "nzbsorg_uid": headphones.CONFIG.NZBSORG_UID,
"nzbsorg_hash": headphones.CONFIG.NZBSORG_HASH, "nzbsorg_hash": headphones.CONFIG.NZBSORG_HASH,
@@ -1041,6 +1106,8 @@ class WebInterface(object):
"whatcd_username": headphones.CONFIG.WHATCD_USERNAME, "whatcd_username": headphones.CONFIG.WHATCD_USERNAME,
"whatcd_password": headphones.CONFIG.WHATCD_PASSWORD, "whatcd_password": headphones.CONFIG.WHATCD_PASSWORD,
"whatcd_ratio": headphones.CONFIG.WHATCD_RATIO, "whatcd_ratio": headphones.CONFIG.WHATCD_RATIO,
"use_strike": checked(headphones.CONFIG.STRIKE),
"strike_ratio": headphones.CONFIG.STRIKE_RATIO,
"pref_qual_0": radio(headphones.CONFIG.PREFERRED_QUALITY, 0), "pref_qual_0": radio(headphones.CONFIG.PREFERRED_QUALITY, 0),
"pref_qual_1": radio(headphones.CONFIG.PREFERRED_QUALITY, 1), "pref_qual_1": radio(headphones.CONFIG.PREFERRED_QUALITY, 1),
"pref_qual_2": radio(headphones.CONFIG.PREFERRED_QUALITY, 2), "pref_qual_2": radio(headphones.CONFIG.PREFERRED_QUALITY, 2),
@@ -1066,15 +1133,19 @@ class WebInterface(object):
"embed_album_art": checked(headphones.CONFIG.EMBED_ALBUM_ART), "embed_album_art": checked(headphones.CONFIG.EMBED_ALBUM_ART),
"embed_lyrics": checked(headphones.CONFIG.EMBED_LYRICS), "embed_lyrics": checked(headphones.CONFIG.EMBED_LYRICS),
"replace_existing_folders": checked(headphones.CONFIG.REPLACE_EXISTING_FOLDERS), "replace_existing_folders": checked(headphones.CONFIG.REPLACE_EXISTING_FOLDERS),
"keep_original_folder" : checked(headphones.CONFIG.KEEP_ORIGINAL_FOLDER),
"destination_dir": headphones.CONFIG.DESTINATION_DIR, "destination_dir": headphones.CONFIG.DESTINATION_DIR,
"lossless_destination_dir": headphones.CONFIG.LOSSLESS_DESTINATION_DIR, "lossless_destination_dir": headphones.CONFIG.LOSSLESS_DESTINATION_DIR,
"folder_format": headphones.CONFIG.FOLDER_FORMAT, "folder_format": headphones.CONFIG.FOLDER_FORMAT,
"file_format": headphones.CONFIG.FILE_FORMAT, "file_format": headphones.CONFIG.FILE_FORMAT,
"file_underscores": checked(headphones.CONFIG.FILE_UNDERSCORES), "file_underscores": checked(headphones.CONFIG.FILE_UNDERSCORES),
"include_extras": checked(headphones.CONFIG.INCLUDE_EXTRAS), "include_extras": checked(headphones.CONFIG.INCLUDE_EXTRAS),
"official_releases_only": checked(headphones.CONFIG.OFFICIAL_RELEASES_ONLY),
"wait_until_release_date": checked(headphones.CONFIG.WAIT_UNTIL_RELEASE_DATE),
"autowant_upcoming": checked(headphones.CONFIG.AUTOWANT_UPCOMING), "autowant_upcoming": checked(headphones.CONFIG.AUTOWANT_UPCOMING),
"autowant_all": checked(headphones.CONFIG.AUTOWANT_ALL), "autowant_all": checked(headphones.CONFIG.AUTOWANT_ALL),
"autowant_manually_added": checked(headphones.CONFIG.AUTOWANT_MANUALLY_ADDED), "autowant_manually_added": checked(headphones.CONFIG.AUTOWANT_MANUALLY_ADDED),
"do_not_process_unmatched": checked(headphones.CONFIG.DO_NOT_PROCESS_UNMATCHED),
"keep_torrent_files": checked(headphones.CONFIG.KEEP_TORRENT_FILES), "keep_torrent_files": checked(headphones.CONFIG.KEEP_TORRENT_FILES),
"prefer_torrents_0": radio(headphones.CONFIG.PREFER_TORRENTS, 0), "prefer_torrents_0": radio(headphones.CONFIG.PREFER_TORRENTS, 0),
"prefer_torrents_1": radio(headphones.CONFIG.PREFER_TORRENTS, 1), "prefer_torrents_1": radio(headphones.CONFIG.PREFER_TORRENTS, 1),
@@ -1120,6 +1191,7 @@ class WebInterface(object):
"plex_client_host": headphones.CONFIG.PLEX_CLIENT_HOST, "plex_client_host": headphones.CONFIG.PLEX_CLIENT_HOST,
"plex_username": headphones.CONFIG.PLEX_USERNAME, "plex_username": headphones.CONFIG.PLEX_USERNAME,
"plex_password": headphones.CONFIG.PLEX_PASSWORD, "plex_password": headphones.CONFIG.PLEX_PASSWORD,
"plex_token": headphones.CONFIG.PLEX_TOKEN,
"plex_update": checked(headphones.CONFIG.PLEX_UPDATE), "plex_update": checked(headphones.CONFIG.PLEX_UPDATE),
"plex_notify": checked(headphones.CONFIG.PLEX_NOTIFY), "plex_notify": checked(headphones.CONFIG.PLEX_NOTIFY),
"nma_enabled": checked(headphones.CONFIG.NMA_ENABLED), "nma_enabled": checked(headphones.CONFIG.NMA_ENABLED),
@@ -1214,11 +1286,12 @@ class WebInterface(object):
# Handle the variable config options. Note - keys with False values aren't getting passed # Handle the variable config options. Note - keys with False values aren't getting passed
checked_configs = [ checked_configs = [
"launch_browser", "enable_https", "api_enabled", "use_blackhole", "headphones_indexer", "use_newznab", "newznab_enabled", "launch_browser", "enable_https", "api_enabled", "use_blackhole", "headphones_indexer", "use_newznab", "newznab_enabled", "use_torznab", "torznab_enabled",
"use_nzbsorg", "use_omgwtfnzbs", "use_kat", "use_piratebay", "use_oldpiratebay", "use_mininova", "use_waffles", "use_rutracker", "use_nzbsorg", "use_omgwtfnzbs", "use_kat", "use_piratebay", "use_oldpiratebay", "use_mininova", "use_waffles", "use_rutracker",
"use_whatcd", "preferred_bitrate_allow_lossless", "detect_bitrate", "ignore_clean_releases", "freeze_db", "cue_split", "move_files", "rename_files", "use_whatcd", "use_strike", "preferred_bitrate_allow_lossless", "detect_bitrate", "ignore_clean_releases", "freeze_db", "cue_split", "move_files",
"correct_metadata", "cleanup_files", "keep_nfo", "add_album_art", "embed_album_art", "embed_lyrics", "replace_existing_folders", "rename_files", "correct_metadata", "cleanup_files", "keep_nfo", "add_album_art", "embed_album_art", "embed_lyrics",
"file_underscores", "include_extras", "autowant_upcoming", "autowant_all", "autowant_manually_added", "keep_torrent_files", "music_encoder", "replace_existing_folders", "keep_original_folder", "file_underscores", "include_extras", "official_releases_only",
"wait_until_release_date", "autowant_upcoming", "autowant_all", "autowant_manually_added", "do_not_process_unmatched", "keep_torrent_files", "music_encoder",
"encoderlossless", "encoder_multicore", "delete_lossless_files", "growl_enabled", "growl_onsnatch", "prowl_enabled", "encoderlossless", "encoder_multicore", "delete_lossless_files", "growl_enabled", "growl_onsnatch", "prowl_enabled",
"prowl_onsnatch", "xbmc_enabled", "xbmc_update", "xbmc_notify", "lms_enabled", "plex_enabled", "plex_update", "plex_notify", "prowl_onsnatch", "xbmc_enabled", "xbmc_update", "xbmc_notify", "lms_enabled", "plex_enabled", "plex_update", "plex_notify",
"nma_enabled", "nma_onsnatch", "pushalot_enabled", "pushalot_onsnatch", "synoindex_enabled", "pushover_enabled", "nma_enabled", "nma_onsnatch", "pushalot_enabled", "pushalot_onsnatch", "synoindex_enabled", "pushover_enabled",
@@ -1251,6 +1324,21 @@ class WebInterface(object):
del kwargs[key] del kwargs[key]
extra_newznabs.append((newznab_host, newznab_api, newznab_enabled)) extra_newznabs.append((newznab_host, newznab_api, newznab_enabled))
extra_torznabs = []
for kwarg in [x for x in kwargs if x.startswith('torznab_host')]:
torznab_host_key = kwarg
torznab_number = kwarg[12:]
if len(torznab_number):
torznab_api_key = 'torznab_api' + torznab_number
torznab_enabled_key = 'torznab_enabled' + torznab_number
torznab_host = kwargs.get(torznab_host_key, '')
torznab_api = kwargs.get(torznab_api_key, '')
torznab_enabled = int(kwargs.get(torznab_enabled_key, 0))
for key in [torznab_host_key, torznab_api_key, torznab_enabled_key]:
if key in kwargs:
del kwargs[key]
extra_torznabs.append((torznab_host, torznab_api, torznab_enabled))
# Convert the extras to list then string. Coming in as 0 or 1 (append new extras to the end) # Convert the extras to list then string. Coming in as 0 or 1 (append new extras to the end)
temp_extras_list = [] temp_extras_list = []
@@ -1276,11 +1364,18 @@ class WebInterface(object):
del kwargs[extra] del kwargs[extra]
headphones.CONFIG.EXTRAS = ','.join(str(n) for n in temp_extras_list) headphones.CONFIG.EXTRAS = ','.join(str(n) for n in temp_extras_list)
headphones.CONFIG.clear_extra_newznabs() headphones.CONFIG.clear_extra_newznabs()
headphones.CONFIG.clear_extra_torznabs()
headphones.CONFIG.process_kwargs(kwargs) headphones.CONFIG.process_kwargs(kwargs)
for extra_newznab in extra_newznabs: for extra_newznab in extra_newznabs:
headphones.CONFIG.add_extra_newznab(extra_newznab) headphones.CONFIG.add_extra_newznab(extra_newznab)
for extra_torznab in extra_torznabs:
headphones.CONFIG.add_extra_torznab(extra_torznab)
# Sanity checking # Sanity checking
if headphones.CONFIG.SEARCH_INTERVAL and headphones.CONFIG.SEARCH_INTERVAL < 360: if headphones.CONFIG.SEARCH_INTERVAL and headphones.CONFIG.SEARCH_INTERVAL < 360:
logger.info("Search interval too low. Resetting to 6 hour minimum") logger.info("Search interval too low. Resetting to 6 hour minimum")
@@ -1424,6 +1519,25 @@ class WebInterface(object):
logger.warn(msg) logger.warn(msg)
return msg return msg
@cherrypy.expose
def testPushover(self):
logger.info(u"Sending Pushover notification")
pushover = notifiers.PUSHOVER()
result = pushover.notify("hooray!", "This is a test")
return str(result)
@cherrypy.expose
def testPlex(self):
logger.info(u"Testing plex notifications")
plex = notifiers.Plex()
plex.notify("hellooooo", "test album!", "")
@cherrypy.expose
def testPushbullet(self):
logger.info("Testing Pushbullet notifications")
pushbullet = notifiers.PUSHBULLET()
pushbullet.notify("it works!")
class Artwork(object): class Artwork(object):
@cherrypy.expose @cherrypy.expose
def index(self): def index(self):
+77 -15
View File
@@ -17,8 +17,8 @@ http://www.crummy.com/software/BeautifulSoup/bs4/doc/
""" """
__author__ = "Leonard Richardson (leonardr@segfault.org)" __author__ = "Leonard Richardson (leonardr@segfault.org)"
__version__ = "4.3.2" __version__ = "4.4.0"
__copyright__ = "Copyright (c) 2004-2013 Leonard Richardson" __copyright__ = "Copyright (c) 2004-2015 Leonard Richardson"
__license__ = "MIT" __license__ = "MIT"
__all__ = ['BeautifulSoup'] __all__ = ['BeautifulSoup']
@@ -45,7 +45,7 @@ from .element import (
# The very first thing we do is give a useful error if someone is # The very first thing we do is give a useful error if someone is
# running this code under Python 3 without converting it. # running this code under Python 3 without converting it.
syntax_error = u'You are trying to run the Python 2 version of Beautiful Soup under Python 3. This will not work. You need to convert the code, either by installing it (`python setup.py install`) or by running 2to3 (`2to3 -w bs4`).' 'You are trying to run the Python 2 version of Beautiful Soup under Python 3. This will not work.'<>'You need to convert the code, either by installing it (`python setup.py install`) or by running 2to3 (`2to3 -w bs4`).'
class BeautifulSoup(Tag): class BeautifulSoup(Tag):
""" """
@@ -77,8 +77,11 @@ class BeautifulSoup(Tag):
ASCII_SPACES = '\x20\x0a\x09\x0c\x0d' ASCII_SPACES = '\x20\x0a\x09\x0c\x0d'
NO_PARSER_SPECIFIED_WARNING = "No parser was explicitly specified, so I'm using the best available %(markup_type)s parser for this system (\"%(parser)s\"). This usually isn't a problem, but if you run this code on another system, or in a different virtual environment, it may use a different parser and behave differently.\n\nTo get rid of this warning, change this:\n\n BeautifulSoup([your markup])\n\nto this:\n\n BeautifulSoup([your markup], \"%(parser)s\")\n"
def __init__(self, markup="", features=None, builder=None, def __init__(self, markup="", features=None, builder=None,
parse_only=None, from_encoding=None, **kwargs): parse_only=None, from_encoding=None, exclude_encodings=None,
**kwargs):
"""The Soup object is initialized as the 'root tag', and the """The Soup object is initialized as the 'root tag', and the
provided markup (which can be a string or a file-like object) provided markup (which can be a string or a file-like object)
is fed into the underlying parser.""" is fed into the underlying parser."""
@@ -114,9 +117,9 @@ class BeautifulSoup(Tag):
del kwargs['isHTML'] del kwargs['isHTML']
warnings.warn( warnings.warn(
"BS4 does not respect the isHTML argument to the " "BS4 does not respect the isHTML argument to the "
"BeautifulSoup constructor. You can pass in features='html' " "BeautifulSoup constructor. Suggest you use "
"or features='xml' to get a builder capable of handling " "features='lxml' for HTML and features='lxml-xml' for "
"one or the other.") "XML.")
def deprecated_argument(old_name, new_name): def deprecated_argument(old_name, new_name):
if old_name in kwargs: if old_name in kwargs:
@@ -140,6 +143,7 @@ class BeautifulSoup(Tag):
"__init__() got an unexpected keyword argument '%s'" % arg) "__init__() got an unexpected keyword argument '%s'" % arg)
if builder is None: if builder is None:
original_features = features
if isinstance(features, basestring): if isinstance(features, basestring):
features = [features] features = [features]
if features is None or len(features) == 0: if features is None or len(features) == 0:
@@ -151,6 +155,16 @@ class BeautifulSoup(Tag):
"requested: %s. Do you need to install a parser library?" "requested: %s. Do you need to install a parser library?"
% ",".join(features)) % ",".join(features))
builder = builder_class() builder = builder_class()
if not (original_features == builder.NAME or
original_features in builder.ALTERNATE_NAMES):
if builder.is_xml:
markup_type = "XML"
else:
markup_type = "HTML"
warnings.warn(self.NO_PARSER_SPECIFIED_WARNING % dict(
parser=builder.NAME,
markup_type=markup_type))
self.builder = builder self.builder = builder
self.is_xml = builder.is_xml self.is_xml = builder.is_xml
self.builder.soup = self self.builder.soup = self
@@ -178,6 +192,8 @@ class BeautifulSoup(Tag):
# system. Just let it go. # system. Just let it go.
pass pass
if is_file: if is_file:
if isinstance(markup, unicode):
markup = markup.encode("utf8")
warnings.warn( warnings.warn(
'"%s" looks like a filename, not markup. You should probably open this file and pass the filehandle into Beautiful Soup.' % markup) '"%s" looks like a filename, not markup. You should probably open this file and pass the filehandle into Beautiful Soup.' % markup)
if markup[:5] == "http:" or markup[:6] == "https:": if markup[:5] == "http:" or markup[:6] == "https:":
@@ -185,12 +201,15 @@ class BeautifulSoup(Tag):
# Python 3 otherwise. # Python 3 otherwise.
if ((isinstance(markup, bytes) and not b' ' in markup) if ((isinstance(markup, bytes) and not b' ' in markup)
or (isinstance(markup, unicode) and not u' ' in markup)): or (isinstance(markup, unicode) and not u' ' in markup)):
if isinstance(markup, unicode):
markup = markup.encode("utf8")
warnings.warn( warnings.warn(
'"%s" looks like a URL. Beautiful Soup is not an HTTP client. You should probably use an HTTP client to get the document behind the URL, and feed that document to Beautiful Soup.' % markup) '"%s" looks like a URL. Beautiful Soup is not an HTTP client. You should probably use an HTTP client to get the document behind the URL, and feed that document to Beautiful Soup.' % markup)
for (self.markup, self.original_encoding, self.declared_html_encoding, for (self.markup, self.original_encoding, self.declared_html_encoding,
self.contains_replacement_characters) in ( self.contains_replacement_characters) in (
self.builder.prepare_markup(markup, from_encoding)): self.builder.prepare_markup(
markup, from_encoding, exclude_encodings=exclude_encodings)):
self.reset() self.reset()
try: try:
self._feed() self._feed()
@@ -203,6 +222,16 @@ class BeautifulSoup(Tag):
self.markup = None self.markup = None
self.builder.soup = None self.builder.soup = None
def __copy__(self):
return type(self)(self.encode(), builder=self.builder)
def __getstate__(self):
# Frequently a tree builder can't be pickled.
d = dict(self.__dict__)
if 'builder' in d and not self.builder.picklable:
del d['builder']
return d
def _feed(self): def _feed(self):
# Convert the document to Unicode. # Convert the document to Unicode.
self.builder.reset() self.builder.reset()
@@ -229,9 +258,7 @@ class BeautifulSoup(Tag):
def new_string(self, s, subclass=NavigableString): def new_string(self, s, subclass=NavigableString):
"""Create a new NavigableString associated with this soup.""" """Create a new NavigableString associated with this soup."""
navigable = subclass(s) return subclass(s)
navigable.setup()
return navigable
def insert_before(self, successor): def insert_before(self, successor):
raise NotImplementedError("BeautifulSoup objects don't support insert_before().") raise NotImplementedError("BeautifulSoup objects don't support insert_before().")
@@ -290,14 +317,49 @@ class BeautifulSoup(Tag):
def object_was_parsed(self, o, parent=None, most_recent_element=None): def object_was_parsed(self, o, parent=None, most_recent_element=None):
"""Add an object to the parse tree.""" """Add an object to the parse tree."""
parent = parent or self.currentTag parent = parent or self.currentTag
most_recent_element = most_recent_element or self._most_recent_element previous_element = most_recent_element or self._most_recent_element
o.setup(parent, most_recent_element)
next_element = previous_sibling = next_sibling = None
if isinstance(o, Tag):
next_element = o.next_element
next_sibling = o.next_sibling
previous_sibling = o.previous_sibling
if not previous_element:
previous_element = o.previous_element
o.setup(parent, previous_element, next_element, previous_sibling, next_sibling)
if most_recent_element is not None:
most_recent_element.next_element = o
self._most_recent_element = o self._most_recent_element = o
parent.contents.append(o) parent.contents.append(o)
if parent.next_sibling:
# This node is being inserted into an element that has
# already been parsed. Deal with any dangling references.
index = parent.contents.index(o)
if index == 0:
previous_element = parent
previous_sibling = None
else:
previous_element = previous_sibling = parent.contents[index-1]
if index == len(parent.contents)-1:
next_element = parent.next_sibling
next_sibling = None
else:
next_element = next_sibling = parent.contents[index+1]
o.previous_element = previous_element
if previous_element:
previous_element.next_element = o
o.next_element = next_element
if next_element:
next_element.previous_element = o
o.next_sibling = next_sibling
if next_sibling:
next_sibling.previous_sibling = o
o.previous_sibling = previous_sibling
if previous_sibling:
previous_sibling.next_sibling = o
def _popToTag(self, name, nsprefix=None, inclusivePop=True): def _popToTag(self, name, nsprefix=None, inclusivePop=True):
"""Pops the tag stack up to and including the most recent """Pops the tag stack up to and including the most recent
instance of the given tag. If inclusivePop is false, pops the tag instance of the given tag. If inclusivePop is false, pops the tag
+3
View File
@@ -80,9 +80,12 @@ builder_registry = TreeBuilderRegistry()
class TreeBuilder(object): class TreeBuilder(object):
"""Turn a document into a Beautiful Soup object tree.""" """Turn a document into a Beautiful Soup object tree."""
NAME = "[Unknown tree builder]"
ALTERNATE_NAMES = []
features = [] features = []
is_xml = False is_xml = False
picklable = False
preserve_whitespace_tags = set() preserve_whitespace_tags = set()
empty_element_tags = None # A tag will be considered an empty-element empty_element_tags = None # A tag will be considered an empty-element
# tag when and only when it has no contents. # tag when and only when it has no contents.
+51 -7
View File
@@ -2,6 +2,7 @@ __all__ = [
'HTML5TreeBuilder', 'HTML5TreeBuilder',
] ]
from pdb import set_trace
import warnings import warnings
from bs4.builder import ( from bs4.builder import (
PERMISSIVE, PERMISSIVE,
@@ -9,7 +10,10 @@ from bs4.builder import (
HTML_5, HTML_5,
HTMLTreeBuilder, HTMLTreeBuilder,
) )
from bs4.element import NamespacedAttribute from bs4.element import (
NamespacedAttribute,
whitespace_re,
)
import html5lib import html5lib
from html5lib.constants import namespaces from html5lib.constants import namespaces
from bs4.element import ( from bs4.element import (
@@ -22,11 +26,20 @@ from bs4.element import (
class HTML5TreeBuilder(HTMLTreeBuilder): class HTML5TreeBuilder(HTMLTreeBuilder):
"""Use html5lib to build a tree.""" """Use html5lib to build a tree."""
features = ['html5lib', PERMISSIVE, HTML_5, HTML] NAME = "html5lib"
def prepare_markup(self, markup, user_specified_encoding): features = [NAME, PERMISSIVE, HTML_5, HTML]
def prepare_markup(self, markup, user_specified_encoding,
document_declared_encoding=None, exclude_encodings=None):
# Store the user-specified encoding for use later on. # Store the user-specified encoding for use later on.
self.user_specified_encoding = user_specified_encoding self.user_specified_encoding = user_specified_encoding
# document_declared_encoding and exclude_encodings aren't used
# ATM because the html5lib TreeBuilder doesn't use
# UnicodeDammit.
if exclude_encodings:
warnings.warn("You provided a value for exclude_encoding, but the html5lib tree builder doesn't support exclude_encoding.")
yield (markup, None, None, False) yield (markup, None, None, False)
# These methods are defined by Beautiful Soup. # These methods are defined by Beautiful Soup.
@@ -101,7 +114,13 @@ class AttrList(object):
def __iter__(self): def __iter__(self):
return list(self.attrs.items()).__iter__() return list(self.attrs.items()).__iter__()
def __setitem__(self, name, value): def __setitem__(self, name, value):
"set attr", name, value # If this attribute is a multi-valued attribute for this element,
# turn its value into a list.
list_attr = HTML5TreeBuilder.cdata_list_attributes
if (name in list_attr['*']
or (self.element.name in list_attr
and name in list_attr[self.element.name])):
value = whitespace_re.split(value)
self.element[name] = value self.element[name] = value
def items(self): def items(self):
return list(self.attrs.items()) return list(self.attrs.items())
@@ -161,6 +180,12 @@ class Element(html5lib.treebuilders._base.Node):
# immediately after the parent, if it has no children.) # immediately after the parent, if it has no children.)
if self.element.contents: if self.element.contents:
most_recent_element = self.element._last_descendant(False) most_recent_element = self.element._last_descendant(False)
elif self.element.next_element is not None:
# Something from further ahead in the parse tree is
# being inserted into this earlier element. This is
# very annoying because it means an expensive search
# for the last element in the tree.
most_recent_element = self.soup._last_descendant()
else: else:
most_recent_element = self.element most_recent_element = self.element
@@ -172,6 +197,7 @@ class Element(html5lib.treebuilders._base.Node):
return AttrList(self.element) return AttrList(self.element)
def setAttributes(self, attributes): def setAttributes(self, attributes):
if attributes is not None and len(attributes) > 0: if attributes is not None and len(attributes) > 0:
converted_attributes = [] converted_attributes = []
@@ -218,6 +244,9 @@ class Element(html5lib.treebuilders._base.Node):
def reparentChildren(self, new_parent): def reparentChildren(self, new_parent):
"""Move all of this tag's children into another tag.""" """Move all of this tag's children into another tag."""
# print "MOVE", self.element.contents
# print "FROM", self.element
# print "TO", new_parent.element
element = self.element element = self.element
new_parent_element = new_parent.element new_parent_element = new_parent.element
# Determine what this tag's next_element will be once all the children # Determine what this tag's next_element will be once all the children
@@ -236,17 +265,28 @@ class Element(html5lib.treebuilders._base.Node):
new_parents_last_descendant_next_element = new_parent_element.next_element new_parents_last_descendant_next_element = new_parent_element.next_element
to_append = element.contents to_append = element.contents
append_after = new_parent.element.contents append_after = new_parent_element.contents
if len(to_append) > 0: if len(to_append) > 0:
# Set the first child's previous_element and previous_sibling # Set the first child's previous_element and previous_sibling
# to elements within the new parent # to elements within the new parent
first_child = to_append[0] first_child = to_append[0]
first_child.previous_element = new_parents_last_descendant if new_parents_last_descendant:
first_child.previous_element = new_parents_last_descendant
else:
first_child.previous_element = new_parent_element
first_child.previous_sibling = new_parents_last_child first_child.previous_sibling = new_parents_last_child
if new_parents_last_descendant:
new_parents_last_descendant.next_element = first_child
else:
new_parent_element.next_element = first_child
if new_parents_last_child:
new_parents_last_child.next_sibling = first_child
# Fix the last child's next_element and next_sibling # Fix the last child's next_element and next_sibling
last_child = to_append[-1] last_child = to_append[-1]
last_child.next_element = new_parents_last_descendant_next_element last_child.next_element = new_parents_last_descendant_next_element
if new_parents_last_descendant_next_element:
new_parents_last_descendant_next_element.previous_element = last_child
last_child.next_sibling = None last_child.next_sibling = None
for child in to_append: for child in to_append:
@@ -257,6 +297,10 @@ class Element(html5lib.treebuilders._base.Node):
element.contents = [] element.contents = []
element.next_element = final_next_element element.next_element = final_next_element
# print "DONE WITH MOVE"
# print "FROM", self.element
# print "TO", new_parent_element
def cloneNode(self): def cloneNode(self):
tag = self.soup.new_tag(self.element.name, self.namespace) tag = self.soup.new_tag(self.element.name, self.namespace)
node = Element(tag, self.soup, self.namespace) node = Element(tag, self.soup, self.namespace)
@@ -268,7 +312,7 @@ class Element(html5lib.treebuilders._base.Node):
return self.element.contents return self.element.contents
def getNameTuple(self): def getNameTuple(self):
if self.namespace is None: if self.namespace == None:
return namespaces["html"], self.name return namespaces["html"], self.name
else: else:
return self.namespace, self.name return self.namespace, self.name
+25 -21
View File
@@ -4,10 +4,16 @@ __all__ = [
'HTMLParserTreeBuilder', 'HTMLParserTreeBuilder',
] ]
from HTMLParser import ( from HTMLParser import HTMLParser
HTMLParser,
HTMLParseError, try:
) from HTMLParser import HTMLParseError
except ImportError, e:
# HTMLParseError is removed in Python 3.5. Since it can never be
# thrown in 3.5, we can just define our own class as a placeholder.
class HTMLParseError(Exception):
pass
import sys import sys
import warnings import warnings
@@ -19,10 +25,10 @@ import warnings
# At the end of this file, we monkeypatch HTMLParser so that # At the end of this file, we monkeypatch HTMLParser so that
# strict=True works well on Python 3.2.2. # strict=True works well on Python 3.2.2.
major, minor, release = sys.version_info[:3] major, minor, release = sys.version_info[:3]
CONSTRUCTOR_TAKES_STRICT = ( CONSTRUCTOR_TAKES_STRICT = major == 3 and minor == 2 and release >= 3
major > 3 CONSTRUCTOR_STRICT_IS_DEPRECATED = major == 3 and minor == 3
or (major == 3 and minor > 2) CONSTRUCTOR_TAKES_CONVERT_CHARREFS = major == 3 and minor >= 4
or (major == 3 and minor == 2 and release >= 3))
from bs4.element import ( from bs4.element import (
CData, CData,
@@ -63,7 +69,8 @@ class BeautifulSoupHTMLParser(HTMLParser):
def handle_charref(self, name): def handle_charref(self, name):
# XXX workaround for a bug in HTMLParser. Remove this once # XXX workaround for a bug in HTMLParser. Remove this once
# it's fixed. # it's fixed in all supported versions.
# http://bugs.python.org/issue13633
if name.startswith('x'): if name.startswith('x'):
real_name = int(name.lstrip('x'), 16) real_name = int(name.lstrip('x'), 16)
elif name.startswith('X'): elif name.startswith('X'):
@@ -113,14 +120,6 @@ class BeautifulSoupHTMLParser(HTMLParser):
def handle_pi(self, data): def handle_pi(self, data):
self.soup.endData() self.soup.endData()
if data.endswith("?") and data.lower().startswith("xml"):
# "An XHTML processing instruction using the trailing '?'
# will cause the '?' to be included in data." - HTMLParser
# docs.
#
# Strip the question mark so we don't end up with two
# question marks.
data = data[:-1]
self.soup.handle_data(data) self.soup.handle_data(data)
self.soup.endData(ProcessingInstruction) self.soup.endData(ProcessingInstruction)
@@ -128,15 +127,19 @@ class BeautifulSoupHTMLParser(HTMLParser):
class HTMLParserTreeBuilder(HTMLTreeBuilder): class HTMLParserTreeBuilder(HTMLTreeBuilder):
is_xml = False is_xml = False
features = [HTML, STRICT, HTMLPARSER] picklable = True
NAME = HTMLPARSER
features = [NAME, HTML, STRICT]
def __init__(self, *args, **kwargs): def __init__(self, *args, **kwargs):
if CONSTRUCTOR_TAKES_STRICT: if CONSTRUCTOR_TAKES_STRICT and not CONSTRUCTOR_STRICT_IS_DEPRECATED:
kwargs['strict'] = False kwargs['strict'] = False
if CONSTRUCTOR_TAKES_CONVERT_CHARREFS:
kwargs['convert_charrefs'] = False
self.parser_args = (args, kwargs) self.parser_args = (args, kwargs)
def prepare_markup(self, markup, user_specified_encoding=None, def prepare_markup(self, markup, user_specified_encoding=None,
document_declared_encoding=None): document_declared_encoding=None, exclude_encodings=None):
""" """
:return: A 4-tuple (markup, original encoding, encoding :return: A 4-tuple (markup, original encoding, encoding
declared within markup, whether any characters had to be declared within markup, whether any characters had to be
@@ -147,7 +150,8 @@ class HTMLParserTreeBuilder(HTMLTreeBuilder):
return return
try_encodings = [user_specified_encoding, document_declared_encoding] try_encodings = [user_specified_encoding, document_declared_encoding]
dammit = UnicodeDammit(markup, try_encodings, is_html=True) dammit = UnicodeDammit(markup, try_encodings, is_html=True,
exclude_encodings=exclude_encodings)
yield (dammit.markup, dammit.original_encoding, yield (dammit.markup, dammit.original_encoding,
dammit.declared_html_encoding, dammit.declared_html_encoding,
dammit.contains_replacement_characters) dammit.contains_replacement_characters)
+20 -5
View File
@@ -7,7 +7,12 @@ from io import BytesIO
from StringIO import StringIO from StringIO import StringIO
import collections import collections
from lxml import etree from lxml import etree
from bs4.element import Comment, Doctype, NamespacedAttribute from bs4.element import (
Comment,
Doctype,
NamespacedAttribute,
ProcessingInstruction,
)
from bs4.builder import ( from bs4.builder import (
FAST, FAST,
HTML, HTML,
@@ -25,8 +30,11 @@ class LXMLTreeBuilderForXML(TreeBuilder):
is_xml = True is_xml = True
NAME = "lxml-xml"
ALTERNATE_NAMES = ["xml"]
# Well, it's permissive by XML parser standards. # Well, it's permissive by XML parser standards.
features = [LXML, XML, FAST, PERMISSIVE] features = [NAME, LXML, XML, FAST, PERMISSIVE]
CHUNK_SIZE = 512 CHUNK_SIZE = 512
@@ -70,6 +78,7 @@ class LXMLTreeBuilderForXML(TreeBuilder):
return (None, tag) return (None, tag)
def prepare_markup(self, markup, user_specified_encoding=None, def prepare_markup(self, markup, user_specified_encoding=None,
exclude_encodings=None,
document_declared_encoding=None): document_declared_encoding=None):
""" """
:yield: A series of 4-tuples. :yield: A series of 4-tuples.
@@ -95,7 +104,8 @@ class LXMLTreeBuilderForXML(TreeBuilder):
# the document as each one in turn. # the document as each one in turn.
is_html = not self.is_xml is_html = not self.is_xml
try_encodings = [user_specified_encoding, document_declared_encoding] try_encodings = [user_specified_encoding, document_declared_encoding]
detector = EncodingDetector(markup, try_encodings, is_html) detector = EncodingDetector(
markup, try_encodings, is_html, exclude_encodings)
for encoding in detector.encodings: for encoding in detector.encodings:
yield (detector.markup, encoding, document_declared_encoding, False) yield (detector.markup, encoding, document_declared_encoding, False)
@@ -189,7 +199,9 @@ class LXMLTreeBuilderForXML(TreeBuilder):
self.nsmaps.pop() self.nsmaps.pop()
def pi(self, target, data): def pi(self, target, data):
pass self.soup.endData()
self.soup.handle_data(target + ' ' + data)
self.soup.endData(ProcessingInstruction)
def data(self, content): def data(self, content):
self.soup.handle_data(content) self.soup.handle_data(content)
@@ -212,7 +224,10 @@ class LXMLTreeBuilderForXML(TreeBuilder):
class LXMLTreeBuilder(HTMLTreeBuilder, LXMLTreeBuilderForXML): class LXMLTreeBuilder(HTMLTreeBuilder, LXMLTreeBuilderForXML):
features = [LXML, HTML, FAST, PERMISSIVE] NAME = LXML
ALTERNATE_NAMES = ["lxml-html"]
features = ALTERNATE_NAMES + [NAME, HTML, FAST, PERMISSIVE]
is_xml = False is_xml = False
def default_parser(self, encoding): def default_parser(self, encoding):
+15 -5
View File
@@ -3,10 +3,11 @@
This library converts a bytestream to Unicode through any means This library converts a bytestream to Unicode through any means
necessary. It is heavily based on code from Mark Pilgrim's Universal necessary. It is heavily based on code from Mark Pilgrim's Universal
Feed Parser. It works best on XML and XML, but it does not rewrite the Feed Parser. It works best on XML and HTML, but it does not rewrite the
XML or HTML to reflect a new encoding; that's the tree builder's job. XML or HTML to reflect a new encoding; that's the tree builder's job.
""" """
from pdb import set_trace
import codecs import codecs
from htmlentitydefs import codepoint2name from htmlentitydefs import codepoint2name
import re import re
@@ -212,8 +213,11 @@ class EncodingDetector:
5. Windows-1252. 5. Windows-1252.
""" """
def __init__(self, markup, override_encodings=None, is_html=False): def __init__(self, markup, override_encodings=None, is_html=False,
exclude_encodings=None):
self.override_encodings = override_encodings or [] self.override_encodings = override_encodings or []
exclude_encodings = exclude_encodings or []
self.exclude_encodings = set([x.lower() for x in exclude_encodings])
self.chardet_encoding = None self.chardet_encoding = None
self.is_html = is_html self.is_html = is_html
self.declared_encoding = None self.declared_encoding = None
@@ -224,6 +228,8 @@ class EncodingDetector:
def _usable(self, encoding, tried): def _usable(self, encoding, tried):
if encoding is not None: if encoding is not None:
encoding = encoding.lower() encoding = encoding.lower()
if encoding in self.exclude_encodings:
return False
if encoding not in tried: if encoding not in tried:
tried.add(encoding) tried.add(encoding)
return True return True
@@ -266,6 +272,9 @@ class EncodingDetector:
def strip_byte_order_mark(cls, data): def strip_byte_order_mark(cls, data):
"""If a byte-order mark is present, strip it and return the encoding it implies.""" """If a byte-order mark is present, strip it and return the encoding it implies."""
encoding = None encoding = None
if isinstance(data, unicode):
# Unicode data cannot have a byte-order mark.
return data, encoding
if (len(data) >= 4) and (data[:2] == b'\xfe\xff') \ if (len(data) >= 4) and (data[:2] == b'\xfe\xff') \
and (data[2:4] != '\x00\x00'): and (data[2:4] != '\x00\x00'):
encoding = 'utf-16be' encoding = 'utf-16be'
@@ -306,7 +315,7 @@ class EncodingDetector:
declared_encoding_match = html_meta_re.search(markup, endpos=html_endpos) declared_encoding_match = html_meta_re.search(markup, endpos=html_endpos)
if declared_encoding_match is not None: if declared_encoding_match is not None:
declared_encoding = declared_encoding_match.groups()[0].decode( declared_encoding = declared_encoding_match.groups()[0].decode(
'ascii') 'ascii', 'replace')
if declared_encoding: if declared_encoding:
return declared_encoding.lower() return declared_encoding.lower()
return None return None
@@ -331,13 +340,14 @@ class UnicodeDammit:
] ]
def __init__(self, markup, override_encodings=[], def __init__(self, markup, override_encodings=[],
smart_quotes_to=None, is_html=False): smart_quotes_to=None, is_html=False, exclude_encodings=[]):
self.smart_quotes_to = smart_quotes_to self.smart_quotes_to = smart_quotes_to
self.tried_encodings = [] self.tried_encodings = []
self.contains_replacement_characters = False self.contains_replacement_characters = False
self.is_html = is_html self.is_html = is_html
self.detector = EncodingDetector(markup, override_encodings, is_html) self.detector = EncodingDetector(
markup, override_encodings, is_html, exclude_encodings)
# Short-circuit if the data is in Unicode to begin with. # Short-circuit if the data is in Unicode to begin with.
if isinstance(markup, unicode) or markup == '': if isinstance(markup, unicode) or markup == '':
+13 -4
View File
@@ -33,12 +33,21 @@ def diagnose(data):
if 'lxml' in basic_parsers: if 'lxml' in basic_parsers:
basic_parsers.append(["lxml", "xml"]) basic_parsers.append(["lxml", "xml"])
from lxml import etree try:
print "Found lxml version %s" % ".".join(map(str,etree.LXML_VERSION)) from lxml import etree
print "Found lxml version %s" % ".".join(map(str,etree.LXML_VERSION))
except ImportError, e:
print (
"lxml is not installed or couldn't be imported.")
if 'html5lib' in basic_parsers: if 'html5lib' in basic_parsers:
import html5lib try:
print "Found html5lib version %s" % html5lib.__version__ import html5lib
print "Found html5lib version %s" % html5lib.__version__
except ImportError, e:
print (
"html5lib is not installed or couldn't be imported.")
if hasattr(data, 'read'): if hasattr(data, 'read'):
data = data.read() data = data.read()
+271 -169
View File
@@ -1,3 +1,4 @@
from pdb import set_trace
import collections import collections
import re import re
import sys import sys
@@ -185,24 +186,40 @@ class PageElement(object):
return self.HTML_FORMATTERS.get( return self.HTML_FORMATTERS.get(
name, HTMLAwareEntitySubstitution.substitute_xml) name, HTMLAwareEntitySubstitution.substitute_xml)
def setup(self, parent=None, previous_element=None): def setup(self, parent=None, previous_element=None, next_element=None,
previous_sibling=None, next_sibling=None):
"""Sets up the initial relations between this element and """Sets up the initial relations between this element and
other elements.""" other elements."""
self.parent = parent self.parent = parent
self.previous_element = previous_element self.previous_element = previous_element
if previous_element is not None: if previous_element is not None:
self.previous_element.next_element = self self.previous_element.next_element = self
self.next_element = None
self.previous_sibling = None self.next_element = next_element
self.next_sibling = None if self.next_element:
if self.parent is not None and self.parent.contents: self.next_element.previous_element = self
self.previous_sibling = self.parent.contents[-1]
self.next_sibling = next_sibling
if self.next_sibling:
self.next_sibling.previous_sibling = self
if (not previous_sibling
and self.parent is not None and self.parent.contents):
previous_sibling = self.parent.contents[-1]
self.previous_sibling = previous_sibling
if previous_sibling:
self.previous_sibling.next_sibling = self self.previous_sibling.next_sibling = self
nextSibling = _alias("next_sibling") # BS3 nextSibling = _alias("next_sibling") # BS3
previousSibling = _alias("previous_sibling") # BS3 previousSibling = _alias("previous_sibling") # BS3
def replace_with(self, replace_with): def replace_with(self, replace_with):
if not self.parent:
raise ValueError(
"Cannot replace one element with another when the"
"element to be replaced is not part of a tree.")
if replace_with is self: if replace_with is self:
return return
if replace_with is self.parent: if replace_with is self.parent:
@@ -216,6 +233,10 @@ class PageElement(object):
def unwrap(self): def unwrap(self):
my_parent = self.parent my_parent = self.parent
if not self.parent:
raise ValueError(
"Cannot replace an element with its contents when that"
"element is not part of a tree.")
my_index = self.parent.index(self) my_index = self.parent.index(self)
self.extract() self.extract()
for child in reversed(self.contents[:]): for child in reversed(self.contents[:]):
@@ -240,17 +261,20 @@ class PageElement(object):
last_child = self._last_descendant() last_child = self._last_descendant()
next_element = last_child.next_element next_element = last_child.next_element
if self.previous_element is not None: if (self.previous_element is not None and
self.previous_element != next_element):
self.previous_element.next_element = next_element self.previous_element.next_element = next_element
if next_element is not None: if next_element is not None and next_element != self.previous_element:
next_element.previous_element = self.previous_element next_element.previous_element = self.previous_element
self.previous_element = None self.previous_element = None
last_child.next_element = None last_child.next_element = None
self.parent = None self.parent = None
if self.previous_sibling is not None: if (self.previous_sibling is not None
and self.previous_sibling != self.next_sibling):
self.previous_sibling.next_sibling = self.next_sibling self.previous_sibling.next_sibling = self.next_sibling
if self.next_sibling is not None: if (self.next_sibling is not None
and self.next_sibling != self.previous_sibling):
self.next_sibling.previous_sibling = self.previous_sibling self.next_sibling.previous_sibling = self.previous_sibling
self.previous_sibling = self.next_sibling = None self.previous_sibling = self.next_sibling = None
return self return self
@@ -478,6 +502,10 @@ class PageElement(object):
def _find_all(self, name, attrs, text, limit, generator, **kwargs): def _find_all(self, name, attrs, text, limit, generator, **kwargs):
"Iterates over a generator looking for things that match." "Iterates over a generator looking for things that match."
if text is None and 'string' in kwargs:
text = kwargs['string']
del kwargs['string']
if isinstance(name, SoupStrainer): if isinstance(name, SoupStrainer):
strainer = name strainer = name
else: else:
@@ -548,17 +576,17 @@ class PageElement(object):
# Methods for supporting CSS selectors. # Methods for supporting CSS selectors.
tag_name_re = re.compile('^[a-z0-9]+$') tag_name_re = re.compile('^[a-zA-Z0-9][-.a-zA-Z0-9:_]*$')
# /^(\w+)\[(\w+)([=~\|\^\$\*]?)=?"?([^\]"]*)"?\]$/ # /^([a-zA-Z0-9][-.a-zA-Z0-9:_]*)\[(\w+)([=~\|\^\$\*]?)=?"?([^\]"]*)"?\]$/
# \---/ \---/\-------------/ \-------/ # \---------------------------/ \---/\-------------/ \-------/
# | | | | # | | | |
# | | | The value # | | | The value
# | | ~,|,^,$,* or = # | | ~,|,^,$,* or =
# | Attribute # | Attribute
# Tag # Tag
attribselect_re = re.compile( attribselect_re = re.compile(
r'^(?P<tag>\w+)?\[(?P<attribute>\w+)(?P<operator>[=~\|\^\$\*]?)' + r'^(?P<tag>[a-zA-Z0-9][-.a-zA-Z0-9:_]*)?\[(?P<attribute>[\w-]+)(?P<operator>[=~\|\^\$\*]?)' +
r'=?"?(?P<value>[^\]"]*)"?\]$' r'=?"?(?P<value>[^\]"]*)"?\]$'
) )
@@ -654,11 +682,17 @@ class NavigableString(unicode, PageElement):
how to handle non-ASCII characters. how to handle non-ASCII characters.
""" """
if isinstance(value, unicode): if isinstance(value, unicode):
return unicode.__new__(cls, value) u = unicode.__new__(cls, value)
return unicode.__new__(cls, value, DEFAULT_OUTPUT_ENCODING) else:
u = unicode.__new__(cls, value, DEFAULT_OUTPUT_ENCODING)
u.setup()
return u
def __copy__(self): def __copy__(self):
return self """A copy of a NavigableString has the same contents and class
as the original, but it is not connected to the parse tree.
"""
return type(self)(self)
def __getnewargs__(self): def __getnewargs__(self):
return (unicode(self),) return (unicode(self),)
@@ -707,7 +741,7 @@ class CData(PreformattedString):
class ProcessingInstruction(PreformattedString): class ProcessingInstruction(PreformattedString):
PREFIX = u'<?' PREFIX = u'<?'
SUFFIX = u'?>' SUFFIX = u'>'
class Comment(PreformattedString): class Comment(PreformattedString):
@@ -759,9 +793,12 @@ class Tag(PageElement):
self.prefix = prefix self.prefix = prefix
if attrs is None: if attrs is None:
attrs = {} attrs = {}
elif attrs and builder.cdata_list_attributes: elif attrs:
attrs = builder._replace_cdata_list_attribute_values( if builder is not None and builder.cdata_list_attributes:
self.name, attrs) attrs = builder._replace_cdata_list_attribute_values(
self.name, attrs)
else:
attrs = dict(attrs)
else: else:
attrs = dict(attrs) attrs = dict(attrs)
self.attrs = attrs self.attrs = attrs
@@ -778,6 +815,18 @@ class Tag(PageElement):
parserClass = _alias("parser_class") # BS3 parserClass = _alias("parser_class") # BS3
def __copy__(self):
"""A copy of a Tag is a new Tag, unconnected to the parse tree.
Its contents are a copy of the old Tag's contents.
"""
clone = type(self)(None, self.builder, self.name, self.namespace,
self.nsprefix, self.attrs)
for attr in ('can_be_empty_element', 'hidden'):
setattr(clone, attr, getattr(self, attr))
for child in self.contents:
clone.append(child.__copy__())
return clone
@property @property
def is_empty_element(self): def is_empty_element(self):
"""Is this tag an empty-element tag? (aka a self-closing tag) """Is this tag an empty-element tag? (aka a self-closing tag)
@@ -971,15 +1020,25 @@ class Tag(PageElement):
as defined in __eq__.""" as defined in __eq__."""
return not self == other return not self == other
def __repr__(self, encoding=DEFAULT_OUTPUT_ENCODING): def __repr__(self, encoding="unicode-escape"):
"""Renders this tag as a string.""" """Renders this tag as a string."""
return self.encode(encoding) if PY3K:
# "The return value must be a string object", i.e. Unicode
return self.decode()
else:
# "The return value must be a string object", i.e. a bytestring.
# By convention, the return value of __repr__ should also be
# an ASCII string.
return self.encode(encoding)
def __unicode__(self): def __unicode__(self):
return self.decode() return self.decode()
def __str__(self): def __str__(self):
return self.encode() if PY3K:
return self.decode()
else:
return self.encode()
if PY3K: if PY3K:
__str__ = __repr__ = __unicode__ __str__ = __repr__ = __unicode__
@@ -1103,12 +1162,18 @@ class Tag(PageElement):
formatter="minimal"): formatter="minimal"):
"""Renders the contents of this tag as a Unicode string. """Renders the contents of this tag as a Unicode string.
:param indent_level: Each line of the rendering will be
indented this many spaces.
:param eventual_encoding: The tag is destined to be :param eventual_encoding: The tag is destined to be
encoded into this encoding. This method is _not_ encoded into this encoding. This method is _not_
responsible for performing that encoding. This information responsible for performing that encoding. This information
is passed in so that it can be substituted in if the is passed in so that it can be substituted in if the
document contains a <META> tag that mentions the document's document contains a <META> tag that mentions the document's
encoding. encoding.
:param formatter: The output formatter responsible for converting
entities to Unicode characters.
""" """
# First off, turn a string formatter into a function. This # First off, turn a string formatter into a function. This
# will stop the lookup from happening over and over again. # will stop the lookup from happening over and over again.
@@ -1137,7 +1202,17 @@ class Tag(PageElement):
def encode_contents( def encode_contents(
self, indent_level=None, encoding=DEFAULT_OUTPUT_ENCODING, self, indent_level=None, encoding=DEFAULT_OUTPUT_ENCODING,
formatter="minimal"): formatter="minimal"):
"""Renders the contents of this tag as a bytestring.""" """Renders the contents of this tag as a bytestring.
:param indent_level: Each line of the rendering will be
indented this many spaces.
:param eventual_encoding: The bytestring will be in this encoding.
:param formatter: The output formatter responsible for converting
entities to Unicode characters.
"""
contents = self.decode_contents(indent_level, encoding, formatter) contents = self.decode_contents(indent_level, encoding, formatter)
return contents.encode(encoding) return contents.encode(encoding)
@@ -1201,63 +1276,89 @@ class Tag(PageElement):
_selector_combinators = ['>', '+', '~'] _selector_combinators = ['>', '+', '~']
_select_debug = False _select_debug = False
def select(self, selector, _candidate_generator=None): def select_one(self, selector):
"""Perform a CSS selection operation on the current element.""" """Perform a CSS selection operation on the current element."""
tokens = selector.split() value = self.select(selector, limit=1)
if value:
return value[0]
return None
def select(self, selector, _candidate_generator=None, limit=None):
"""Perform a CSS selection operation on the current element."""
# Remove whitespace directly after the grouping operator ','
# then split into tokens.
tokens = re.sub(',[\s]*',',', selector).split()
current_context = [self] current_context = [self]
if tokens[-1] in self._selector_combinators: if tokens[-1] in self._selector_combinators:
raise ValueError( raise ValueError(
'Final combinator "%s" is missing an argument.' % tokens[-1]) 'Final combinator "%s" is missing an argument.' % tokens[-1])
if self._select_debug: if self._select_debug:
print 'Running CSS selector "%s"' % selector print 'Running CSS selector "%s"' % selector
for index, token in enumerate(tokens):
if self._select_debug: for index, token_group in enumerate(tokens):
print ' Considering token "%s"' % token new_context = []
recursive_candidate_generator = None new_context_ids = set([])
tag_name = None
# Grouping selectors, ie: p,a
grouped_tokens = token_group.split(',')
if '' in grouped_tokens:
raise ValueError('Invalid group selection syntax: %s' % token_group)
if tokens[index-1] in self._selector_combinators: if tokens[index-1] in self._selector_combinators:
# This token was consumed by the previous combinator. Skip it. # This token was consumed by the previous combinator. Skip it.
if self._select_debug: if self._select_debug:
print ' Token was consumed by the previous combinator.' print ' Token was consumed by the previous combinator.'
continue continue
# Each operation corresponds to a checker function, a rule
# for determining whether a candidate matches the
# selector. Candidates are generated by the active
# iterator.
checker = None
m = self.attribselect_re.match(token) for token in grouped_tokens:
if m is not None: if self._select_debug:
# Attribute selector print ' Considering token "%s"' % token
tag_name, attribute, operator, value = m.groups() recursive_candidate_generator = None
checker = self._attribute_checker(operator, attribute, value) tag_name = None
elif '#' in token: # Each operation corresponds to a checker function, a rule
# ID selector # for determining whether a candidate matches the
tag_name, tag_id = token.split('#', 1) # selector. Candidates are generated by the active
def id_matches(tag): # iterator.
return tag.get('id', None) == tag_id checker = None
checker = id_matches
elif '.' in token: m = self.attribselect_re.match(token)
# Class selector if m is not None:
tag_name, klass = token.split('.', 1) # Attribute selector
classes = set(klass.split('.')) tag_name, attribute, operator, value = m.groups()
def classes_match(candidate): checker = self._attribute_checker(operator, attribute, value)
return classes.issubset(candidate.get('class', []))
checker = classes_match
elif ':' in token: elif '#' in token:
# Pseudo-class # ID selector
tag_name, pseudo = token.split(':', 1) tag_name, tag_id = token.split('#', 1)
if tag_name == '': def id_matches(tag):
raise ValueError( return tag.get('id', None) == tag_id
"A pseudo-class must be prefixed with a tag name.") checker = id_matches
pseudo_attributes = re.match('([a-zA-Z\d-]+)\(([a-zA-Z\d]+)\)', pseudo)
found = [] elif '.' in token:
if pseudo_attributes is not None: # Class selector
pseudo_type, pseudo_value = pseudo_attributes.groups() tag_name, klass = token.split('.', 1)
classes = set(klass.split('.'))
def classes_match(candidate):
return classes.issubset(candidate.get('class', []))
checker = classes_match
elif ':' in token:
# Pseudo-class
tag_name, pseudo = token.split(':', 1)
if tag_name == '':
raise ValueError(
"A pseudo-class must be prefixed with a tag name.")
pseudo_attributes = re.match('([a-zA-Z\d-]+)\(([a-zA-Z\d]+)\)', pseudo)
found = []
if pseudo_attributes is None:
pseudo_type = pseudo
pseudo_value = None
else:
pseudo_type, pseudo_value = pseudo_attributes.groups()
if pseudo_type == 'nth-of-type': if pseudo_type == 'nth-of-type':
try: try:
pseudo_value = int(pseudo_value) pseudo_value = int(pseudo_value)
@@ -1286,109 +1387,110 @@ class Tag(PageElement):
raise NotImplementedError( raise NotImplementedError(
'Only the following pseudo-classes are implemented: nth-of-type.') 'Only the following pseudo-classes are implemented: nth-of-type.')
elif token == '*': elif token == '*':
# Star selector -- matches everything # Star selector -- matches everything
pass pass
elif token == '>': elif token == '>':
# Run the next token as a CSS selector against the # Run the next token as a CSS selector against the
# direct children of each tag in the current context. # direct children of each tag in the current context.
recursive_candidate_generator = lambda tag: tag.children recursive_candidate_generator = lambda tag: tag.children
elif token == '~': elif token == '~':
# Run the next token as a CSS selector against the # Run the next token as a CSS selector against the
# siblings of each tag in the current context. # siblings of each tag in the current context.
recursive_candidate_generator = lambda tag: tag.next_siblings recursive_candidate_generator = lambda tag: tag.next_siblings
elif token == '+': elif token == '+':
# For each tag in the current context, run the next # For each tag in the current context, run the next
# token as a CSS selector against the tag's next # token as a CSS selector against the tag's next
# sibling that's a tag. # sibling that's a tag.
def next_tag_sibling(tag): def next_tag_sibling(tag):
yield tag.find_next_sibling(True) yield tag.find_next_sibling(True)
recursive_candidate_generator = next_tag_sibling recursive_candidate_generator = next_tag_sibling
elif self.tag_name_re.match(token): elif self.tag_name_re.match(token):
# Just a tag name. # Just a tag name.
tag_name = token tag_name = token
else:
raise ValueError(
'Unsupported or invalid CSS selector: "%s"' % token)
if recursive_candidate_generator:
# This happens when the selector looks like "> foo".
#
# The generator calls select() recursively on every
# member of the current context, passing in a different
# candidate generator and a different selector.
#
# In the case of "> foo", the candidate generator is
# one that yields a tag's direct children (">"), and
# the selector is "foo".
next_token = tokens[index+1]
def recursive_select(tag):
if self._select_debug:
print ' Calling select("%s") recursively on %s %s' % (next_token, tag.name, tag.attrs)
print '-' * 40
for i in tag.select(next_token, recursive_candidate_generator):
if self._select_debug:
print '(Recursive select picked up candidate %s %s)' % (i.name, i.attrs)
yield i
if self._select_debug:
print '-' * 40
_use_candidate_generator = recursive_select
elif _candidate_generator is None:
# By default, a tag's candidates are all of its
# children. If tag_name is defined, only yield tags
# with that name.
if self._select_debug:
if tag_name:
check = "[any]"
else:
check = tag_name
print ' Default candidate generator, tag name="%s"' % check
if self._select_debug:
# This is redundant with later code, but it stops
# a bunch of bogus tags from cluttering up the
# debug log.
def default_candidate_generator(tag):
for child in tag.descendants:
if not isinstance(child, Tag):
continue
if tag_name and not child.name == tag_name:
continue
yield child
_use_candidate_generator = default_candidate_generator
else: else:
_use_candidate_generator = lambda tag: tag.descendants raise ValueError(
else: 'Unsupported or invalid CSS selector: "%s"' % token)
_use_candidate_generator = _candidate_generator if recursive_candidate_generator:
# This happens when the selector looks like "> foo".
new_context = [] #
new_context_ids = set([]) # The generator calls select() recursively on every
for tag in current_context: # member of the current context, passing in a different
if self._select_debug: # candidate generator and a different selector.
print " Running candidate generator on %s %s" % ( #
tag.name, repr(tag.attrs)) # In the case of "> foo", the candidate generator is
for candidate in _use_candidate_generator(tag): # one that yields a tag's direct children (">"), and
if not isinstance(candidate, Tag): # the selector is "foo".
continue next_token = tokens[index+1]
if tag_name and candidate.name != tag_name: def recursive_select(tag):
continue
if checker is not None:
try:
result = checker(candidate)
except StopIteration:
# The checker has decided we should no longer
# run the generator.
break
if checker is None or result:
if self._select_debug: if self._select_debug:
print " SUCCESS %s %s" % (candidate.name, repr(candidate.attrs)) print ' Calling select("%s") recursively on %s %s' % (next_token, tag.name, tag.attrs)
if id(candidate) not in new_context_ids: print '-' * 40
# If a tag matches a selector more than once, for i in tag.select(next_token, recursive_candidate_generator):
# don't include it in the context more than once. if self._select_debug:
new_context.append(candidate) print '(Recursive select picked up candidate %s %s)' % (i.name, i.attrs)
new_context_ids.add(id(candidate)) yield i
elif self._select_debug: if self._select_debug:
print " FAILURE %s %s" % (candidate.name, repr(candidate.attrs)) print '-' * 40
_use_candidate_generator = recursive_select
elif _candidate_generator is None:
# By default, a tag's candidates are all of its
# children. If tag_name is defined, only yield tags
# with that name.
if self._select_debug:
if tag_name:
check = "[any]"
else:
check = tag_name
print ' Default candidate generator, tag name="%s"' % check
if self._select_debug:
# This is redundant with later code, but it stops
# a bunch of bogus tags from cluttering up the
# debug log.
def default_candidate_generator(tag):
for child in tag.descendants:
if not isinstance(child, Tag):
continue
if tag_name and not child.name == tag_name:
continue
yield child
_use_candidate_generator = default_candidate_generator
else:
_use_candidate_generator = lambda tag: tag.descendants
else:
_use_candidate_generator = _candidate_generator
count = 0
for tag in current_context:
if self._select_debug:
print " Running candidate generator on %s %s" % (
tag.name, repr(tag.attrs))
for candidate in _use_candidate_generator(tag):
if not isinstance(candidate, Tag):
continue
if tag_name and candidate.name != tag_name:
continue
if checker is not None:
try:
result = checker(candidate)
except StopIteration:
# The checker has decided we should no longer
# run the generator.
break
if checker is None or result:
if self._select_debug:
print " SUCCESS %s %s" % (candidate.name, repr(candidate.attrs))
if id(candidate) not in new_context_ids:
# If a tag matches a selector more than once,
# don't include it in the context more than once.
new_context.append(candidate)
new_context_ids.add(id(candidate))
if limit and len(new_context) >= limit:
break
elif self._select_debug:
print " FAILURE %s %s" % (candidate.name, repr(candidate.attrs))
current_context = new_context current_context = new_context
+89 -1
View File
@@ -1,5 +1,6 @@
"""Helper classes for tests.""" """Helper classes for tests."""
import pickle
import copy import copy
import functools import functools
import unittest import unittest
@@ -43,6 +44,16 @@ class SoupTest(unittest.TestCase):
self.assertEqual(obj.decode(), self.document_for(compare_parsed_to)) self.assertEqual(obj.decode(), self.document_for(compare_parsed_to))
def assertConnectedness(self, element):
"""Ensure that next_element and previous_element are properly
set for all descendants of the given element.
"""
earlier = None
for e in element.descendants:
if earlier:
self.assertEqual(e, earlier.next_element)
self.assertEqual(earlier, e.previous_element)
earlier = e
class HTMLTreeBuilderSmokeTest(object): class HTMLTreeBuilderSmokeTest(object):
@@ -54,6 +65,15 @@ class HTMLTreeBuilderSmokeTest(object):
markup in these tests, there's not much room for interpretation. markup in these tests, there's not much room for interpretation.
""" """
def test_pickle_and_unpickle_identity(self):
# Pickling a tree, then unpickling it, yields a tree identical
# to the original.
tree = self.soup("<a><b>foo</a>")
dumped = pickle.dumps(tree, 2)
loaded = pickle.loads(dumped)
self.assertEqual(loaded.__class__, BeautifulSoup)
self.assertEqual(loaded.decode(), tree.decode())
def assertDoctypeHandled(self, doctype_fragment): def assertDoctypeHandled(self, doctype_fragment):
"""Assert that a given doctype string is handled correctly.""" """Assert that a given doctype string is handled correctly."""
doctype_str, soup = self._document_with_doctype(doctype_fragment) doctype_str, soup = self._document_with_doctype(doctype_fragment)
@@ -114,6 +134,11 @@ class HTMLTreeBuilderSmokeTest(object):
soup.encode("utf-8").replace(b"\n", b""), soup.encode("utf-8").replace(b"\n", b""),
markup.replace(b"\n", b"")) markup.replace(b"\n", b""))
def test_processing_instruction(self):
markup = b"""<?PITarget PIContent?>"""
soup = self.soup(markup)
self.assertEqual(markup, soup.encode("utf8"))
def test_deepcopy(self): def test_deepcopy(self):
"""Make sure you can copy the tree builder. """Make sure you can copy the tree builder.
@@ -155,6 +180,23 @@ class HTMLTreeBuilderSmokeTest(object):
def test_nested_formatting_elements(self): def test_nested_formatting_elements(self):
self.assertSoupEquals("<em><em></em></em>") self.assertSoupEquals("<em><em></em></em>")
def test_double_head(self):
html = '''<!DOCTYPE html>
<html>
<head>
<title>Ordinary HEAD element test</title>
</head>
<script type="text/javascript">
alert("Help!");
</script>
<body>
Hello, world!
</body>
</html>
'''
soup = self.soup(html)
self.assertEqual("text/javascript", soup.find('script')['type'])
def test_comment(self): def test_comment(self):
# Comments are represented as Comment objects. # Comments are represented as Comment objects.
markup = "<p>foo<!--foobar-->baz</p>" markup = "<p>foo<!--foobar-->baz</p>"
@@ -221,6 +263,14 @@ class HTMLTreeBuilderSmokeTest(object):
soup = self.soup(markup) soup = self.soup(markup)
self.assertEqual(["css"], soup.div.div['class']) self.assertEqual(["css"], soup.div.div['class'])
def test_multivalued_attribute_on_html(self):
# html5lib uses a different API to set the attributes ot the
# <html> tag. This has caused problems with multivalued
# attributes.
markup = '<html class="a b"></html>'
soup = self.soup(markup)
self.assertEqual(["a", "b"], soup.html['class'])
def test_angle_brackets_in_attribute_values_are_escaped(self): def test_angle_brackets_in_attribute_values_are_escaped(self):
self.assertSoupEquals('<a b="<a>"></a>', '<a b="&lt;a&gt;"></a>') self.assertSoupEquals('<a b="<a>"></a>', '<a b="&lt;a&gt;"></a>')
@@ -253,6 +303,35 @@ class HTMLTreeBuilderSmokeTest(object):
soup = self.soup("<html><h2>\nfoo</h2><p></p></html>") soup = self.soup("<html><h2>\nfoo</h2><p></p></html>")
self.assertEqual("p", soup.h2.string.next_element.name) self.assertEqual("p", soup.h2.string.next_element.name)
self.assertEqual("p", soup.p.name) self.assertEqual("p", soup.p.name)
self.assertConnectedness(soup)
def test_head_tag_between_head_and_body(self):
"Prevent recurrence of a bug in the html5lib treebuilder."
content = """<html><head></head>
<link></link>
<body>foo</body>
</html>
"""
soup = self.soup(content)
self.assertNotEqual(None, soup.html.body)
self.assertConnectedness(soup)
def test_multiple_copies_of_a_tag(self):
"Prevent recurrence of a bug in the html5lib treebuilder."
content = """<!DOCTYPE html>
<html>
<body>
<article id="a" >
<div><a href="1"></div>
<footer>
<a href="2"></a>
</footer>
</article>
</body>
</html>
"""
soup = self.soup(content)
self.assertConnectedness(soup.article)
def test_basic_namespaces(self): def test_basic_namespaces(self):
"""Parsers don't need to *understand* namespaces, but at the """Parsers don't need to *understand* namespaces, but at the
@@ -463,6 +542,15 @@ class HTMLTreeBuilderSmokeTest(object):
class XMLTreeBuilderSmokeTest(object): class XMLTreeBuilderSmokeTest(object):
def test_pickle_and_unpickle_identity(self):
# Pickling a tree, then unpickling it, yields a tree identical
# to the original.
tree = self.soup("<a><b>foo</a>")
dumped = pickle.dumps(tree, 2)
loaded = pickle.loads(dumped)
self.assertEqual(loaded.__class__, BeautifulSoup)
self.assertEqual(loaded.decode(), tree.decode())
def test_docstring_generated(self): def test_docstring_generated(self):
soup = self.soup("<root/>") soup = self.soup("<root/>")
self.assertEqual( self.assertEqual(
@@ -485,7 +573,7 @@ class XMLTreeBuilderSmokeTest(object):
<script type="text/javascript"> <script type="text/javascript">
</script> </script>
""" """
soup = BeautifulSoup(doc, "xml") soup = BeautifulSoup(doc, "lxml-xml")
# lxml would have stripped this while parsing, but we can add # lxml would have stripped this while parsing, but we can add
# it later. # it later.
soup.script.string = 'console.log("< < hey > > ");' soup.script.string = 'console.log("< < hey > > ");'
+3 -1
View File
@@ -20,4 +20,6 @@ from .serializer import serialize
__all__ = ["HTMLParser", "parse", "parseFragment", "getTreeBuilder", __all__ = ["HTMLParser", "parse", "parseFragment", "getTreeBuilder",
"getTreeWalker", "serialize"] "getTreeWalker", "serialize"]
__version__ = "0.999"
# this has to be at the top level, see how setup.py parses this
__version__ = "0.999999"
+193 -195
View File
@@ -1,292 +1,290 @@
from __future__ import absolute_import, division, unicode_literals from __future__ import absolute_import, division, unicode_literals
import string import string
import gettext
_ = gettext.gettext
EOF = None EOF = None
E = { E = {
"null-character": "null-character":
_("Null character in input stream, replaced with U+FFFD."), "Null character in input stream, replaced with U+FFFD.",
"invalid-codepoint": "invalid-codepoint":
_("Invalid codepoint in stream."), "Invalid codepoint in stream.",
"incorrectly-placed-solidus": "incorrectly-placed-solidus":
_("Solidus (/) incorrectly placed in tag."), "Solidus (/) incorrectly placed in tag.",
"incorrect-cr-newline-entity": "incorrect-cr-newline-entity":
_("Incorrect CR newline entity, replaced with LF."), "Incorrect CR newline entity, replaced with LF.",
"illegal-windows-1252-entity": "illegal-windows-1252-entity":
_("Entity used with illegal number (windows-1252 reference)."), "Entity used with illegal number (windows-1252 reference).",
"cant-convert-numeric-entity": "cant-convert-numeric-entity":
_("Numeric entity couldn't be converted to character " "Numeric entity couldn't be converted to character "
"(codepoint U+%(charAsInt)08x)."), "(codepoint U+%(charAsInt)08x).",
"illegal-codepoint-for-numeric-entity": "illegal-codepoint-for-numeric-entity":
_("Numeric entity represents an illegal codepoint: " "Numeric entity represents an illegal codepoint: "
"U+%(charAsInt)08x."), "U+%(charAsInt)08x.",
"numeric-entity-without-semicolon": "numeric-entity-without-semicolon":
_("Numeric entity didn't end with ';'."), "Numeric entity didn't end with ';'.",
"expected-numeric-entity-but-got-eof": "expected-numeric-entity-but-got-eof":
_("Numeric entity expected. Got end of file instead."), "Numeric entity expected. Got end of file instead.",
"expected-numeric-entity": "expected-numeric-entity":
_("Numeric entity expected but none found."), "Numeric entity expected but none found.",
"named-entity-without-semicolon": "named-entity-without-semicolon":
_("Named entity didn't end with ';'."), "Named entity didn't end with ';'.",
"expected-named-entity": "expected-named-entity":
_("Named entity expected. Got none."), "Named entity expected. Got none.",
"attributes-in-end-tag": "attributes-in-end-tag":
_("End tag contains unexpected attributes."), "End tag contains unexpected attributes.",
'self-closing-flag-on-end-tag': 'self-closing-flag-on-end-tag':
_("End tag contains unexpected self-closing flag."), "End tag contains unexpected self-closing flag.",
"expected-tag-name-but-got-right-bracket": "expected-tag-name-but-got-right-bracket":
_("Expected tag name. Got '>' instead."), "Expected tag name. Got '>' instead.",
"expected-tag-name-but-got-question-mark": "expected-tag-name-but-got-question-mark":
_("Expected tag name. Got '?' instead. (HTML doesn't " "Expected tag name. Got '?' instead. (HTML doesn't "
"support processing instructions.)"), "support processing instructions.)",
"expected-tag-name": "expected-tag-name":
_("Expected tag name. Got something else instead"), "Expected tag name. Got something else instead",
"expected-closing-tag-but-got-right-bracket": "expected-closing-tag-but-got-right-bracket":
_("Expected closing tag. Got '>' instead. Ignoring '</>'."), "Expected closing tag. Got '>' instead. Ignoring '</>'.",
"expected-closing-tag-but-got-eof": "expected-closing-tag-but-got-eof":
_("Expected closing tag. Unexpected end of file."), "Expected closing tag. Unexpected end of file.",
"expected-closing-tag-but-got-char": "expected-closing-tag-but-got-char":
_("Expected closing tag. Unexpected character '%(data)s' found."), "Expected closing tag. Unexpected character '%(data)s' found.",
"eof-in-tag-name": "eof-in-tag-name":
_("Unexpected end of file in the tag name."), "Unexpected end of file in the tag name.",
"expected-attribute-name-but-got-eof": "expected-attribute-name-but-got-eof":
_("Unexpected end of file. Expected attribute name instead."), "Unexpected end of file. Expected attribute name instead.",
"eof-in-attribute-name": "eof-in-attribute-name":
_("Unexpected end of file in attribute name."), "Unexpected end of file in attribute name.",
"invalid-character-in-attribute-name": "invalid-character-in-attribute-name":
_("Invalid character in attribute name"), "Invalid character in attribute name",
"duplicate-attribute": "duplicate-attribute":
_("Dropped duplicate attribute on tag."), "Dropped duplicate attribute on tag.",
"expected-end-of-tag-name-but-got-eof": "expected-end-of-tag-name-but-got-eof":
_("Unexpected end of file. Expected = or end of tag."), "Unexpected end of file. Expected = or end of tag.",
"expected-attribute-value-but-got-eof": "expected-attribute-value-but-got-eof":
_("Unexpected end of file. Expected attribute value."), "Unexpected end of file. Expected attribute value.",
"expected-attribute-value-but-got-right-bracket": "expected-attribute-value-but-got-right-bracket":
_("Expected attribute value. Got '>' instead."), "Expected attribute value. Got '>' instead.",
'equals-in-unquoted-attribute-value': 'equals-in-unquoted-attribute-value':
_("Unexpected = in unquoted attribute"), "Unexpected = in unquoted attribute",
'unexpected-character-in-unquoted-attribute-value': 'unexpected-character-in-unquoted-attribute-value':
_("Unexpected character in unquoted attribute"), "Unexpected character in unquoted attribute",
"invalid-character-after-attribute-name": "invalid-character-after-attribute-name":
_("Unexpected character after attribute name."), "Unexpected character after attribute name.",
"unexpected-character-after-attribute-value": "unexpected-character-after-attribute-value":
_("Unexpected character after attribute value."), "Unexpected character after attribute value.",
"eof-in-attribute-value-double-quote": "eof-in-attribute-value-double-quote":
_("Unexpected end of file in attribute value (\")."), "Unexpected end of file in attribute value (\").",
"eof-in-attribute-value-single-quote": "eof-in-attribute-value-single-quote":
_("Unexpected end of file in attribute value (')."), "Unexpected end of file in attribute value (').",
"eof-in-attribute-value-no-quotes": "eof-in-attribute-value-no-quotes":
_("Unexpected end of file in attribute value."), "Unexpected end of file in attribute value.",
"unexpected-EOF-after-solidus-in-tag": "unexpected-EOF-after-solidus-in-tag":
_("Unexpected end of file in tag. Expected >"), "Unexpected end of file in tag. Expected >",
"unexpected-character-after-solidus-in-tag": "unexpected-character-after-solidus-in-tag":
_("Unexpected character after / in tag. Expected >"), "Unexpected character after / in tag. Expected >",
"expected-dashes-or-doctype": "expected-dashes-or-doctype":
_("Expected '--' or 'DOCTYPE'. Not found."), "Expected '--' or 'DOCTYPE'. Not found.",
"unexpected-bang-after-double-dash-in-comment": "unexpected-bang-after-double-dash-in-comment":
_("Unexpected ! after -- in comment"), "Unexpected ! after -- in comment",
"unexpected-space-after-double-dash-in-comment": "unexpected-space-after-double-dash-in-comment":
_("Unexpected space after -- in comment"), "Unexpected space after -- in comment",
"incorrect-comment": "incorrect-comment":
_("Incorrect comment."), "Incorrect comment.",
"eof-in-comment": "eof-in-comment":
_("Unexpected end of file in comment."), "Unexpected end of file in comment.",
"eof-in-comment-end-dash": "eof-in-comment-end-dash":
_("Unexpected end of file in comment (-)"), "Unexpected end of file in comment (-)",
"unexpected-dash-after-double-dash-in-comment": "unexpected-dash-after-double-dash-in-comment":
_("Unexpected '-' after '--' found in comment."), "Unexpected '-' after '--' found in comment.",
"eof-in-comment-double-dash": "eof-in-comment-double-dash":
_("Unexpected end of file in comment (--)."), "Unexpected end of file in comment (--).",
"eof-in-comment-end-space-state": "eof-in-comment-end-space-state":
_("Unexpected end of file in comment."), "Unexpected end of file in comment.",
"eof-in-comment-end-bang-state": "eof-in-comment-end-bang-state":
_("Unexpected end of file in comment."), "Unexpected end of file in comment.",
"unexpected-char-in-comment": "unexpected-char-in-comment":
_("Unexpected character in comment found."), "Unexpected character in comment found.",
"need-space-after-doctype": "need-space-after-doctype":
_("No space after literal string 'DOCTYPE'."), "No space after literal string 'DOCTYPE'.",
"expected-doctype-name-but-got-right-bracket": "expected-doctype-name-but-got-right-bracket":
_("Unexpected > character. Expected DOCTYPE name."), "Unexpected > character. Expected DOCTYPE name.",
"expected-doctype-name-but-got-eof": "expected-doctype-name-but-got-eof":
_("Unexpected end of file. Expected DOCTYPE name."), "Unexpected end of file. Expected DOCTYPE name.",
"eof-in-doctype-name": "eof-in-doctype-name":
_("Unexpected end of file in DOCTYPE name."), "Unexpected end of file in DOCTYPE name.",
"eof-in-doctype": "eof-in-doctype":
_("Unexpected end of file in DOCTYPE."), "Unexpected end of file in DOCTYPE.",
"expected-space-or-right-bracket-in-doctype": "expected-space-or-right-bracket-in-doctype":
_("Expected space or '>'. Got '%(data)s'"), "Expected space or '>'. Got '%(data)s'",
"unexpected-end-of-doctype": "unexpected-end-of-doctype":
_("Unexpected end of DOCTYPE."), "Unexpected end of DOCTYPE.",
"unexpected-char-in-doctype": "unexpected-char-in-doctype":
_("Unexpected character in DOCTYPE."), "Unexpected character in DOCTYPE.",
"eof-in-innerhtml": "eof-in-innerhtml":
_("XXX innerHTML EOF"), "XXX innerHTML EOF",
"unexpected-doctype": "unexpected-doctype":
_("Unexpected DOCTYPE. Ignored."), "Unexpected DOCTYPE. Ignored.",
"non-html-root": "non-html-root":
_("html needs to be the first start tag."), "html needs to be the first start tag.",
"expected-doctype-but-got-eof": "expected-doctype-but-got-eof":
_("Unexpected End of file. Expected DOCTYPE."), "Unexpected End of file. Expected DOCTYPE.",
"unknown-doctype": "unknown-doctype":
_("Erroneous DOCTYPE."), "Erroneous DOCTYPE.",
"expected-doctype-but-got-chars": "expected-doctype-but-got-chars":
_("Unexpected non-space characters. Expected DOCTYPE."), "Unexpected non-space characters. Expected DOCTYPE.",
"expected-doctype-but-got-start-tag": "expected-doctype-but-got-start-tag":
_("Unexpected start tag (%(name)s). Expected DOCTYPE."), "Unexpected start tag (%(name)s). Expected DOCTYPE.",
"expected-doctype-but-got-end-tag": "expected-doctype-but-got-end-tag":
_("Unexpected end tag (%(name)s). Expected DOCTYPE."), "Unexpected end tag (%(name)s). Expected DOCTYPE.",
"end-tag-after-implied-root": "end-tag-after-implied-root":
_("Unexpected end tag (%(name)s) after the (implied) root element."), "Unexpected end tag (%(name)s) after the (implied) root element.",
"expected-named-closing-tag-but-got-eof": "expected-named-closing-tag-but-got-eof":
_("Unexpected end of file. Expected end tag (%(name)s)."), "Unexpected end of file. Expected end tag (%(name)s).",
"two-heads-are-not-better-than-one": "two-heads-are-not-better-than-one":
_("Unexpected start tag head in existing head. Ignored."), "Unexpected start tag head in existing head. Ignored.",
"unexpected-end-tag": "unexpected-end-tag":
_("Unexpected end tag (%(name)s). Ignored."), "Unexpected end tag (%(name)s). Ignored.",
"unexpected-start-tag-out-of-my-head": "unexpected-start-tag-out-of-my-head":
_("Unexpected start tag (%(name)s) that can be in head. Moved."), "Unexpected start tag (%(name)s) that can be in head. Moved.",
"unexpected-start-tag": "unexpected-start-tag":
_("Unexpected start tag (%(name)s)."), "Unexpected start tag (%(name)s).",
"missing-end-tag": "missing-end-tag":
_("Missing end tag (%(name)s)."), "Missing end tag (%(name)s).",
"missing-end-tags": "missing-end-tags":
_("Missing end tags (%(name)s)."), "Missing end tags (%(name)s).",
"unexpected-start-tag-implies-end-tag": "unexpected-start-tag-implies-end-tag":
_("Unexpected start tag (%(startName)s) " "Unexpected start tag (%(startName)s) "
"implies end tag (%(endName)s)."), "implies end tag (%(endName)s).",
"unexpected-start-tag-treated-as": "unexpected-start-tag-treated-as":
_("Unexpected start tag (%(originalName)s). Treated as %(newName)s."), "Unexpected start tag (%(originalName)s). Treated as %(newName)s.",
"deprecated-tag": "deprecated-tag":
_("Unexpected start tag %(name)s. Don't use it!"), "Unexpected start tag %(name)s. Don't use it!",
"unexpected-start-tag-ignored": "unexpected-start-tag-ignored":
_("Unexpected start tag %(name)s. Ignored."), "Unexpected start tag %(name)s. Ignored.",
"expected-one-end-tag-but-got-another": "expected-one-end-tag-but-got-another":
_("Unexpected end tag (%(gotName)s). " "Unexpected end tag (%(gotName)s). "
"Missing end tag (%(expectedName)s)."), "Missing end tag (%(expectedName)s).",
"end-tag-too-early": "end-tag-too-early":
_("End tag (%(name)s) seen too early. Expected other end tag."), "End tag (%(name)s) seen too early. Expected other end tag.",
"end-tag-too-early-named": "end-tag-too-early-named":
_("Unexpected end tag (%(gotName)s). Expected end tag (%(expectedName)s)."), "Unexpected end tag (%(gotName)s). Expected end tag (%(expectedName)s).",
"end-tag-too-early-ignored": "end-tag-too-early-ignored":
_("End tag (%(name)s) seen too early. Ignored."), "End tag (%(name)s) seen too early. Ignored.",
"adoption-agency-1.1": "adoption-agency-1.1":
_("End tag (%(name)s) violates step 1, " "End tag (%(name)s) violates step 1, "
"paragraph 1 of the adoption agency algorithm."), "paragraph 1 of the adoption agency algorithm.",
"adoption-agency-1.2": "adoption-agency-1.2":
_("End tag (%(name)s) violates step 1, " "End tag (%(name)s) violates step 1, "
"paragraph 2 of the adoption agency algorithm."), "paragraph 2 of the adoption agency algorithm.",
"adoption-agency-1.3": "adoption-agency-1.3":
_("End tag (%(name)s) violates step 1, " "End tag (%(name)s) violates step 1, "
"paragraph 3 of the adoption agency algorithm."), "paragraph 3 of the adoption agency algorithm.",
"adoption-agency-4.4": "adoption-agency-4.4":
_("End tag (%(name)s) violates step 4, " "End tag (%(name)s) violates step 4, "
"paragraph 4 of the adoption agency algorithm."), "paragraph 4 of the adoption agency algorithm.",
"unexpected-end-tag-treated-as": "unexpected-end-tag-treated-as":
_("Unexpected end tag (%(originalName)s). Treated as %(newName)s."), "Unexpected end tag (%(originalName)s). Treated as %(newName)s.",
"no-end-tag": "no-end-tag":
_("This element (%(name)s) has no end tag."), "This element (%(name)s) has no end tag.",
"unexpected-implied-end-tag-in-table": "unexpected-implied-end-tag-in-table":
_("Unexpected implied end tag (%(name)s) in the table phase."), "Unexpected implied end tag (%(name)s) in the table phase.",
"unexpected-implied-end-tag-in-table-body": "unexpected-implied-end-tag-in-table-body":
_("Unexpected implied end tag (%(name)s) in the table body phase."), "Unexpected implied end tag (%(name)s) in the table body phase.",
"unexpected-char-implies-table-voodoo": "unexpected-char-implies-table-voodoo":
_("Unexpected non-space characters in " "Unexpected non-space characters in "
"table context caused voodoo mode."), "table context caused voodoo mode.",
"unexpected-hidden-input-in-table": "unexpected-hidden-input-in-table":
_("Unexpected input with type hidden in table context."), "Unexpected input with type hidden in table context.",
"unexpected-form-in-table": "unexpected-form-in-table":
_("Unexpected form in table context."), "Unexpected form in table context.",
"unexpected-start-tag-implies-table-voodoo": "unexpected-start-tag-implies-table-voodoo":
_("Unexpected start tag (%(name)s) in " "Unexpected start tag (%(name)s) in "
"table context caused voodoo mode."), "table context caused voodoo mode.",
"unexpected-end-tag-implies-table-voodoo": "unexpected-end-tag-implies-table-voodoo":
_("Unexpected end tag (%(name)s) in " "Unexpected end tag (%(name)s) in "
"table context caused voodoo mode."), "table context caused voodoo mode.",
"unexpected-cell-in-table-body": "unexpected-cell-in-table-body":
_("Unexpected table cell start tag (%(name)s) " "Unexpected table cell start tag (%(name)s) "
"in the table body phase."), "in the table body phase.",
"unexpected-cell-end-tag": "unexpected-cell-end-tag":
_("Got table cell end tag (%(name)s) " "Got table cell end tag (%(name)s) "
"while required end tags are missing."), "while required end tags are missing.",
"unexpected-end-tag-in-table-body": "unexpected-end-tag-in-table-body":
_("Unexpected end tag (%(name)s) in the table body phase. Ignored."), "Unexpected end tag (%(name)s) in the table body phase. Ignored.",
"unexpected-implied-end-tag-in-table-row": "unexpected-implied-end-tag-in-table-row":
_("Unexpected implied end tag (%(name)s) in the table row phase."), "Unexpected implied end tag (%(name)s) in the table row phase.",
"unexpected-end-tag-in-table-row": "unexpected-end-tag-in-table-row":
_("Unexpected end tag (%(name)s) in the table row phase. Ignored."), "Unexpected end tag (%(name)s) in the table row phase. Ignored.",
"unexpected-select-in-select": "unexpected-select-in-select":
_("Unexpected select start tag in the select phase " "Unexpected select start tag in the select phase "
"treated as select end tag."), "treated as select end tag.",
"unexpected-input-in-select": "unexpected-input-in-select":
_("Unexpected input start tag in the select phase."), "Unexpected input start tag in the select phase.",
"unexpected-start-tag-in-select": "unexpected-start-tag-in-select":
_("Unexpected start tag token (%(name)s in the select phase. " "Unexpected start tag token (%(name)s in the select phase. "
"Ignored."), "Ignored.",
"unexpected-end-tag-in-select": "unexpected-end-tag-in-select":
_("Unexpected end tag (%(name)s) in the select phase. Ignored."), "Unexpected end tag (%(name)s) in the select phase. Ignored.",
"unexpected-table-element-start-tag-in-select-in-table": "unexpected-table-element-start-tag-in-select-in-table":
_("Unexpected table element start tag (%(name)s) in the select in table phase."), "Unexpected table element start tag (%(name)s) in the select in table phase.",
"unexpected-table-element-end-tag-in-select-in-table": "unexpected-table-element-end-tag-in-select-in-table":
_("Unexpected table element end tag (%(name)s) in the select in table phase."), "Unexpected table element end tag (%(name)s) in the select in table phase.",
"unexpected-char-after-body": "unexpected-char-after-body":
_("Unexpected non-space characters in the after body phase."), "Unexpected non-space characters in the after body phase.",
"unexpected-start-tag-after-body": "unexpected-start-tag-after-body":
_("Unexpected start tag token (%(name)s)" "Unexpected start tag token (%(name)s)"
" in the after body phase."), " in the after body phase.",
"unexpected-end-tag-after-body": "unexpected-end-tag-after-body":
_("Unexpected end tag token (%(name)s)" "Unexpected end tag token (%(name)s)"
" in the after body phase."), " in the after body phase.",
"unexpected-char-in-frameset": "unexpected-char-in-frameset":
_("Unexpected characters in the frameset phase. Characters ignored."), "Unexpected characters in the frameset phase. Characters ignored.",
"unexpected-start-tag-in-frameset": "unexpected-start-tag-in-frameset":
_("Unexpected start tag token (%(name)s)" "Unexpected start tag token (%(name)s)"
" in the frameset phase. Ignored."), " in the frameset phase. Ignored.",
"unexpected-frameset-in-frameset-innerhtml": "unexpected-frameset-in-frameset-innerhtml":
_("Unexpected end tag token (frameset) " "Unexpected end tag token (frameset) "
"in the frameset phase (innerHTML)."), "in the frameset phase (innerHTML).",
"unexpected-end-tag-in-frameset": "unexpected-end-tag-in-frameset":
_("Unexpected end tag token (%(name)s)" "Unexpected end tag token (%(name)s)"
" in the frameset phase. Ignored."), " in the frameset phase. Ignored.",
"unexpected-char-after-frameset": "unexpected-char-after-frameset":
_("Unexpected non-space characters in the " "Unexpected non-space characters in the "
"after frameset phase. Ignored."), "after frameset phase. Ignored.",
"unexpected-start-tag-after-frameset": "unexpected-start-tag-after-frameset":
_("Unexpected start tag (%(name)s)" "Unexpected start tag (%(name)s)"
" in the after frameset phase. Ignored."), " in the after frameset phase. Ignored.",
"unexpected-end-tag-after-frameset": "unexpected-end-tag-after-frameset":
_("Unexpected end tag (%(name)s)" "Unexpected end tag (%(name)s)"
" in the after frameset phase. Ignored."), " in the after frameset phase. Ignored.",
"unexpected-end-tag-after-body-innerhtml": "unexpected-end-tag-after-body-innerhtml":
_("Unexpected end tag after body(innerHtml)"), "Unexpected end tag after body(innerHtml)",
"expected-eof-but-got-char": "expected-eof-but-got-char":
_("Unexpected non-space characters. Expected end of file."), "Unexpected non-space characters. Expected end of file.",
"expected-eof-but-got-start-tag": "expected-eof-but-got-start-tag":
_("Unexpected start tag (%(name)s)" "Unexpected start tag (%(name)s)"
". Expected end of file."), ". Expected end of file.",
"expected-eof-but-got-end-tag": "expected-eof-but-got-end-tag":
_("Unexpected end tag (%(name)s)" "Unexpected end tag (%(name)s)"
". Expected end of file."), ". Expected end of file.",
"eof-in-table": "eof-in-table":
_("Unexpected end of file. Expected table content."), "Unexpected end of file. Expected table content.",
"eof-in-select": "eof-in-select":
_("Unexpected end of file. Expected select content."), "Unexpected end of file. Expected select content.",
"eof-in-frameset": "eof-in-frameset":
_("Unexpected end of file. Expected frameset content."), "Unexpected end of file. Expected frameset content.",
"eof-in-script-in-script": "eof-in-script-in-script":
_("Unexpected end of file. Expected script content."), "Unexpected end of file. Expected script content.",
"eof-in-foreign-lands": "eof-in-foreign-lands":
_("Unexpected end of file. Expected foreign content"), "Unexpected end of file. Expected foreign content",
"non-void-element-with-trailing-solidus": "non-void-element-with-trailing-solidus":
_("Trailing solidus not allowed on element %(name)s"), "Trailing solidus not allowed on element %(name)s",
"unexpected-html-element-in-foreign-content": "unexpected-html-element-in-foreign-content":
_("Element %(name)s not allowed in a non-html context"), "Element %(name)s not allowed in a non-html context",
"unexpected-end-tag-before-html": "unexpected-end-tag-before-html":
_("Unexpected end tag (%(name)s) before html."), "Unexpected end tag (%(name)s) before html.",
"XXX-undefined-error": "XXX-undefined-error":
_("Undefined error (this sucks and should be fixed)"), "Undefined error (this sucks and should be fixed)",
} }
namespaces = { namespaces = {
@@ -298,7 +296,7 @@ namespaces = {
"xmlns": "http://www.w3.org/2000/xmlns/" "xmlns": "http://www.w3.org/2000/xmlns/"
} }
scopingElements = frozenset(( scopingElements = frozenset([
(namespaces["html"], "applet"), (namespaces["html"], "applet"),
(namespaces["html"], "caption"), (namespaces["html"], "caption"),
(namespaces["html"], "html"), (namespaces["html"], "html"),
@@ -316,9 +314,9 @@ scopingElements = frozenset((
(namespaces["svg"], "foreignObject"), (namespaces["svg"], "foreignObject"),
(namespaces["svg"], "desc"), (namespaces["svg"], "desc"),
(namespaces["svg"], "title"), (namespaces["svg"], "title"),
)) ])
formattingElements = frozenset(( formattingElements = frozenset([
(namespaces["html"], "a"), (namespaces["html"], "a"),
(namespaces["html"], "b"), (namespaces["html"], "b"),
(namespaces["html"], "big"), (namespaces["html"], "big"),
@@ -333,9 +331,9 @@ formattingElements = frozenset((
(namespaces["html"], "strong"), (namespaces["html"], "strong"),
(namespaces["html"], "tt"), (namespaces["html"], "tt"),
(namespaces["html"], "u") (namespaces["html"], "u")
)) ])
specialElements = frozenset(( specialElements = frozenset([
(namespaces["html"], "address"), (namespaces["html"], "address"),
(namespaces["html"], "applet"), (namespaces["html"], "applet"),
(namespaces["html"], "area"), (namespaces["html"], "area"),
@@ -416,22 +414,22 @@ specialElements = frozenset((
(namespaces["html"], "wbr"), (namespaces["html"], "wbr"),
(namespaces["html"], "xmp"), (namespaces["html"], "xmp"),
(namespaces["svg"], "foreignObject") (namespaces["svg"], "foreignObject")
)) ])
htmlIntegrationPointElements = frozenset(( htmlIntegrationPointElements = frozenset([
(namespaces["mathml"], "annotaion-xml"), (namespaces["mathml"], "annotaion-xml"),
(namespaces["svg"], "foreignObject"), (namespaces["svg"], "foreignObject"),
(namespaces["svg"], "desc"), (namespaces["svg"], "desc"),
(namespaces["svg"], "title") (namespaces["svg"], "title")
)) ])
mathmlTextIntegrationPointElements = frozenset(( mathmlTextIntegrationPointElements = frozenset([
(namespaces["mathml"], "mi"), (namespaces["mathml"], "mi"),
(namespaces["mathml"], "mo"), (namespaces["mathml"], "mo"),
(namespaces["mathml"], "mn"), (namespaces["mathml"], "mn"),
(namespaces["mathml"], "ms"), (namespaces["mathml"], "ms"),
(namespaces["mathml"], "mtext") (namespaces["mathml"], "mtext")
)) ])
adjustForeignAttributes = { adjustForeignAttributes = {
"xlink:actuate": ("xlink", "actuate", namespaces["xlink"]), "xlink:actuate": ("xlink", "actuate", namespaces["xlink"]),
@@ -451,21 +449,21 @@ adjustForeignAttributes = {
unadjustForeignAttributes = dict([((ns, local), qname) for qname, (prefix, local, ns) in unadjustForeignAttributes = dict([((ns, local), qname) for qname, (prefix, local, ns) in
adjustForeignAttributes.items()]) adjustForeignAttributes.items()])
spaceCharacters = frozenset(( spaceCharacters = frozenset([
"\t", "\t",
"\n", "\n",
"\u000C", "\u000C",
" ", " ",
"\r" "\r"
)) ])
tableInsertModeElements = frozenset(( tableInsertModeElements = frozenset([
"table", "table",
"tbody", "tbody",
"tfoot", "tfoot",
"thead", "thead",
"tr" "tr"
)) ])
asciiLowercase = frozenset(string.ascii_lowercase) asciiLowercase = frozenset(string.ascii_lowercase)
asciiUppercase = frozenset(string.ascii_uppercase) asciiUppercase = frozenset(string.ascii_uppercase)
@@ -486,7 +484,7 @@ headingElements = (
"h6" "h6"
) )
voidElements = frozenset(( voidElements = frozenset([
"base", "base",
"command", "command",
"event-source", "event-source",
@@ -502,11 +500,11 @@ voidElements = frozenset((
"input", "input",
"source", "source",
"track" "track"
)) ])
cdataElements = frozenset(('title', 'textarea')) cdataElements = frozenset(['title', 'textarea'])
rcdataElements = frozenset(( rcdataElements = frozenset([
'style', 'style',
'script', 'script',
'xmp', 'xmp',
@@ -514,27 +512,27 @@ rcdataElements = frozenset((
'noembed', 'noembed',
'noframes', 'noframes',
'noscript' 'noscript'
)) ])
booleanAttributes = { booleanAttributes = {
"": frozenset(("irrelevant",)), "": frozenset(["irrelevant"]),
"style": frozenset(("scoped",)), "style": frozenset(["scoped"]),
"img": frozenset(("ismap",)), "img": frozenset(["ismap"]),
"audio": frozenset(("autoplay", "controls")), "audio": frozenset(["autoplay", "controls"]),
"video": frozenset(("autoplay", "controls")), "video": frozenset(["autoplay", "controls"]),
"script": frozenset(("defer", "async")), "script": frozenset(["defer", "async"]),
"details": frozenset(("open",)), "details": frozenset(["open"]),
"datagrid": frozenset(("multiple", "disabled")), "datagrid": frozenset(["multiple", "disabled"]),
"command": frozenset(("hidden", "disabled", "checked", "default")), "command": frozenset(["hidden", "disabled", "checked", "default"]),
"hr": frozenset(("noshade")), "hr": frozenset(["noshade"]),
"menu": frozenset(("autosubmit",)), "menu": frozenset(["autosubmit"]),
"fieldset": frozenset(("disabled", "readonly")), "fieldset": frozenset(["disabled", "readonly"]),
"option": frozenset(("disabled", "readonly", "selected")), "option": frozenset(["disabled", "readonly", "selected"]),
"optgroup": frozenset(("disabled", "readonly")), "optgroup": frozenset(["disabled", "readonly"]),
"button": frozenset(("disabled", "autofocus")), "button": frozenset(["disabled", "autofocus"]),
"input": frozenset(("disabled", "readonly", "required", "autofocus", "checked", "ismap")), "input": frozenset(["disabled", "readonly", "required", "autofocus", "checked", "ismap"]),
"select": frozenset(("disabled", "readonly", "autofocus", "multiple")), "select": frozenset(["disabled", "readonly", "autofocus", "multiple"]),
"output": frozenset(("disabled", "readonly")), "output": frozenset(["disabled", "readonly"]),
} }
# entitiesWindows1252 has to be _ordered_ and needs to have an index. It # entitiesWindows1252 has to be _ordered_ and needs to have an index. It
@@ -574,7 +572,7 @@ entitiesWindows1252 = (
376 # 0x9F 0x0178 LATIN CAPITAL LETTER Y WITH DIAERESIS 376 # 0x9F 0x0178 LATIN CAPITAL LETTER Y WITH DIAERESIS
) )
xmlEntities = frozenset(('lt;', 'gt;', 'amp;', 'apos;', 'quot;')) xmlEntities = frozenset(['lt;', 'gt;', 'amp;', 'apos;', 'quot;'])
entities = { entities = {
"AElig": "\xc6", "AElig": "\xc6",
@@ -3088,8 +3086,8 @@ tokenTypes = {
"ParseError": 7 "ParseError": 7
} }
tagTokenTypes = frozenset((tokenTypes["StartTag"], tokenTypes["EndTag"], tagTokenTypes = frozenset([tokenTypes["StartTag"], tokenTypes["EndTag"],
tokenTypes["EmptyTag"])) tokenTypes["EmptyTag"]])
prefixes = dict([(v, k) for k, v in namespaces.items()]) prefixes = dict([(v, k) for k, v in namespaces.items()])
+19 -22
View File
@@ -1,8 +1,5 @@
from __future__ import absolute_import, division, unicode_literals from __future__ import absolute_import, division, unicode_literals
from gettext import gettext
_ = gettext
from . import _base from . import _base
from ..constants import cdataElements, rcdataElements, voidElements from ..constants import cdataElements, rcdataElements, voidElements
@@ -23,24 +20,24 @@ class Filter(_base.Filter):
if type in ("StartTag", "EmptyTag"): if type in ("StartTag", "EmptyTag"):
name = token["name"] name = token["name"]
if contentModelFlag != "PCDATA": if contentModelFlag != "PCDATA":
raise LintError(_("StartTag not in PCDATA content model flag: %(tag)s") % {"tag": name}) raise LintError("StartTag not in PCDATA content model flag: %(tag)s" % {"tag": name})
if not isinstance(name, str): if not isinstance(name, str):
raise LintError(_("Tag name is not a string: %(tag)r") % {"tag": name}) raise LintError("Tag name is not a string: %(tag)r" % {"tag": name})
if not name: if not name:
raise LintError(_("Empty tag name")) raise LintError("Empty tag name")
if type == "StartTag" and name in voidElements: if type == "StartTag" and name in voidElements:
raise LintError(_("Void element reported as StartTag token: %(tag)s") % {"tag": name}) raise LintError("Void element reported as StartTag token: %(tag)s" % {"tag": name})
elif type == "EmptyTag" and name not in voidElements: elif type == "EmptyTag" and name not in voidElements:
raise LintError(_("Non-void element reported as EmptyTag token: %(tag)s") % {"tag": token["name"]}) raise LintError("Non-void element reported as EmptyTag token: %(tag)s" % {"tag": token["name"]})
if type == "StartTag": if type == "StartTag":
open_elements.append(name) open_elements.append(name)
for name, value in token["data"]: for name, value in token["data"]:
if not isinstance(name, str): if not isinstance(name, str):
raise LintError(_("Attribute name is not a string: %(name)r") % {"name": name}) raise LintError("Attribute name is not a string: %(name)r" % {"name": name})
if not name: if not name:
raise LintError(_("Empty attribute name")) raise LintError("Empty attribute name")
if not isinstance(value, str): if not isinstance(value, str):
raise LintError(_("Attribute value is not a string: %(value)r") % {"value": value}) raise LintError("Attribute value is not a string: %(value)r" % {"value": value})
if name in cdataElements: if name in cdataElements:
contentModelFlag = "CDATA" contentModelFlag = "CDATA"
elif name in rcdataElements: elif name in rcdataElements:
@@ -51,43 +48,43 @@ class Filter(_base.Filter):
elif type == "EndTag": elif type == "EndTag":
name = token["name"] name = token["name"]
if not isinstance(name, str): if not isinstance(name, str):
raise LintError(_("Tag name is not a string: %(tag)r") % {"tag": name}) raise LintError("Tag name is not a string: %(tag)r" % {"tag": name})
if not name: if not name:
raise LintError(_("Empty tag name")) raise LintError("Empty tag name")
if name in voidElements: if name in voidElements:
raise LintError(_("Void element reported as EndTag token: %(tag)s") % {"tag": name}) raise LintError("Void element reported as EndTag token: %(tag)s" % {"tag": name})
start_name = open_elements.pop() start_name = open_elements.pop()
if start_name != name: if start_name != name:
raise LintError(_("EndTag (%(end)s) does not match StartTag (%(start)s)") % {"end": name, "start": start_name}) raise LintError("EndTag (%(end)s) does not match StartTag (%(start)s)" % {"end": name, "start": start_name})
contentModelFlag = "PCDATA" contentModelFlag = "PCDATA"
elif type == "Comment": elif type == "Comment":
if contentModelFlag != "PCDATA": if contentModelFlag != "PCDATA":
raise LintError(_("Comment not in PCDATA content model flag")) raise LintError("Comment not in PCDATA content model flag")
elif type in ("Characters", "SpaceCharacters"): elif type in ("Characters", "SpaceCharacters"):
data = token["data"] data = token["data"]
if not isinstance(data, str): if not isinstance(data, str):
raise LintError(_("Attribute name is not a string: %(name)r") % {"name": data}) raise LintError("Attribute name is not a string: %(name)r" % {"name": data})
if not data: if not data:
raise LintError(_("%(type)s token with empty data") % {"type": type}) raise LintError("%(type)s token with empty data" % {"type": type})
if type == "SpaceCharacters": if type == "SpaceCharacters":
data = data.strip(spaceCharacters) data = data.strip(spaceCharacters)
if data: if data:
raise LintError(_("Non-space character(s) found in SpaceCharacters token: %(token)r") % {"token": data}) raise LintError("Non-space character(s) found in SpaceCharacters token: %(token)r" % {"token": data})
elif type == "Doctype": elif type == "Doctype":
name = token["name"] name = token["name"]
if contentModelFlag != "PCDATA": if contentModelFlag != "PCDATA":
raise LintError(_("Doctype not in PCDATA content model flag: %(name)s") % {"name": name}) raise LintError("Doctype not in PCDATA content model flag: %(name)s" % {"name": name})
if not isinstance(name, str): if not isinstance(name, str):
raise LintError(_("Tag name is not a string: %(tag)r") % {"tag": name}) raise LintError("Tag name is not a string: %(tag)r" % {"tag": name})
# XXX: what to do with token["data"] ? # XXX: what to do with token["data"] ?
elif type in ("ParseError", "SerializeError"): elif type in ("ParseError", "SerializeError"):
pass pass
else: else:
raise LintError(_("Unknown token type: %(type)s") % {"type": type}) raise LintError("Unknown token type: %(type)s" % {"type": type})
yield token yield token
+15 -4
View File
@@ -18,6 +18,7 @@ from .constants import cdataElements, rcdataElements
from .constants import tokenTypes, ReparseException, namespaces from .constants import tokenTypes, ReparseException, namespaces
from .constants import htmlIntegrationPointElements, mathmlTextIntegrationPointElements from .constants import htmlIntegrationPointElements, mathmlTextIntegrationPointElements
from .constants import adjustForeignAttributes as adjustForeignAttributesMap from .constants import adjustForeignAttributes as adjustForeignAttributesMap
from .constants import E
def parse(doc, treebuilder="etree", encoding=None, def parse(doc, treebuilder="etree", encoding=None,
@@ -129,6 +130,17 @@ class HTMLParser(object):
self.framesetOK = True self.framesetOK = True
@property
def documentEncoding(self):
"""The name of the character encoding
that was used to decode the input stream,
or :obj:`None` if that is not determined yet.
"""
if not hasattr(self, 'tokenizer'):
return None
return self.tokenizer.stream.charEncoding[0]
def isHTMLIntegrationPoint(self, element): def isHTMLIntegrationPoint(self, element):
if (element.name == "annotation-xml" and if (element.name == "annotation-xml" and
element.namespace == namespaces["mathml"]): element.namespace == namespaces["mathml"]):
@@ -245,7 +257,7 @@ class HTMLParser(object):
# XXX The idea is to make errorcode mandatory. # XXX The idea is to make errorcode mandatory.
self.errors.append((self.tokenizer.stream.position(), errorcode, datavars)) self.errors.append((self.tokenizer.stream.position(), errorcode, datavars))
if self.strict: if self.strict:
raise ParseError raise ParseError(E[errorcode] % datavars)
def normalizeToken(self, token): def normalizeToken(self, token):
""" HTML5 specific normalizations to the token stream """ """ HTML5 specific normalizations to the token stream """
@@ -868,7 +880,7 @@ def getPhases(debug):
self.startTagHandler = utils.MethodDispatcher([ self.startTagHandler = utils.MethodDispatcher([
("html", self.startTagHtml), ("html", self.startTagHtml),
(("base", "basefont", "bgsound", "command", "link", "meta", (("base", "basefont", "bgsound", "command", "link", "meta",
"noframes", "script", "style", "title"), "script", "style", "title"),
self.startTagProcessInHead), self.startTagProcessInHead),
("body", self.startTagBody), ("body", self.startTagBody),
("frameset", self.startTagFrameset), ("frameset", self.startTagFrameset),
@@ -1205,8 +1217,7 @@ def getPhases(debug):
attributes["name"] = "isindex" attributes["name"] = "isindex"
self.processStartTag(impliedTagToken("input", "StartTag", self.processStartTag(impliedTagToken("input", "StartTag",
attributes=attributes, attributes=attributes,
selfClosing= selfClosing=token["selfClosing"]))
token["selfClosing"]))
self.processEndTag(impliedTagToken("label")) self.processEndTag(impliedTagToken("label"))
self.processStartTag(impliedTagToken("hr", "StartTag")) self.processStartTag(impliedTagToken("hr", "StartTag"))
self.processEndTag(impliedTagToken("form")) self.processEndTag(impliedTagToken("form"))
+26 -9
View File
@@ -28,7 +28,18 @@ asciiLettersBytes = frozenset([item.encode("ascii") for item in asciiLetters])
asciiUppercaseBytes = frozenset([item.encode("ascii") for item in asciiUppercase]) asciiUppercaseBytes = frozenset([item.encode("ascii") for item in asciiUppercase])
spacesAngleBrackets = spaceCharactersBytes | frozenset([b">", b"<"]) spacesAngleBrackets = spaceCharactersBytes | frozenset([b">", b"<"])
invalid_unicode_re = re.compile("[\u0001-\u0008\u000B\u000E-\u001F\u007F-\u009F\uD800-\uDFFF\uFDD0-\uFDEF\uFFFE\uFFFF\U0001FFFE\U0001FFFF\U0002FFFE\U0002FFFF\U0003FFFE\U0003FFFF\U0004FFFE\U0004FFFF\U0005FFFE\U0005FFFF\U0006FFFE\U0006FFFF\U0007FFFE\U0007FFFF\U0008FFFE\U0008FFFF\U0009FFFE\U0009FFFF\U000AFFFE\U000AFFFF\U000BFFFE\U000BFFFF\U000CFFFE\U000CFFFF\U000DFFFE\U000DFFFF\U000EFFFE\U000EFFFF\U000FFFFE\U000FFFFF\U0010FFFE\U0010FFFF]")
invalid_unicode_no_surrogate = "[\u0001-\u0008\u000B\u000E-\u001F\u007F-\u009F\uFDD0-\uFDEF\uFFFE\uFFFF\U0001FFFE\U0001FFFF\U0002FFFE\U0002FFFF\U0003FFFE\U0003FFFF\U0004FFFE\U0004FFFF\U0005FFFE\U0005FFFF\U0006FFFE\U0006FFFF\U0007FFFE\U0007FFFF\U0008FFFE\U0008FFFF\U0009FFFE\U0009FFFF\U000AFFFE\U000AFFFF\U000BFFFE\U000BFFFF\U000CFFFE\U000CFFFF\U000DFFFE\U000DFFFF\U000EFFFE\U000EFFFF\U000FFFFE\U000FFFFF\U0010FFFE\U0010FFFF]"
if utils.supports_lone_surrogates:
# Use one extra step of indirection and create surrogates with
# unichr. Not using this indirection would introduce an illegal
# unicode literal on platforms not supporting such lone
# surrogates.
invalid_unicode_re = re.compile(invalid_unicode_no_surrogate +
eval('"\\uD800-\\uDFFF"'))
else:
invalid_unicode_re = re.compile(invalid_unicode_no_surrogate)
non_bmp_invalid_codepoints = set([0x1FFFE, 0x1FFFF, 0x2FFFE, 0x2FFFF, 0x3FFFE, non_bmp_invalid_codepoints = set([0x1FFFE, 0x1FFFF, 0x2FFFE, 0x2FFFF, 0x3FFFE,
0x3FFFF, 0x4FFFE, 0x4FFFF, 0x5FFFE, 0x5FFFF, 0x3FFFF, 0x4FFFE, 0x4FFFF, 0x5FFFE, 0x5FFFF,
@@ -164,13 +175,18 @@ class HTMLUnicodeInputStream(object):
""" """
# Craziness if not utils.supports_lone_surrogates:
if len("\U0010FFFF") == 1: # Such platforms will have already checked for such
# surrogate errors, so no need to do this checking.
self.reportCharacterErrors = None
self.replaceCharactersRegexp = None
elif len("\U0010FFFF") == 1:
self.reportCharacterErrors = self.characterErrorsUCS4 self.reportCharacterErrors = self.characterErrorsUCS4
self.replaceCharactersRegexp = re.compile("[\uD800-\uDFFF]") self.replaceCharactersRegexp = re.compile(eval('"[\\uD800-\\uDFFF]"'))
else: else:
self.reportCharacterErrors = self.characterErrorsUCS2 self.reportCharacterErrors = self.characterErrorsUCS2
self.replaceCharactersRegexp = re.compile("([\uD800-\uDBFF](?![\uDC00-\uDFFF])|(?<![\uD800-\uDBFF])[\uDC00-\uDFFF])") self.replaceCharactersRegexp = re.compile(
eval('"([\\uD800-\\uDBFF](?![\\uDC00-\\uDFFF])|(?<![\\uD800-\\uDBFF])[\\uDC00-\\uDFFF])"'))
# List of where new lines occur # List of where new lines occur
self.newLines = [0] self.newLines = [0]
@@ -265,11 +281,12 @@ class HTMLUnicodeInputStream(object):
self._bufferedCharacter = data[-1] self._bufferedCharacter = data[-1]
data = data[:-1] data = data[:-1]
self.reportCharacterErrors(data) if self.reportCharacterErrors:
self.reportCharacterErrors(data)
# Replace invalid characters # Replace invalid characters
# Note U+0000 is dealt with in the tokenizer # Note U+0000 is dealt with in the tokenizer
data = self.replaceCharactersRegexp.sub("\ufffd", data) data = self.replaceCharactersRegexp.sub("\ufffd", data)
data = data.replace("\r\n", "\n") data = data.replace("\r\n", "\n")
data = data.replace("\r", "\n") data = data.replace("\r", "\n")
+33 -8
View File
@@ -2,11 +2,26 @@ from __future__ import absolute_import, division, unicode_literals
import re import re
from xml.sax.saxutils import escape, unescape from xml.sax.saxutils import escape, unescape
from six.moves import urllib_parse as urlparse
from .tokenizer import HTMLTokenizer from .tokenizer import HTMLTokenizer
from .constants import tokenTypes from .constants import tokenTypes
content_type_rgx = re.compile(r'''
^
# Match a content type <application>/<type>
(?P<content_type>[-a-zA-Z0-9.]+/[-a-zA-Z0-9.]+)
# Match any character set and encoding
(?:(?:;charset=(?:[-a-zA-Z0-9]+)(?:;(?:base64))?)
|(?:;(?:base64))?(?:;charset=(?:[-a-zA-Z0-9]+))?)
# Assume the rest is data
,.*
$
''',
re.VERBOSE)
class HTMLSanitizerMixin(object): class HTMLSanitizerMixin(object):
""" sanitization of XHTML+MathML+SVG and of inline style attributes.""" """ sanitization of XHTML+MathML+SVG and of inline style attributes."""
@@ -100,8 +115,8 @@ class HTMLSanitizerMixin(object):
'xml:base', 'xml:lang', 'xml:space', 'xmlns', 'xmlns:xlink', 'y', 'xml:base', 'xml:lang', 'xml:space', 'xmlns', 'xmlns:xlink', 'y',
'y1', 'y2', 'zoomAndPan'] 'y1', 'y2', 'zoomAndPan']
attr_val_is_uri = ['href', 'src', 'cite', 'action', 'longdesc', 'poster', attr_val_is_uri = ['href', 'src', 'cite', 'action', 'longdesc', 'poster', 'background', 'datasrc',
'xlink:href', 'xml:base'] 'dynsrc', 'lowsrc', 'ping', 'poster', 'xlink:href', 'xml:base']
svg_attr_val_allows_ref = ['clip-path', 'color-profile', 'cursor', 'fill', svg_attr_val_allows_ref = ['clip-path', 'color-profile', 'cursor', 'fill',
'filter', 'marker', 'marker-start', 'marker-mid', 'marker-end', 'filter', 'marker', 'marker-start', 'marker-mid', 'marker-end',
@@ -138,7 +153,9 @@ class HTMLSanitizerMixin(object):
acceptable_protocols = ['ed2k', 'ftp', 'http', 'https', 'irc', acceptable_protocols = ['ed2k', 'ftp', 'http', 'https', 'irc',
'mailto', 'news', 'gopher', 'nntp', 'telnet', 'webcal', 'mailto', 'news', 'gopher', 'nntp', 'telnet', 'webcal',
'xmpp', 'callto', 'feed', 'urn', 'aim', 'rsync', 'tag', 'xmpp', 'callto', 'feed', 'urn', 'aim', 'rsync', 'tag',
'ssh', 'sftp', 'rtsp', 'afs'] 'ssh', 'sftp', 'rtsp', 'afs', 'data']
acceptable_content_types = ['image/png', 'image/jpeg', 'image/gif', 'image/webp', 'image/bmp', 'text/plain']
# subclasses may define their own versions of these constants # subclasses may define their own versions of these constants
allowed_elements = acceptable_elements + mathml_elements + svg_elements allowed_elements = acceptable_elements + mathml_elements + svg_elements
@@ -147,6 +164,7 @@ class HTMLSanitizerMixin(object):
allowed_css_keywords = acceptable_css_keywords allowed_css_keywords = acceptable_css_keywords
allowed_svg_properties = acceptable_svg_properties allowed_svg_properties = acceptable_svg_properties
allowed_protocols = acceptable_protocols allowed_protocols = acceptable_protocols
allowed_content_types = acceptable_content_types
# Sanitize the +html+, escaping all elements not in ALLOWED_ELEMENTS, and # Sanitize the +html+, escaping all elements not in ALLOWED_ELEMENTS, and
# stripping out all # attributes not in ALLOWED_ATTRIBUTES. Style # stripping out all # attributes not in ALLOWED_ATTRIBUTES. Style
@@ -189,10 +207,17 @@ class HTMLSanitizerMixin(object):
unescape(attrs[attr])).lower() unescape(attrs[attr])).lower()
# remove replacement characters from unescaped characters # remove replacement characters from unescaped characters
val_unescaped = val_unescaped.replace("\ufffd", "") val_unescaped = val_unescaped.replace("\ufffd", "")
if (re.match("^[a-z0-9][-+.a-z0-9]*:", val_unescaped) and uri = urlparse.urlparse(val_unescaped)
(val_unescaped.split(':')[0] not in if uri and uri.scheme:
self.allowed_protocols)): if uri.scheme not in self.allowed_protocols:
del attrs[attr] del attrs[attr]
if uri.scheme == 'data':
m = content_type_rgx.match(uri.path)
if not m:
del attrs[attr]
elif m.group('content_type') not in self.allowed_content_types:
del attrs[attr]
for attr in self.svg_attr_val_allows_ref: for attr in self.svg_attr_val_allows_ref:
if attr in attrs: if attr in attrs:
attrs[attr] = re.sub(r'url\s*\(\s*[^#\s][^)]+?\)', attrs[attr] = re.sub(r'url\s*\(\s*[^#\s][^)]+?\)',
@@ -245,7 +270,7 @@ class HTMLSanitizerMixin(object):
elif prop.split('-')[0].lower() in ['background', 'border', 'margin', elif prop.split('-')[0].lower() in ['background', 'border', 'margin',
'padding']: 'padding']:
for keyword in value.split(): for keyword in value.split():
if not keyword in self.acceptable_css_keywords and \ if keyword not in self.acceptable_css_keywords and \
not re.match("^(#[0-9a-f]+|rgb\(\d+%?,\d*%?,?\d*%?\)?|\d{0,2}\.?\d{0,2}(cm|em|ex|in|mm|pc|pt|px|%|,|\))?)$", keyword): not re.match("^(#[0-9a-f]+|rgb\(\d+%?,\d*%?,?\d*%?\)?|\d{0,2}\.?\d{0,2}(cm|em|ex|in|mm|pc|pt|px|%|,|\))?)$", keyword):
break break
else: else:
+8 -11
View File
@@ -1,9 +1,6 @@
from __future__ import absolute_import, division, unicode_literals from __future__ import absolute_import, division, unicode_literals
from six import text_type from six import text_type
import gettext
_ = gettext.gettext
try: try:
from functools import reduce from functools import reduce
except ImportError: except ImportError:
@@ -35,7 +32,7 @@ else:
v = utils.surrogatePairToCodepoint(v) v = utils.surrogatePairToCodepoint(v)
else: else:
v = ord(v) v = ord(v)
if not v in encode_entity_map or k.islower(): if v not in encode_entity_map or k.islower():
# prefer &lt; over &LT; and similarly for &amp;, &gt;, etc. # prefer &lt; over &LT; and similarly for &amp;, &gt;, etc.
encode_entity_map[v] = k encode_entity_map[v] = k
@@ -208,7 +205,7 @@ class HTMLSerializer(object):
if token["systemId"]: if token["systemId"]:
if token["systemId"].find('"') >= 0: if token["systemId"].find('"') >= 0:
if token["systemId"].find("'") >= 0: if token["systemId"].find("'") >= 0:
self.serializeError(_("System identifer contains both single and double quote characters")) self.serializeError("System identifer contains both single and double quote characters")
quote_char = "'" quote_char = "'"
else: else:
quote_char = '"' quote_char = '"'
@@ -220,7 +217,7 @@ class HTMLSerializer(object):
elif type in ("Characters", "SpaceCharacters"): elif type in ("Characters", "SpaceCharacters"):
if type == "SpaceCharacters" or in_cdata: if type == "SpaceCharacters" or in_cdata:
if in_cdata and token["data"].find("</") >= 0: if in_cdata and token["data"].find("</") >= 0:
self.serializeError(_("Unexpected </ in CDATA")) self.serializeError("Unexpected </ in CDATA")
yield self.encode(token["data"]) yield self.encode(token["data"])
else: else:
yield self.encode(escape(token["data"])) yield self.encode(escape(token["data"]))
@@ -231,7 +228,7 @@ class HTMLSerializer(object):
if name in rcdataElements and not self.escape_rcdata: if name in rcdataElements and not self.escape_rcdata:
in_cdata = True in_cdata = True
elif in_cdata: elif in_cdata:
self.serializeError(_("Unexpected child element of a CDATA element")) self.serializeError("Unexpected child element of a CDATA element")
for (attr_namespace, attr_name), attr_value in token["data"].items(): for (attr_namespace, attr_name), attr_value in token["data"].items():
# TODO: Add namespace support here # TODO: Add namespace support here
k = attr_name k = attr_name
@@ -279,20 +276,20 @@ class HTMLSerializer(object):
if name in rcdataElements: if name in rcdataElements:
in_cdata = False in_cdata = False
elif in_cdata: elif in_cdata:
self.serializeError(_("Unexpected child element of a CDATA element")) self.serializeError("Unexpected child element of a CDATA element")
yield self.encodeStrict("</%s>" % name) yield self.encodeStrict("</%s>" % name)
elif type == "Comment": elif type == "Comment":
data = token["data"] data = token["data"]
if data.find("--") >= 0: if data.find("--") >= 0:
self.serializeError(_("Comment contains --")) self.serializeError("Comment contains --")
yield self.encodeStrict("<!--%s-->" % token["data"]) yield self.encodeStrict("<!--%s-->" % token["data"])
elif type == "Entity": elif type == "Entity":
name = token["name"] name = token["name"]
key = name + ";" key = name + ";"
if not key in entities: if key not in entities:
self.serializeError(_("Entity %s not recognized" % name)) self.serializeError("Entity %s not recognized" % name)
if self.resolve_entities and key not in xmlEntities: if self.resolve_entities and key not in xmlEntities:
data = entities[key] data = entities[key]
else: else:
+1 -1
View File
@@ -158,7 +158,7 @@ def getDomBuilder(DomImplementation):
else: else:
# HACK: allow text nodes as children of the document node # HACK: allow text nodes as children of the document node
if hasattr(self.dom, '_child_node_types'): if hasattr(self.dom, '_child_node_types'):
if not Node.TEXT_NODE in self.dom._child_node_types: if Node.TEXT_NODE not in self.dom._child_node_types:
self.dom._child_node_types = list(self.dom._child_node_types) self.dom._child_node_types = list(self.dom._child_node_types)
self.dom._child_node_types.append(Node.TEXT_NODE) self.dom._child_node_types.append(Node.TEXT_NODE)
self.dom.appendChild(self.dom.createTextNode(data)) self.dom.appendChild(self.dom.createTextNode(data))
+90
View File
@@ -10,8 +10,12 @@ returning an iterator generating tokens.
from __future__ import absolute_import, division, unicode_literals from __future__ import absolute_import, division, unicode_literals
__all__ = ["getTreeWalker", "pprint", "dom", "etree", "genshistream", "lxmletree",
"pulldom"]
import sys import sys
from .. import constants
from ..utils import default_etree from ..utils import default_etree
treeWalkerCache = {} treeWalkerCache = {}
@@ -55,3 +59,89 @@ def getTreeWalker(treeType, implementation=None, **kwargs):
# XXX: NEVER cache here, caching is done in the etree submodule # XXX: NEVER cache here, caching is done in the etree submodule
return etree.getETreeModule(implementation, **kwargs).TreeWalker return etree.getETreeModule(implementation, **kwargs).TreeWalker
return treeWalkerCache.get(treeType) return treeWalkerCache.get(treeType)
def concatenateCharacterTokens(tokens):
pendingCharacters = []
for token in tokens:
type = token["type"]
if type in ("Characters", "SpaceCharacters"):
pendingCharacters.append(token["data"])
else:
if pendingCharacters:
yield {"type": "Characters", "data": "".join(pendingCharacters)}
pendingCharacters = []
yield token
if pendingCharacters:
yield {"type": "Characters", "data": "".join(pendingCharacters)}
def pprint(walker):
"""Pretty printer for tree walkers"""
output = []
indent = 0
for token in concatenateCharacterTokens(walker):
type = token["type"]
if type in ("StartTag", "EmptyTag"):
# tag name
if token["namespace"] and token["namespace"] != constants.namespaces["html"]:
if token["namespace"] in constants.prefixes:
ns = constants.prefixes[token["namespace"]]
else:
ns = token["namespace"]
name = "%s %s" % (ns, token["name"])
else:
name = token["name"]
output.append("%s<%s>" % (" " * indent, name))
indent += 2
# attributes (sorted for consistent ordering)
attrs = token["data"]
for (namespace, localname), value in sorted(attrs.items()):
if namespace:
if namespace in constants.prefixes:
ns = constants.prefixes[namespace]
else:
ns = namespace
name = "%s %s" % (ns, localname)
else:
name = localname
output.append("%s%s=\"%s\"" % (" " * indent, name, value))
# self-closing
if type == "EmptyTag":
indent -= 2
elif type == "EndTag":
indent -= 2
elif type == "Comment":
output.append("%s<!-- %s -->" % (" " * indent, token["data"]))
elif type == "Doctype":
if token["name"]:
if token["publicId"]:
output.append("""%s<!DOCTYPE %s "%s" "%s">""" %
(" " * indent,
token["name"],
token["publicId"],
token["systemId"] if token["systemId"] else ""))
elif token["systemId"]:
output.append("""%s<!DOCTYPE %s "" "%s">""" %
(" " * indent,
token["name"],
token["systemId"]))
else:
output.append("%s<!DOCTYPE %s>" % (" " * indent,
token["name"]))
else:
output.append("%s<!DOCTYPE >" % (" " * indent,))
elif type == "Characters":
output.append("%s\"%s\"" % (" " * indent, token["data"]))
elif type == "SpaceCharacters":
assert False, "concatenateCharacterTokens should have got rid of all Space tokens"
else:
raise ValueError("Unknown token type, %s" % type)
return "\n".join(output)
+4 -4
View File
@@ -1,8 +1,8 @@
from __future__ import absolute_import, division, unicode_literals from __future__ import absolute_import, division, unicode_literals
from six import text_type, string_types from six import text_type, string_types
import gettext __all__ = ["DOCUMENT", "DOCTYPE", "TEXT", "ELEMENT", "COMMENT", "ENTITY", "UNKNOWN",
_ = gettext.gettext "TreeWalker", "NonRecursiveTreeWalker"]
from xml.dom import Node from xml.dom import Node
@@ -58,7 +58,7 @@ class TreeWalker(object):
"namespace": to_text(namespace), "namespace": to_text(namespace),
"data": attrs} "data": attrs}
if hasChildren: if hasChildren:
yield self.error(_("Void element has children")) yield self.error("Void element has children")
def startTag(self, namespace, name, attrs): def startTag(self, namespace, name, attrs):
assert namespace is None or isinstance(namespace, string_types), type(namespace) assert namespace is None or isinstance(namespace, string_types), type(namespace)
@@ -122,7 +122,7 @@ class TreeWalker(object):
return {"type": "Entity", "name": text_type(name)} return {"type": "Entity", "name": text_type(name)}
def unknown(self, nodeType): def unknown(self, nodeType):
return self.error(_("Unknown node type: ") + nodeType) return self.error("Unknown node type: " + nodeType)
class NonRecursiveTreeWalker(TreeWalker): class NonRecursiveTreeWalker(TreeWalker):
-3
View File
@@ -2,9 +2,6 @@ from __future__ import absolute_import, division, unicode_literals
from xml.dom import Node from xml.dom import Node
import gettext
_ = gettext.gettext
from . import _base from . import _base
+2 -4
View File
@@ -7,12 +7,10 @@ except ImportError:
from ordereddict import OrderedDict from ordereddict import OrderedDict
except ImportError: except ImportError:
OrderedDict = dict OrderedDict = dict
import gettext
_ = gettext.gettext
import re import re
from six import text_type from six import string_types
from . import _base from . import _base
from ..utils import moduleFactoryFactory from ..utils import moduleFactoryFactory
@@ -60,7 +58,7 @@ def getETreeBuilder(ElementTreeImplementation):
return _base.COMMENT, node.text return _base.COMMENT, node.text
else: else:
assert type(node.tag) == text_type, type(node.tag) assert isinstance(node.tag, string_types), type(node.tag)
# This is assumed to be an ordinary element # This is assumed to be an ordinary element
match = tag_regexp.match(node.tag) match = tag_regexp.match(node.tag)
if match: if match:
+4 -7
View File
@@ -4,9 +4,6 @@ from six import text_type
from lxml import etree from lxml import etree
from ..treebuilders.etree import tag_regexp from ..treebuilders.etree import tag_regexp
from gettext import gettext
_ = gettext
from . import _base from . import _base
from .. import ihatexml from .. import ihatexml
@@ -130,7 +127,7 @@ class TreeWalker(_base.NonRecursiveTreeWalker):
def getNodeDetails(self, node): def getNodeDetails(self, node):
if isinstance(node, tuple): # Text node if isinstance(node, tuple): # Text node
node, key = node node, key = node
assert key in ("text", "tail"), _("Text nodes are text or tail, found %s") % key assert key in ("text", "tail"), "Text nodes are text or tail, found %s" % key
return _base.TEXT, ensure_str(getattr(node, key)) return _base.TEXT, ensure_str(getattr(node, key))
elif isinstance(node, Root): elif isinstance(node, Root):
@@ -169,7 +166,7 @@ class TreeWalker(_base.NonRecursiveTreeWalker):
attrs, len(node) > 0 or node.text) attrs, len(node) > 0 or node.text)
def getFirstChild(self, node): def getFirstChild(self, node):
assert not isinstance(node, tuple), _("Text nodes have no children") assert not isinstance(node, tuple), "Text nodes have no children"
assert len(node) or node.text, "Node has no children" assert len(node) or node.text, "Node has no children"
if node.text: if node.text:
@@ -180,7 +177,7 @@ class TreeWalker(_base.NonRecursiveTreeWalker):
def getNextSibling(self, node): def getNextSibling(self, node):
if isinstance(node, tuple): # Text node if isinstance(node, tuple): # Text node
node, key = node node, key = node
assert key in ("text", "tail"), _("Text nodes are text or tail, found %s") % key assert key in ("text", "tail"), "Text nodes are text or tail, found %s" % key
if key == "text": if key == "text":
# XXX: we cannot use a "bool(node) and node[0] or None" construct here # XXX: we cannot use a "bool(node) and node[0] or None" construct here
# because node[0] might evaluate to False if it has no child element # because node[0] might evaluate to False if it has no child element
@@ -196,7 +193,7 @@ class TreeWalker(_base.NonRecursiveTreeWalker):
def getParentNode(self, node): def getParentNode(self, node):
if isinstance(node, tuple): # Text node if isinstance(node, tuple): # Text node
node, key = node node, key = node
assert key in ("text", "tail"), _("Text nodes are text or tail, found %s") % key assert key in ("text", "tail"), "Text nodes are text or tail, found %s" % key
if key == "text": if key == "text":
return node return node
# else: fallback to "normal" processing # else: fallback to "normal" processing
+22 -1
View File
@@ -2,6 +2,8 @@ from __future__ import absolute_import, division, unicode_literals
from types import ModuleType from types import ModuleType
from six import text_type
try: try:
import xml.etree.cElementTree as default_etree import xml.etree.cElementTree as default_etree
except ImportError: except ImportError:
@@ -9,7 +11,26 @@ except ImportError:
__all__ = ["default_etree", "MethodDispatcher", "isSurrogatePair", __all__ = ["default_etree", "MethodDispatcher", "isSurrogatePair",
"surrogatePairToCodepoint", "moduleFactoryFactory"] "surrogatePairToCodepoint", "moduleFactoryFactory",
"supports_lone_surrogates"]
# Platforms not supporting lone surrogates (\uD800-\uDFFF) should be
# caught by the below test. In general this would be any platform
# using UTF-16 as its encoding of unicode strings, such as
# Jython. This is because UTF-16 itself is based on the use of such
# surrogates, and there is no mechanism to further escape such
# escapes.
try:
_x = eval('"\\uD800"')
if not isinstance(_x, text_type):
# We need this with u"" because of http://bugs.jython.org/issue2039
_x = eval('u"\\uD800"')
assert isinstance(_x, text_type)
except:
supports_lone_surrogates = False
else:
supports_lone_surrogates = True
class MethodDispatcher(dict): class MethodDispatcher(dict):
+1 -1
View File
@@ -1,4 +1,4 @@
#!/usr/bin/python #!/usr/bin/python
from pynma import PyNMA from .pynma import PyNMA
Regular → Executable
+39 -28
View File
@@ -1,12 +1,20 @@
#!/usr/bin/python #!/usr/bin/python
from xml.dom.minidom import parseString from xml.dom.minidom import parseString
from httplib import HTTPSConnection
from urllib import urlencode
__version__ = "0.1" try:
from http.client import HTTPSConnection
except ImportError:
from httplib import HTTPSConnection
API_SERVER = 'nma.usk.bz' try:
from urllib.parse import urlencode
except ImportError:
from urllib import urlencode
__version__ = "1.0"
API_SERVER = 'www.notifymyandroid.com'
ADD_PATH = '/publicapi/notify' ADD_PATH = '/publicapi/notify'
USER_AGENT="PyNMA/v%s"%__version__ USER_AGENT="PyNMA/v%s"%__version__
@@ -18,14 +26,14 @@ def uniq_preserve(seq): # Dave Kirby
def uniq(seq): def uniq(seq):
# Not order preserving # Not order preserving
return {}.fromkeys(seq).keys() return list({}.fromkeys(seq).keys())
class PyNMA(object): class PyNMA(object):
"""PyNMA(apikey=[], developerkey=None) """PyNMA(apikey=[], developerkey=None)
takes 2 optional arguments: takes 2 optional arguments:
- (opt) apykey: might me a string containing 1 key or an array of keys - (opt) apykey: might me a string containing 1 key or an array of keys
- (opt) developerkey: where you can store your developer key - (opt) developerkey: where you can store your developer key
""" """
def __init__(self, apikey=[], developerkey=None): def __init__(self, apikey=[], developerkey=None):
self._developerkey = None self._developerkey = None
@@ -60,19 +68,20 @@ class PyNMA(object):
if type(developerkey) == str and len(developerkey) == 48: if type(developerkey) == str and len(developerkey) == 48:
self._developerkey = developerkey self._developerkey = developerkey
def push(self, application="", event="", description="", url="", priority=0, batch_mode=False): def push(self, application="", event="", description="", url="", contenttype=None, priority=0, batch_mode=False, html=False):
"""Pushes a message on the registered API keys. """Pushes a message on the registered API keys.
takes 5 arguments: takes 5 arguments:
- (req) application: application name [256] - (req) application: application name [256]
- (req) event: event name [1000] - (req) event: event name [1000]
- (req) description: description [10000] - (req) description: description [10000]
- (opt) url: url [512] - (opt) url: url [512]
- (opt) priority: from -2 (lowest) to 2 (highest) (def:0) - (opt) contenttype: Content Type (act: None (plain text) or text/html)
- (opt) batch_mode: call API 5 by 5 (def:False) - (opt) priority: from -2 (lowest) to 2 (highest) (def:0)
- (opt) batch_mode: push to all keys at once (def:False)
Warning: using batch_mode will return error only if all API keys are bad - (opt) html: shortcut for contenttype=text/html
cf: http://nma.usk.bz/api.php Warning: using batch_mode will return error only if all API keys are bad
""" cf: http://nma.usk.bz/api.php
"""
datas = { datas = {
'application': application[:256].encode('utf8'), 'application': application[:256].encode('utf8'),
'event': event[:1024].encode('utf8'), 'event': event[:1024].encode('utf8'),
@@ -83,6 +92,9 @@ class PyNMA(object):
if url: if url:
datas['url'] = url[:512] datas['url'] = url[:512]
if contenttype == "text/html" or html == True: # Currently only accepted content type
datas['content-type'] = "text/html"
if self._developerkey: if self._developerkey:
datas['developerkey'] = self._developerkey datas['developerkey'] = self._developerkey
@@ -94,10 +106,9 @@ class PyNMA(object):
res = self.callapi('POST', ADD_PATH, datas) res = self.callapi('POST', ADD_PATH, datas)
results[key] = res results[key] = res
else: else:
for i in range(0, len(self._apikey), 5): datas['apikey'] = ",".join(self._apikey)
datas['apikey'] = ",".join(self._apikey[i:i+5]) res = self.callapi('POST', ADD_PATH, datas)
res = self.callapi('POST', ADD_PATH, datas) results[datas['apikey']] = res
results[datas['apikey']] = res
return results return results
def callapi(self, method, path, args): def callapi(self, method, path, args):
@@ -110,7 +121,7 @@ class PyNMA(object):
try: try:
res = self._parse_reponse(resp.read()) res = self._parse_reponse(resp.read())
except Exception, e: except Exception as e:
res = {'type': "pynmaerror", res = {'type': "pynmaerror",
'code': 600, 'code': 600,
'message': str(e) 'message': str(e)
@@ -124,12 +135,12 @@ class PyNMA(object):
for elem in root.childNodes: for elem in root.childNodes:
if elem.nodeType == elem.TEXT_NODE: continue if elem.nodeType == elem.TEXT_NODE: continue
if elem.tagName == 'success': if elem.tagName == 'success':
res = dict(elem.attributes.items()) res = dict(list(elem.attributes.items()))
res['message'] = "" res['message'] = ""
res['type'] = elem.tagName res['type'] = elem.tagName
return res return res
if elem.tagName == 'error': if elem.tagName == 'error':
res = dict(elem.attributes.items()) res = dict(list(elem.attributes.items()))
res['message'] = elem.firstChild.nodeValue res['message'] = elem.firstChild.nodeValue
res['type'] = elem.tagName res['type'] = elem.tagName
return res return res