Wayback Machine Small Bug Fixes

Fixes encoded ampersands on Wayback Machine's captures graph and problems that arise when trailing slashes are missing in an URL and other small issues universally present in all crawled sites

Tính đến 20-01-2016. Xem phiên bản mới nhất.

You will need to install an extension such as Tampermonkey, Greasemonkey or Violentmonkey to install this script.

Bạn sẽ cần cài đặt một tiện ích mở rộng như Tampermonkey hoặc Violentmonkey để cài đặt kịch bản này.

You will need to install an extension such as Tampermonkey or Violentmonkey to install this script.

You will need to install an extension such as Tampermonkey or Userscripts to install this script.

You will need to install an extension such as Tampermonkey to install this script.

You will need to install a user script manager extension to install this script.

(Tôi đã có Trình quản lý tập lệnh người dùng, hãy cài đặt nó!)

You will need to install an extension such as Stylus to install this style.

You will need to install an extension such as Stylus to install this style.

You will need to install an extension such as Stylus to install this style.

You will need to install a user style manager extension to install this style.

You will need to install a user style manager extension to install this style.

You will need to install a user style manager extension to install this style.

(I already have a user style manager, let me install it!)

// ==UserScript==
// @name          Wayback Machine Small Bug Fixes
// @namespace     DoomTay
// @description   Fixes encoded ampersands on Wayback Machine's captures graph and problems that arise when trailing slashes are missing in an URL and other small issues universally present in all crawled sites
// @version       1.2.6
// @include       http://web.archive.org/web/*
// @include       http://wayback.archive.org/web/*
// @include       https://web.archive.org/web/*
// @include       https://wayback.archive.org/web/*
// @run-at        document-start
// @exclude       /\*/
// @grant         none

// ==/UserScript==

var toolbarNav = document.getElementById("wm-graph-anchor");
var lastFolder = window.location.href.substring(window.location.href.lastIndexOf("/") + 1);
var pics = document.images;
var backgrounds = document.querySelectorAll("[background]");
var shouldHaveTrailingSlash = (window.location.href.lastIndexOf(".") < window.location.href.lastIndexOf("/") || window.location.href.substring(window.location.href.lastIndexOf("//") + 2) == lastFolder) && lastFolder.indexOf("?") == -1;
var hasTrailingSlash = window.location.href.lastIndexOf("/") == window.location.href.length - 1;
var domain = window.location.href.substring(0,window.location.href.indexOf("/",window.location.href.lastIndexOf("//") + 2));

function fixToolbar()
{
	while(toolbarNav.href.indexOf("&amp;") > -1) toolbarNav.href = toolbarNav.href.replace("&amp;","&");
}

//Fix cases of &amp; in the capture graph
if(toolbarNav) fixToolbar();

if(!document.getElementsByTagName("base")[0])
{
	var base = document.createElement("base");
	if(shouldHaveTrailingSlash && !hasTrailingSlash) base.href = window.location.href + "/";
	else if((!hasTrailingSlash && !shouldHaveTrailingSlash) || hasTrailingSlash) base.href = document.baseURI;
	else base.href = domain + "/";
	document.head.appendChild(base);
}
	

for(var i = 0; i < pics.length; i++)
{
	//Skip over stuff related to the Wayback Machine toolbar and data URIs
	if((document.getElementById("wm-ipp") && document.getElementById("wm-ipp").contains(pics[i])) || pics[i].src.indexOf("data:") > -1) continue;
	//Refresh images in case the "base url" had to be modified.
	pics[i].src = pics[i].src;
	//For whatever reason, some images will point to within Internet Archive's "main" servers, instead of the crawled site. This attempts to fix that.
	if(pics[i].src.indexOf(document.domain + "/web") == -1) pics[i].src = domain.substring(0,domain.indexOf("/http") + 1) + pics[i].src.substring(pics[i].src.indexOf("/http") + 1);
}

for(var b = 0; b < backgrounds.length; b++)
{
	var bg = backgrounds[b].background || backgrounds[b].getAttribute("background");
	//Skip over stuff related to the Wayback Machine toolbar and data URIs
	if((document.getElementById("wm-ipp") && document.getElementById("wm-ipp").contains(backgrounds[b])) || bg.indexOf("data:") > -1) continue;
	//Refresh images in case the "base url" had to be modified.
	changeBackground(backgrounds[b],bg);
	//For whatever reason, some images will point to within Internet Archive's "main" servers, instead of the crawled site. This attempts to fix that.
	if(relativeToAbsolute(bg).indexOf(document.domain + "/web") == -1) 
	{
		var absoluteBG = relativeToAbsolute(bg)
		changeBackground(backgrounds[b],domain.substring(0,domain.indexOf("/http") + 1) + absoluteBG.substring(absoluteBG.indexOf("/http") + 1));
	}
}

function relativeToAbsolute(bgURL)
{
	var img = new Image();
	img.src = bgURL;
	return img.src;
}

function changeBackground(node, newBackground)
{
	if(node.background) node.background = newBackground;
	else if(backgrounds[b].getAttribute("background")) backgrounds[b].setAttribute("background",newBackground);
}

var observer = new MutationObserver(function(mutations) {
	mutations.forEach(function(mutation) {
		if(mutation.target.id == "wm-graph-anchor") toolbarNav = mutation.target;
		if(mutation.type == "attributes" && mutation.target == toolbarNav) fixToolbar();
		if(mutation.target.nodeName == "IMG" && mutation.attributeName == "src" && mutation.target.src.indexOf(document.domain + "/web") == -1)
		{
			mutation.target.src = domain.substring(0,domain.indexOf("/http") + 1) + mutation.target.src.substring(mutation.target.src.indexOf("/http") + 1);
		}
		if(mutation.attributeName == "background" && (mutation.target.getAttribute("background") || mutation.target.background).indexOf(document.domain + "/web") == -1)
		{
			var bg = mutation.target.background || mutation.target.getAttribute("background");
			var absoluteBG = relativeToAbsolute(bg);
			changeBackground(mutation.target,domain.substring(0,domain.indexOf("/http") + 1) + absoluteBG.substring(absoluteBG.indexOf("/http") + 1));
		}
	});    
});

var config = { attributes: true, childList: true, characterData: true, subtree: true };
observer.observe(document.body || document, config);