June 2018 Update
This commit is contained in:
parent
ba8067c3b7
commit
22f33d4004
5278 changed files with 84726 additions and 14379 deletions
|
|
@ -23,13 +23,13 @@ The relevant part of that page source looked like this:
|
|||
|
||||
<TITLE>What time is it?</TITLE>
|
||||
<H2> US Naval Observatory Master Clock Time</H2> <H3><PRE>
|
||||
<BR>Jul. 27, 22:57:22 UTC Universal Time
|
||||
<BR>Jul. 27, 06:57:22 PM EDT Eastern Time
|
||||
<BR>Jul. 27, 05:57:22 PM CDT Central Time
|
||||
<BR>Jul. 27, 04:57:22 PM MDT Mountain Time
|
||||
<BR>Jul. 27, 03:57:22 PM PDT Pacific Time
|
||||
<BR>Jul. 27, 02:57:22 PM AKDT Alaska Time
|
||||
<BR>Jul. 27, 12:57:22 PM HAST Hawaii-Aleutian Time
|
||||
<BR>Jul. 27, 22:57:22 UTC Universal Time
|
||||
<BR>Jul. 27, 06:57:22 PM EDT Eastern Time
|
||||
<BR>Jul. 27, 05:57:22 PM CDT Central Time
|
||||
<BR>Jul. 27, 04:57:22 PM MDT Mountain Time
|
||||
<BR>Jul. 27, 03:57:22 PM PDT Pacific Time
|
||||
<BR>Jul. 27, 02:57:22 PM AKDT Alaska Time
|
||||
<BR>Jul. 27, 12:57:22 PM HAST Hawaii-Aleutian Time
|
||||
|
||||
...
|
||||
</pre>
|
||||
|
|
|
|||
|
|
@ -6,7 +6,7 @@
|
|||
|
||||
: get-time
|
||||
read-url
|
||||
/<BR>.*?(\d{2}:\d{2}:\d{2})\sUTC/
|
||||
/<BR>.*?(\d{2}:\d{2}:\d{2})\sUTC/
|
||||
tuck r:match if
|
||||
1 r:@ . cr
|
||||
then ;
|
||||
|
|
|
|||
|
|
@ -1,9 +1,9 @@
|
|||
when ScrapeButton.Click do
|
||||
set ScrapeWeb.Url to SourceTextBox.Text
|
||||
call ScrapeWeb.Get
|
||||
set ScrapeWeb.Url to SourceTextBox.Text
|
||||
call ScrapeWeb.Get
|
||||
|
||||
when ScrapeWeb.GotText url,responseCode,responseType,responseContent do
|
||||
initialize local Left to split at first text (text: get responseContent, at: PreTextBox.Text)
|
||||
initialize local Right to "" in
|
||||
set Right to select list item (list: get Left, index: 2)
|
||||
set ResultLabel.Text to select list item (list: split at first (text:get Right, at: PostTextBox.Text), index: 1)
|
||||
initialize local Left to split at first text (text: get responseContent, at: PreTextBox.Text)
|
||||
initialize local Right to "" in
|
||||
set Right to select list item (list: get Left, index: 2)
|
||||
set ResultLabel.Text to select list item (list: split at first (text:get Right, at: PostTextBox.Text), index: 1)
|
||||
|
|
|
|||
|
|
@ -3,58 +3,58 @@ Class Utils.Net [ Abstract ]
|
|||
|
||||
ClassMethod ExtractHTMLData(pHost As %String = "", pPath As %String = "", pRegEx As %String = "", Output list As %List) As %Status
|
||||
{
|
||||
// implement error handling
|
||||
Try {
|
||||
// implement error handling
|
||||
Try {
|
||||
|
||||
// some initialisation
|
||||
Set list="", sc=$$$OK
|
||||
|
||||
// check input parameters
|
||||
If $Match(pHost, "^([a-zA-Z0-9]([a-zA-Z0-9\-]{0,61}[a-zA-Z0-9])?\.)+[a-zA-Z]{2,6}$")=0 {
|
||||
Set sc=$$$ERROR($$$GeneralError, "Invalid host name.")
|
||||
Quit
|
||||
}
|
||||
|
||||
// create http request and get page
|
||||
Set req=##class(%Net.HttpRequest).%New()
|
||||
Set req.Server=pHost
|
||||
Do req.Get(pPath)
|
||||
|
||||
// check for success
|
||||
If $Extract(req.HttpResponse.StatusCode)'=2 {
|
||||
Set sc=$$$ERROR($$$GeneralError, "Page not loaded.")
|
||||
Quit
|
||||
}
|
||||
|
||||
// read http response stream
|
||||
Set html=req.HttpResponse.Data
|
||||
Set html.LineTerminator=$Char(10)
|
||||
Set sc=html.Rewind()
|
||||
|
||||
// read http response stream
|
||||
While 'html.AtEnd {
|
||||
Set line=html.ReadLine(, .sc, .eol)
|
||||
Set pos=$Locate(line, pRegEx)
|
||||
If pos {
|
||||
Set parse=$Piece($Extract(line, pos, *), $Char(9))
|
||||
Set slot=$ListLength(list)+1
|
||||
Set $List(list, slot)=parse
|
||||
}
|
||||
}
|
||||
|
||||
} Catch err {
|
||||
|
||||
// an error has occurred
|
||||
If err.Name="<REGULAR EXPRESSION>" {
|
||||
Set sc=$$$ERROR($$$GeneralError, "Invalid regular expression.")
|
||||
} Else {
|
||||
Set sc=$$$ERROR($$$CacheError, $ZError)
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
// return status
|
||||
Quit sc
|
||||
// some initialisation
|
||||
Set list="", sc=$$$OK
|
||||
|
||||
// check input parameters
|
||||
If $Match(pHost, "^([a-zA-Z0-9]([a-zA-Z0-9\-]{0,61}[a-zA-Z0-9])?\.)+[a-zA-Z]{2,6}$")=0 {
|
||||
Set sc=$$$ERROR($$$GeneralError, "Invalid host name.")
|
||||
Quit
|
||||
}
|
||||
|
||||
// create http request and get page
|
||||
Set req=##class(%Net.HttpRequest).%New()
|
||||
Set req.Server=pHost
|
||||
Do req.Get(pPath)
|
||||
|
||||
// check for success
|
||||
If $Extract(req.HttpResponse.StatusCode)'=2 {
|
||||
Set sc=$$$ERROR($$$GeneralError, "Page not loaded.")
|
||||
Quit
|
||||
}
|
||||
|
||||
// read http response stream
|
||||
Set html=req.HttpResponse.Data
|
||||
Set html.LineTerminator=$Char(10)
|
||||
Set sc=html.Rewind()
|
||||
|
||||
// read http response stream
|
||||
While 'html.AtEnd {
|
||||
Set line=html.ReadLine(, .sc, .eol)
|
||||
Set pos=$Locate(line, pRegEx)
|
||||
If pos {
|
||||
Set parse=$Piece($Extract(line, pos, *), $Char(9))
|
||||
Set slot=$ListLength(list)+1
|
||||
Set $List(list, slot)=parse
|
||||
}
|
||||
}
|
||||
|
||||
} Catch err {
|
||||
|
||||
// an error has occurred
|
||||
If err.Name="<REGULAR EXPRESSION>" {
|
||||
Set sc=$$$ERROR($$$GeneralError, "Invalid regular expression.")
|
||||
} Else {
|
||||
Set sc=$$$ERROR($$$CacheError, $ZError)
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
// return status
|
||||
Quit sc
|
||||
}
|
||||
|
||||
}
|
||||
|
|
|
|||
31
Task/Web-scraping/Ceylon/web-scraping.ceylon
Normal file
31
Task/Web-scraping/Ceylon/web-scraping.ceylon
Normal file
|
|
@ -0,0 +1,31 @@
|
|||
import ceylon.uri {
|
||||
parse
|
||||
}
|
||||
import ceylon.http.client {
|
||||
get
|
||||
}
|
||||
|
||||
shared void run() {
|
||||
|
||||
// apparently the cgi link is deprecated?
|
||||
value oldUri = "http://tycho.usno.navy.mil/cgi-bin/timer.pl";
|
||||
value newUri = "http://tycho.usno.navy.mil/timer.pl";
|
||||
|
||||
value contents = downloadContents(newUri);
|
||||
value time = extractTime(contents);
|
||||
print(time else "nothing found");
|
||||
}
|
||||
|
||||
String downloadContents(String uriString) {
|
||||
value uri = parse(uriString);
|
||||
value request = get(uri);
|
||||
value response = request.execute();
|
||||
return response.contents;
|
||||
}
|
||||
|
||||
String? extractTime(String contents) =>
|
||||
contents
|
||||
.lines
|
||||
.filter((String element) => element.contains("UTC"))
|
||||
.first
|
||||
?.substring(4, 21);
|
||||
|
|
@ -146,7 +146,7 @@ begin
|
|||
for i := 0 to Pred(HTTPResponse.Count) do
|
||||
begin
|
||||
{ The line we're looking for is something like this:
|
||||
<BR>May. 04. 21:55:19 UTC Universal Time }
|
||||
<BR>May. 04. 21:55:19 UTC Universal Time }
|
||||
|
||||
// Check each line
|
||||
if Pos('UTC',HTTPResponse[i]) > 0 then
|
||||
|
|
|
|||
|
|
@ -4,7 +4,7 @@
|
|||
-define(Match, "<BR>(.+ UTC)").
|
||||
|
||||
main() ->
|
||||
inets:start(),
|
||||
{ok, {_Status, _Header, HTML}} = httpc:request(?Url),
|
||||
{match, [Time]} = re:run(HTML, ?Match, [{capture, all_but_first, binary}]),
|
||||
io:format("~s~n",[Time]).
|
||||
inets:start(),
|
||||
{ok, {_Status, _Header, HTML}} = httpc:request(?Url),
|
||||
{match, [Time]} = re:run(HTML, ?Match, [{capture, all_but_first, binary}]),
|
||||
io:format("~s~n",[Time]).
|
||||
|
|
|
|||
|
|
@ -6,19 +6,19 @@ import java.net.URLConnection;
|
|||
|
||||
|
||||
public class WebTime{
|
||||
public static void main(String[] args){
|
||||
try{
|
||||
URL address = new URL(
|
||||
"http://tycho.usno.navy.mil/cgi-bin/timer.pl");
|
||||
URLConnection conn = address.openConnection();
|
||||
BufferedReader in = new BufferedReader(
|
||||
new InputStreamReader(conn.getInputStream()));
|
||||
String line;
|
||||
while(!(line = in.readLine()).contains("UTC"));
|
||||
System.out.println(line.substring(4));
|
||||
}catch(IOException e){
|
||||
System.err.println("error connecting to server.");
|
||||
e.printStackTrace();
|
||||
}
|
||||
}
|
||||
public static void main(String[] args){
|
||||
try{
|
||||
URL address = new URL(
|
||||
"http://tycho.usno.navy.mil/cgi-bin/timer.pl");
|
||||
URLConnection conn = address.openConnection();
|
||||
BufferedReader in = new BufferedReader(
|
||||
new InputStreamReader(conn.getInputStream()));
|
||||
String line;
|
||||
while(!(line = in.readLine()).contains("UTC"));
|
||||
System.out.println(line.substring(4));
|
||||
}catch(IOException e){
|
||||
System.err.println("error connecting to server.");
|
||||
e.printStackTrace();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -8,8 +8,8 @@ function getusnotime()
|
|||
@sprintf "get(%s)\n => %s" url err
|
||||
end
|
||||
isa(s, Requests.Response) || return (s, false)
|
||||
t = match(r"(?<=<BR>)(.*?UTC)", readall(s))
|
||||
isa(t, RegexMatch) || return (@sprintf("raw html:\n %s", readall(s)), false)
|
||||
t = match(r"(?<=<BR>)(.*?UTC)", readstring(s))
|
||||
isa(t, RegexMatch) || return (@sprintf("raw html:\n %s", readstring(s)), false)
|
||||
return (t.match, true)
|
||||
end
|
||||
|
||||
|
|
|
|||
|
|
@ -1,13 +1,13 @@
|
|||
/* have to be used
|
||||
local(raw_htmlstring = '<TITLE>What time is it?</TITLE>
|
||||
<H2> US Naval Observatory Master Clock Time</H2> <H3><PRE>
|
||||
<BR>Jul. 27, 22:57:22 UTC Universal Time
|
||||
<BR>Jul. 27, 06:57:22 PM EDT Eastern Time
|
||||
<BR>Jul. 27, 05:57:22 PM CDT Central Time
|
||||
<BR>Jul. 27, 04:57:22 PM MDT Mountain Time
|
||||
<BR>Jul. 27, 03:57:22 PM PDT Pacific Time
|
||||
<BR>Jul. 27, 02:57:22 PM AKDT Alaska Time
|
||||
<BR>Jul. 27, 12:57:22 PM HAST Hawaii-Aleutian Time
|
||||
<BR>Jul. 27, 22:57:22 UTC Universal Time
|
||||
<BR>Jul. 27, 06:57:22 PM EDT Eastern Time
|
||||
<BR>Jul. 27, 05:57:22 PM CDT Central Time
|
||||
<BR>Jul. 27, 04:57:22 PM MDT Mountain Time
|
||||
<BR>Jul. 27, 03:57:22 PM PDT Pacific Time
|
||||
<BR>Jul. 27, 02:57:22 PM AKDT Alaska Time
|
||||
<BR>Jul. 27, 12:57:22 PM HAST Hawaii-Aleutian Time
|
||||
</PRE></H3>
|
||||
')
|
||||
*/
|
||||
|
|
@ -16,8 +16,8 @@ local(raw_htmlstring = '<TITLE>What time is it?</TITLE>
|
|||
local(raw_htmlstring = string(include_url('http://tycho.usno.navy.mil/cgi-bin/timer.pl')))
|
||||
|
||||
local(
|
||||
reg_exp = regexp(-find = `<br>(.*?) UTC`, -input = #raw_htmlstring, -ignorecase),
|
||||
datepart_txt = #reg_exp -> find ? #reg_exp -> matchstring(1) | string
|
||||
reg_exp = regexp(-find = `<br>(.*?) UTC`, -input = #raw_htmlstring, -ignorecase),
|
||||
datepart_txt = #reg_exp -> find ? #reg_exp -> matchstring(1) | string
|
||||
)
|
||||
|
||||
#datepart_txt
|
||||
|
|
|
|||
|
|
@ -3,6 +3,6 @@ ix = [findstr(s,'<BR>'), length(s)+1];
|
|||
for k = 2:length(ix)
|
||||
tok = s(ix(k-1)+4:ix(k)-1);
|
||||
if findstr(tok,'UTC')
|
||||
disp(tok);
|
||||
disp(tok);
|
||||
end;
|
||||
end;
|
||||
|
|
|
|||
2
Task/Web-scraping/Maple/web-scraping.maple
Normal file
2
Task/Web-scraping/Maple/web-scraping.maple
Normal file
|
|
@ -0,0 +1,2 @@
|
|||
text := URL:-Get("http://tycho.usno.navy.mil/cgi-bin/timer.pl"):
|
||||
printf(StringTools:-StringSplit(text,"<BR>")[2]);
|
||||
|
|
@ -1,7 +1,7 @@
|
|||
<?
|
||||
|
||||
echo preg_replace(
|
||||
"/^.*<BR>(.*) UTC.*$/su",
|
||||
"\\1",
|
||||
file_get_contents('http://tycho.usno.navy.mil/cgi-bin/timer.pl')
|
||||
"/^.*<BR>(.*) UTC.*$/su",
|
||||
"\\1",
|
||||
file_get_contents('http://tycho.usno.navy.mil/cgi-bin/timer.pl')
|
||||
);
|
||||
|
|
|
|||
|
|
@ -1,8 +1,8 @@
|
|||
REBOL [
|
||||
Title: "Web Scraping"
|
||||
Author: oofoe
|
||||
Date: 2009-12-07
|
||||
URL: http://rosettacode.org/wiki/Web_Scraping
|
||||
Title: "Web Scraping"
|
||||
Author: oofoe
|
||||
Date: 2009-12-07
|
||||
URL: http://rosettacode.org/wiki/Web_Scraping
|
||||
]
|
||||
|
||||
; Notice that REBOL understands unquoted URL's:
|
||||
|
|
|
|||
|
|
@ -1,10 +1,10 @@
|
|||
import scala.io.Source
|
||||
|
||||
object WebTime extends Application {
|
||||
val text = Source.fromURL("http://tycho.usno.navy.mil/cgi-bin/timer.pl")
|
||||
val utc = text.getLines.find(_.contains("UTC"))
|
||||
utc match {
|
||||
case Some(s) => println(s.substring(4))
|
||||
case _ => println("error")
|
||||
}
|
||||
val text = Source.fromURL("http://tycho.usno.navy.mil/cgi-bin/timer.pl")
|
||||
val utc = text.getLines.find(_.contains("UTC"))
|
||||
utc match {
|
||||
case Some(s) => println(s.substring(4))
|
||||
case _ => println("error")
|
||||
}
|
||||
}
|
||||
|
|
|
|||
38
Task/Web-scraping/VBA/web-scraping.vba
Normal file
38
Task/Web-scraping/VBA/web-scraping.vba
Normal file
|
|
@ -0,0 +1,38 @@
|
|||
Rem add Microsoft VBScript Regular Expression X.X to your Tools References
|
||||
|
||||
Function GetUTC() As String
|
||||
Url = "http://tycho.usno.navy.mil/cgi-bin/timer.pl"
|
||||
With CreateObject("MSXML2.XMLHTTP.6.0")
|
||||
.Open "GET", Url, False
|
||||
.send
|
||||
arrt = Split(.responseText, vbLf)
|
||||
End With
|
||||
For Each t In arrt
|
||||
If InStr(t, "UTC") Then
|
||||
GetUTC = StripHttpTags(t)
|
||||
|
||||
Exit For
|
||||
End If
|
||||
Next
|
||||
End Function
|
||||
|
||||
Function StripHttpTags(s)
|
||||
With New RegExp
|
||||
.Global = True
|
||||
.Pattern = "\<.+?\>"
|
||||
If .Test(s) Then
|
||||
StripHttpTags = .Replace(s, "")
|
||||
Else
|
||||
StripHttpTags = s
|
||||
End If
|
||||
End With
|
||||
End Function
|
||||
|
||||
Sub getTime()
|
||||
Rem starting point
|
||||
Dim ReturnValue As String
|
||||
ReturnValue = GetUTC
|
||||
Rem debug.print can be removed
|
||||
Debug.Print ReturnValue
|
||||
MsgBox (ReturnValue)
|
||||
End Sub
|
||||
|
|
@ -1,28 +1,29 @@
|
|||
Function GetUTC()
|
||||
url = "http://tycho.usno.navy.mil/cgi-bin/timer.pl"
|
||||
With CreateObject("MSXML2.XMLHTTP.6.0")
|
||||
.open "GET", url, False
|
||||
.send
|
||||
arrt = Split(.responseText,vbLf)
|
||||
End With
|
||||
For Each t In arrt
|
||||
If InStr(t,"UTC") Then
|
||||
GetUTC = StripHttpTags(t)
|
||||
Exit For
|
||||
End If
|
||||
Next
|
||||
Function GetUTC() As String
|
||||
Url = "http://tycho.usno.navy.mil/cgi-bin/timer.pl"
|
||||
With CreateObject("MSXML2.XMLHTTP.6.0")
|
||||
.Open "GET", Url, False
|
||||
.send
|
||||
arrt = Split(.responseText, vbLf)
|
||||
End With
|
||||
For Each t In arrt
|
||||
If InStr(t, "UTC") Then
|
||||
GetUTC = StripHttpTags(t)
|
||||
|
||||
Exit For
|
||||
End If
|
||||
Next
|
||||
End Function
|
||||
|
||||
Function StripHttpTags(s)
|
||||
With New RegExp
|
||||
.Global = True
|
||||
.Pattern = "\<.+?\>"
|
||||
If .Test(s) Then
|
||||
StripHttpTags = .Replace(s,"")
|
||||
Else
|
||||
StripHttpTags = s
|
||||
End If
|
||||
End With
|
||||
With New RegExp
|
||||
.Global = True
|
||||
.Pattern = "\<.+?\>"
|
||||
If .Test(s) Then
|
||||
StripHttpTags = .Replace(s, "")
|
||||
Else
|
||||
StripHttpTags = s
|
||||
End If
|
||||
End With
|
||||
End Function
|
||||
|
||||
WScript.StdOut.Write GetUTC
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue