Repository navigation
Expand file tree
/
Copy pathoembed_scraper.php
More file actions
174 lines (155 loc) · 4.67 KB
/
Copy pathoembed_scraper.php
File metadata and controls
174 lines (155 loc) · 4.67 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
<?php
/**
* @file
* Endpoint provider that returns oEmbed JSON from arbitrary URLs.
*
* This is not much more than a layer in front of
* https://github.com/oscarotero/Embed
* which does the semantic scraping and resolves both opengraph and
* Dublin Core metadata into common fields.
*
* Plus a hundred other resolvers, handlers and providers.
*
* I regurgitate these fields in OpenGraph format.
*/
/**
* Call this page directly with parameters per the opengraph spec:
* Required:
* 'url'
* Optional:
* 'maxwidth'
* 'maxheight'
* 'format'
*
* To prevent overloading and all manner of things that could go wrong,
* we accept incoming requests FROM OUR OWN DOMAIN ONLY.
* This provides a tiny oEmbed provider, running locally, that
* provides info about remote resources.
*
* Plugging in to the oEmbed ecosystem by just providing another endpoint
* that we consume ourself is the most-decoupled, least-code way forward today.
*
* Does require that the hosting server can see itself.
*/
// Configurable.
/**
* Clients IPs that are allowed to make requests of this service.
*
* Set this to empty to remove all restrictions, though you should
* replace this security layer with your own access controls. See README.
*
* If your server talks to itself using a different IP, add it here.
*/
$allowed_clients = array(
$_SERVER['REMOTE_ADDR'],
);
// Begin preparation.
$params = $_REQUEST;
// Check incoming client is allowed.
if (!empty($allowed_clients)) {
$client = $_SERVER['REMOTE_ADDR'];
if (!in_array($client, $allowed_clients)) {
http_response_code(403);
print "Requests to this service from $client are currently disallowed";
exit();
}
}
// Validate request.
if (empty($params['url'])) {
// Bad request, required argument not given.
http_response_code(400);
print 'No URL provided. You should pass the ?url= parameter in this request. Hint, you can also set <a href="?format=text">?format=text</a> for debugging.';
print '<form><input type="text" name="url" size="128"/><select name="format"><option>json</option><option>text</option></select><input type="submit"></form>';
exit();
}
$url = $params['url'];
$url_parts = parse_url($url);
// More validation needed? Sanitize expected input at least.
$maxwidth = @(int) $params['maxwidth'];
$maxheight = @(int) $params['maxheight'];
if (!empty($params['format']) && in_array($params['format'], array('json', 'xml', 'text'))) {
$format = $params['format'];
}
else {
$format = 'json';
}
$defaults = array(
"provider_name" => "oembed scraper",
"provider_url" => "http://standards.net.nz/oembed_scraper",
'width' => NULL,
'height' => NULL,
'thumbnail_url' => NULL,
'thumbnail_width' => NULL,
'thumbnail_height' => NULL,
'version' => '1.0',
);
// Begin request.
// @see https://github.com/oscarotero/Embed
// We are either a stand-alone instance that has built composer,
// or a library that composer has pulled in for someone else.
if (file_exists('vendor/autoload.php')) {
include_once 'vendor/autoload.php';
} else if (file_exists('../../autoload.php')) {
include_once '../../autoload.php';
}
use Embed\Embed;
/**
* @var \Embed\Adapters\AdapterInterface $info
*/
$info = Embed::create($url);
// Start building the oEmbed data struct.
$oembed = array(
"type" => "link",
"url" => $info->url,
"title" => $info->title,
"description" => $info->description,
"author_name" => $info->authorName,
"author_url" => $info->authorUrl,
);
// A successful semantic scrape from elsewhere MAY provide 'code' field
// containing markup to use. EG YouTube.
if (!empty("" . $info->code)) {
$oembed['html'] = $info->code;
}
else {
// Otherwise,
// Produce some adequate HTML, if none is provided.
$teaser = "<div class='oembed_scraper'>";
$teaser .= "<h3>" . $oembed['title'] . '</h3>';
$teaser .= "<div>" . $oembed['description'] . '</div>';
$teaser .= "</div>";
$oembed['html'] = $teaser;
}
$fields = array(
'image' => 'image',
'imageWidth' => 'thumbnail_width',
'imageHeight' => 'thumbnail_height',
'providerName' => 'provider_name',
'providerUrl' => 'provider_url',
'providerIcon' => 'provider_icon',
);
foreach ($fields as $source => $destination) {
// empty() on an attribute failed here !?
// Wacky. Must be a magic getter or something.
if (!empty("" . $info->{$source})) {
$oembed[$destination] = $info->{$source};
}
}
$oembed += $defaults;
switch ($format) {
case 'json':
header('Content-type: application/json');
echo json_encode($oembed);
exit;
case 'text':
print '<pre>';
print_r($oembed);
print '</pre>';
print '<hr/>';
exit;
case 'xml':
print '<pre>';
print "XML format TODO";
// @see array2XML or something?
exit;
}