Not a member of Pastebin yet?
Sign Up,
it unlocks many cool features!
- <?php
- /**
- * for latest version of this file, see https://pastebin.com/edit/rt7F6sMv
- *
- * This file will extract memora data from `UA_Data/resources.assets` file,
- * format it as the wiki requires, and either upload to the wiki, or display it.
- *
- * To use:
- * Place in the UA_Data folder and run it. Or run from anywhere, with the path
- * to the resources.assets file as an argument.
- *
- * If you want it to opload to the wiki, be sure to download/install apibot
- * and edit the apibot/logins.php file.
- *
- * Otherwise, set $USE_BOT=false;, and pipe the output to a file, eg:
- * php logs.php > memora.txt
- * The output will be UTF-8. but lacks a BOM so may not be shown that way in
- * some editors.
- */
- ini_set("memory_limit", "-1");
- set_time_limit(0);
- define(MB_CASE_TITLE, 2); // Because my PHP installation is broken.
- // Following requires: http://apibot.zavinagi.org/index.php/Installation
- // Set true if you want it to auto-overwrite the pages for you.
- $USE_BOT=true;
- $numLanguages = 11; // EN, ES, FR, IT, DE, CH, RU, KR, PT, JP
- if ($USE_BOT) {
- if (!is_dir('apibot')) {
- echo "The ./apibot/ folder does not exist.\n";
- echo "You can install it from http://apibot.zavinagi.org\n";
- echo "Alternatively, set \$USE_BOT=false; at the top of this script.\n";
- die(1);
- }
- require_once(dirname(__FILE__).'/apibot/settings.php');
- require_once(dirname(__FILE__).'/apibot/logins.php');
- require_once(dirname(__FILE__).'/apibot/core/core.php');
- require_once(dirname(__FILE__).'/apibot/interfaces/bridge/bridge.php');
- }
- // Default to the obvious file, but allow the filename to be pasted in.
- $filename = 'resources.assets';
- if (2 === $argc) {
- $filename = $argv[1];
- }
- $keywordSearch = [
- 'Abyssal Key' => '/Abyssal Key|Sun Key/i',
- 'Baldred' => '/Baldred/i',
- 'Cataclysm' => '/Cataclysm/',
- 'Cabirus' => '/Cabirus/i',
- 'Circle of Portals' => '/Circle of Portals/i',
- 'Deep Gap' => '/Deep Gap/i',
- 'Disenthralled' => '/Disenthralled/i',
- 'Dwarf' => '/Dwarf|Dwarves|Dwarven/i',
- 'Elf' => '/\b(Elf|Elves|Elven)\b/i',
- 'Executor Rubric' => '/Executor Rubric/',
- 'Expedition' => '/Expedition/',
- 'Galdwain' => '/Galdwain/i',
- 'Gilgamesh' => '/Gilgamesh/i',
- 'Goblin' => '/Goblin|Goblinfolk/i',
- 'Great Work' => '/Great Work/i',
- 'Grue' => '/\bGrues?\b/i',
- 'Hivemind' => '/Hivemind/i',
- 'Horn of Plenty' => '/Horn of Plenty/i',
- 'Ishtass' => '/Ishtass/i',
- 'Izanagi-no-Mikoto' => '/Izanagi-no-Mikoto/i',
- 'Khosnak' => '/Khosnak/i',
- 'Leaf' => '/\bLeaf\b/',
- 'Lich' => '/\bLich/i',
- 'Lower Dark' => '/Lower Dark/i',
- 'Marcaul' => '/Marcaul/i',
- 'Mediator' => '/mediator/i',
- 'Memora' => '/Memora(?!nd)/i',
- 'Mogwawg' => '/Mogwawg/i',
- 'Obsidian' => '/Obsidian/',
- 'Odin' => '/Odin/',
- 'Outcast' => '/Outcast/',
- 'Pir-Tama' => '/Pir-Tama|Pir Tama/i',
- 'Praxor' => '/Praxor/i',
- 'Rotworm' => '/Rot-?worm/i',
- 'Saurian' => '/Saurian|Lizardmen|Lizardman/i',
- 'Seers' => '/\bSeers?\b/',
- 'Slasher of Veils' => '/Slasher of Veils/i',
- 'Shambler' => '/Shambler/i',
- 'Stygian Abyss' => '/Abyss|Underworld|Avernus|Aornum|Xibalba|Ganzer/i',
- 'Sun Key' => '/Sun Key/i',
- 'Tinker' => '/\bTinker/i',
- 'Titans' => '/\bTitans?\b/i',
- 'Typhon' => '/Typhon/i',
- 'Undead' => '/Undead/i',
- 'Xefreyani' => '/Xefreyani/i',
- 'Zeus' => '/\bZeus/i',
- ];
- $extractor = new StringExtractor($filename);
- // Figure out where we can stop looking. +3 ints for label length, label, 0x0b.
- $lastStringStart = $extractor->fileLength - (4 * ($numLanguages + 3));
- // Loop over the file in chunks of 4 bytes (the file is int-aligned).
- for ($i = 0; $i < $lastStringStart; $i += 4) {
- $ptr = $i; // Temp pointer to read the data in, from $i onwards.
- $descData = null; // Data for the optional descriptor string.
- // Speed optimization: scan rapidly until we hit something that might be valid.
- if (
- ("\0" !== $extractor->data[$i+3]) ||
- ("\0" !== $extractor->data[$i+2]) ||
- (
- ("\0" === $extractor->data[$i+1]) &&
- ("\0" === $extractor->data[$i])
- )
- ) {
- continue;
- }
- $labelData = $extractor->readString($ptr); // Data for the initial label string.
- // Fail on error.
- if (0 !== $labelData->ret) {
- continue;
- }
- // Fail if it's not a valid label.
- if (!preg_match('#^Log\w+/[^/][-\'\w ]+$#', $labelData->str)) {
- continue;
- }
- // There are two optional strings here. I don't know what they are for.
- $descData1 = $extractor->readString($ptr, true);
- if (0 !== $labelData->ret) {
- echo '!!0!!' . $labelData->str . ':' . $descData->ret .':' . $extractor->stringError($descData) . " at 0x".dechex($ptr)."\n";
- continue;
- }
- $descData2 = $extractor->readString($ptr, true);
- if (0 !== $labelData->ret) {
- echo '!!1!!' . $labelData->str . ':' . $descData->ret .':' . $extractor->stringError($descData) . " at 0x".dechex($ptr)."\n";
- continue;
- }
- // Then one 0x0000000b, little-endian (so 0b 00 00 00).
- if ($extractor->readInt($ptr) !== 0x0b) {
- continue;
- }
- // Read in one string per language. Some strings may be zero length.
- $strings = [];
- $labelBits = explode('/', $labelData->str);
- for ($s = 0; $s < $numLanguages; $s++) {
- // Add this string to our array.
- $str = $extractor->readString($ptr, true);
- if (0 !== $labelData->ret) {
- echo '!!2!!' . $labelData->str .':' . $extractor->stringError($labelData) . " at 0x".dechex($ptr)."\n";
- continue 2;
- }
- if ('Logs' === $labelBits[0]) {
- $str->str = preg_replace('/\r?\n/', '<br>', $str->str);
- $str->str = preg_replace('/\t/', ' ', $str->str);
- }
- $strings[$s] = trim($str->str);
- }
- // We have found all strings for this label. Output!.
- $output[$labelBits[1]][$labelBits[0]] = $strings;
- // We don't need to scan the strings we already found for the start of other strings!
- // Subtract 4 because we'll add 4 in the loop increment.
- $i = $ptr - 4;
- }
- // If we're gonna be using a bot to output stuff, prepare it here.
- if ($USE_BOT) {
- $bridge = new Standalone_Bridge ($logins['underwiki'], $bot_settings);
- $numBotEdits = 0;
- }
- // Now we've gathered everything into one big array, we can parse out the data.
- foreach ($output as $name => $record) {
- // Record contains two parts, 'logs' and 'logtitles'
- $titleBits = preg_split('/\r?\n| - /', $record['LogTitles'][0], 2);
- switch(trim($titleBits[0])) {
- case 'Xefreyani of The Deep Elves':
- $author = 'Xefreyani';
- $icon = 'Elves_icon.png';
- break;
- case 'Khosnak, Mediator to The Shamblers':
- $author = 'Khosnak';
- $icon = 'Shamblers_icon.png';
- break;
- case 'Executor Rubric of The Expedition':
- $author = 'Executor Rubric';
- $icon = 'Expedition_icon.png';
- break;
- case 'Cabirus':
- $author = 'Cabirus';
- $icon = 'Cabirus_icon.png';
- break;
- default:
- $author = 'Anon';
- $icon = '';
- break;
- }
- // Strip the author names off the beginnings of all titles, and convert to title case.
- foreach ($record['LogTitles'] as $id => $title) {
- if (!empty($title)) {
- $bits = preg_split('/\r?\n| - /', $title, 2);
- if (2 == count($bits)) {
- $record['LogTitles'][$id] = trim($bits[1]);
- }
- $record['LogTitles'][$id] = mb_convert_case($record['LogTitles'][$id], MB_CASE_TITLE, "UTF-8");
- }
- }
- $keywords = [];
- foreach ($keywordSearch as $keyword => $regex) {
- if (preg_match($regex, $record['Logs'][0])) {
- $keywords []= $keyword;
- }
- }
- $memoraPage = "
- {{Infobox Memora
- |author=$author
- |icon=$icon
- |scene=?
- |location=?
- |keywords=" . implode(',', $keywords) . "\n"
- . (empty($record['LogTitles'][0]) ? '' : "|title={$record['LogTitles'][0]}\n")
- . (empty($record['LogTitles'][1]) ? '' : "|title_es={$record['LogTitles'][1]}\n")
- . (empty($record['LogTitles'][2]) ? '' : "|title_fr={$record['LogTitles'][2]}\n")
- . (empty($record['LogTitles'][3]) ? '' : "|title_it={$record['LogTitles'][3]}\n")
- . (empty($record['LogTitles'][4]) ? '' : "|title_de={$record['LogTitles'][4]}\n")
- . (empty($record['LogTitles'][5]) ? '' : "|title_ch={$record['LogTitles'][5]}\n")
- . (empty($record['LogTitles'][6]) ? '' : "|title_ru={$record['LogTitles'][6]}\n")
- . (empty($record['LogTitles'][7]) ? '' : "|title_kr={$record['LogTitles'][7]}\n")
- . (empty($record['LogTitles'][8]) ? '' : "|title_pt={$record['LogTitles'][8]}\n")
- . (empty($record['LogTitles'][9]) ? '' : "|title_jp={$record['LogTitles'][9]}\n")
- . (empty($record['Logs'][0]) ? '' : "|text={$record['Logs'][0]}\n")
- . (empty($record['Logs'][1]) ? '' : "|text_es={$record['Logs'][1]}\n")
- . (empty($record['Logs'][2]) ? '' : "|text_fr={$record['Logs'][2]}\n")
- . (empty($record['Logs'][3]) ? '' : "|text_it={$record['Logs'][3]}\n")
- . (empty($record['Logs'][4]) ? '' : "|text_de={$record['Logs'][4]}\n")
- . (empty($record['Logs'][5]) ? '' : "|text_ch={$record['Logs'][5]}\n")
- . (empty($record['Logs'][6]) ? '' : "|text_ru={$record['Logs'][6]}\n")
- . (empty($record['Logs'][7]) ? '' : "|text_kr={$record['Logs'][7]}\n")
- . (empty($record['Logs'][8]) ? '' : "|text_pt={$record['Logs'][8]}\n")
- . (empty($record['Logs'][9]) ? '' : "|text_jp={$record['Logs'][9]}\n")
- . "}}
- ";
- if (!$USE_BOT) {
- echo "$memoraPage\n";
- continue;
- }
- // Here we use a bot to do the job!
- $page = $bridge->fetch_editable($record['LogTitles'][0]);
- if (false === $page) {
- echo "Couldn't fetch page '{$record['LogTitles'][0]}' - skipping.\n";
- continue;
- }
- $page->text = $memoraPage;
- //echo "Here I would edit the page {$record['LogTitles'][0]} to say:\n{$page->text}\n";
- $result = $bridge->edit(
- $page,
- 'Memora: update by bot',
- true, // Is a minor edit
- true, // Is a bot edit
- true, // Do watch the page
- false, // Don't recreate if deleted.
- false, // Don't only create. Editing is allowed/expected.
- true // Do avoid creating nonexistent pages.
- );
- $numBotEdits++;
- echo "$numBotEdits edits made to wiki pages.\n";
- }
- /** Class to extract the string from file data. */
- class StringExtractor {
- public $fileLength;
- public $data;
- /** Constructor.
- * @param $path The file to load in and extract strings from.
- */
- function __construct($path) {
- $this->data = file_get_contents($path);
- $this->fileLength = strlen($this->data);
- }
- /** Read in a string from our data.
- * @param $ptr Pointer to read from, passed by ref and incremented.
- * @param $allowEmpty True if zero-length strings are acceptable.
- * @return StringData Object describing the string parsed.
- */
- function readString(&$ptr, $allowEmpty = false, $debug=false) {
- $tmpPtr = $ptr; // Use a temp pointer until we're sure the string was valid.
- $result = new StringData();
- // Get string length.
- $result->len = $this->readInt($tmpPtr, $debug);
- if ($debug) { echo __LINE__ . ": length = {$result->len}\n"; }
- // Check string length is valid.
- if (($result->len < 0) || ($result->len > (2 ** 16))) {
- $result->err = "Bad length: {$result->len}";
- $result->ret = "1";
- return $result;
- }
- if ((!$allowEmpty) && ($result->len === 0)) {
- $result->err = "Zero length: {$result->len}";
- $result->ret = "2";
- return $result;
- }
- if ($tmpPtr + $result->len > $this->fileLength) {
- $result->err = "String would pass EoF ({$result->len} bytes).";
- $result->ret = "4";
- return $result;
- }
- // Now we're confident in its length, read the string into our struct.
- $result->str = substr($this->data, $tmpPtr, $result->len);
- $tmpPtr += $result->len;
- $result->ptr = $tmpPtr;
- // Check for forbidden characters in the string.
- if (preg_match('/[\0]/', $result->str)) {
- $result->err = "String contains nulls.";
- $result->ret = "5";
- return $result;
- }
- // And then there must be 0-3 nulls to the next 4-byte boundary.
- $result->pad = (4 - ($result->len % 4)) % 4;
- if ($tmpPtr + $result->pad > $this->fileLength) {
- $result->err = "Padding would pass EoF.";
- $result->ret = "6";
- return $result;
- }
- for ($i = $result->pad; $i > 0; $i--) {
- if ("\0" !== $this->data[$tmpPtr+$result->pad-1]) {
- $result->err .= "Bad padding at 0x".dechex($tmpPtr)." $i/{$result->pad}: data 0x".dechex(ord($this->data[$tmpPtr+$result->pad])).".";
- $result->ret = "7";
- return $result;
- }
- }
- // Update the pointer and return our result.
- $tmpPtr += $result->pad;
- $ptr = $tmpPtr;
- return $result;
- }
- /** Read in an integer and move the pointer past it.
- * @param $ptr Pointer to read the int from, passed by ref and incremented.
- * @return The value read, or -1 if no value could be read.
- */
- public function readInt(&$ptr, $debug=false) {
- if ($debug) {echo __LINE__ . ": ptr=$ptr; reading ints, '"
- . dechex(ord($this->data[$ptr+0])) . ' '
- . dechex(ord($this->data[$ptr+1])) . ' '
- . dechex(ord($this->data[$ptr+2])) . ' '
- . dechex(ord($this->data[$ptr+3])) . "'\n"
- ;
- }
- if ($ptr < $this->fileLength - 4) {
- $result =
- (ord($this->data[$ptr+0]) << 0) +
- (ord($this->data[$ptr+1]) << 8) +
- (ord($this->data[$ptr+2]) << 16) +
- (ord($this->data[$ptr+3]) << 24);
- $ptr += 4;
- return $result;
- }
- else {
- $ptr += 4;
- return -1;
- }
- }
- /** Return an error message that describes this string
- * @param StringData $s The struct that contains the errors.
- * @return string Description of the errors.
- */
- public function stringError(StringData $s) {
- return "'{$s->err}' at 0x".dechex($s->ptr).'-0x'.dechex($s->ptr+$s->len+$s->pad)." ({$s->len}+{$s->pad} long: '{$s->str}')";
- }
- }
- /** POD struct to hold data about a string. */
- class StringData {
- public $str='';
- public $err='';
- public $len=0;
- public $ptr=0;
- public $pad=0;
- public $ret=0;
- }
Advertisement
Add Comment
Please, Sign In to add comment