Farrier

Memora extractor

Mar 3rd, 2019 (edited)
154
0
Never
Not a member of Pastebin yet? Sign Up, it unlocks many cool features!
PHP 13.87 KB | None | 0 0
  1. <?php
  2. /**
  3. * for latest version of this file, see https://pastebin.com/edit/rt7F6sMv
  4. *
  5. * This file will extract memora data from `UA_Data/resources.assets` file,
  6. * format it as the wiki requires, and either upload to the wiki, or display it.
  7. *
  8. * To use:
  9. * Place in the UA_Data folder and run it. Or run from anywhere, with the path
  10. * to the resources.assets file as an argument.
  11. *
  12. * If you want it to opload to the wiki, be sure to download/install apibot
  13. * and edit the apibot/logins.php file.
  14. *
  15. * Otherwise, set $USE_BOT=false;, and pipe the output to a file, eg:
  16. *   php logs.php > memora.txt
  17. * The output will be UTF-8. but lacks a BOM so may not be shown that way in
  18. * some editors.
  19. */
  20. ini_set("memory_limit", "-1");
  21. set_time_limit(0);
  22. define(MB_CASE_TITLE, 2); // Because my PHP installation is broken.
  23. // Following requires: http://apibot.zavinagi.org/index.php/Installation
  24. // Set true if you want it to auto-overwrite the pages for you.
  25. $USE_BOT=true;
  26.  
  27. $numLanguages = 11; // EN, ES, FR, IT, DE, CH, RU, KR, PT, JP
  28.  
  29. if ($USE_BOT) {
  30.   if (!is_dir('apibot')) {
  31.     echo "The ./apibot/ folder does not exist.\n";
  32.     echo "You can install it from http://apibot.zavinagi.org\n";
  33.     echo "Alternatively, set \$USE_BOT=false; at the top of this script.\n";
  34.     die(1);
  35.   }
  36.   require_once(dirname(__FILE__).'/apibot/settings.php');
  37.   require_once(dirname(__FILE__).'/apibot/logins.php');
  38.   require_once(dirname(__FILE__).'/apibot/core/core.php');
  39.   require_once(dirname(__FILE__).'/apibot/interfaces/bridge/bridge.php');
  40. }
  41.  
  42. // Default to the obvious file, but allow the filename to be pasted in.
  43. $filename = 'resources.assets';
  44. if (2 === $argc) {
  45.   $filename = $argv[1];
  46. }
  47.  
  48. $keywordSearch = [
  49.   'Abyssal Key' => '/Abyssal Key|Sun Key/i',
  50.   'Baldred' => '/Baldred/i',
  51.   'Cataclysm' => '/Cataclysm/',
  52.   'Cabirus' => '/Cabirus/i',
  53.   'Circle of Portals' => '/Circle of Portals/i',
  54.   'Deep Gap' => '/Deep Gap/i',
  55.   'Disenthralled' => '/Disenthralled/i',
  56.   'Dwarf' => '/Dwarf|Dwarves|Dwarven/i',
  57.   'Elf' => '/\b(Elf|Elves|Elven)\b/i',
  58.   'Executor Rubric' => '/Executor Rubric/',
  59.   'Expedition' => '/Expedition/',
  60.   'Galdwain' => '/Galdwain/i',
  61.   'Gilgamesh' => '/Gilgamesh/i',
  62.   'Goblin' => '/Goblin|Goblinfolk/i',
  63.   'Great Work' => '/Great Work/i',
  64.   'Grue' => '/\bGrues?\b/i',
  65.   'Hivemind' => '/Hivemind/i',
  66.   'Horn of Plenty' => '/Horn of Plenty/i',
  67.   'Ishtass' => '/Ishtass/i',
  68.   'Izanagi-no-Mikoto' => '/Izanagi-no-Mikoto/i',
  69.   'Khosnak' => '/Khosnak/i',
  70.   'Leaf' => '/\bLeaf\b/',
  71.   'Lich' => '/\bLich/i',
  72.   'Lower Dark' => '/Lower Dark/i',
  73.   'Marcaul' => '/Marcaul/i',
  74.   'Mediator' => '/mediator/i',
  75.   'Memora' => '/Memora(?!nd)/i',
  76.   'Mogwawg' => '/Mogwawg/i',
  77.   'Obsidian' => '/Obsidian/',
  78.   'Odin' => '/Odin/',
  79.   'Outcast' => '/Outcast/',
  80.   'Pir-Tama' => '/Pir-Tama|Pir Tama/i',
  81.   'Praxor' => '/Praxor/i',
  82.   'Rotworm' => '/Rot-?worm/i',
  83.   'Saurian' => '/Saurian|Lizardmen|Lizardman/i',
  84.   'Seers' => '/\bSeers?\b/',
  85.   'Slasher of Veils' => '/Slasher of Veils/i',
  86.   'Shambler' => '/Shambler/i',
  87.   'Stygian Abyss' => '/Abyss|Underworld|Avernus|Aornum|Xibalba|Ganzer/i',
  88.   'Sun Key' => '/Sun Key/i',
  89.   'Tinker' => '/\bTinker/i',
  90.   'Titans' => '/\bTitans?\b/i',
  91.   'Typhon' => '/Typhon/i',
  92.   'Undead' => '/Undead/i',
  93.   'Xefreyani' => '/Xefreyani/i',
  94.   'Zeus' => '/\bZeus/i',
  95. ];
  96.  
  97. $extractor = new StringExtractor($filename);
  98.  
  99. // Figure out where we can stop looking. +3 ints for label length, label, 0x0b.
  100. $lastStringStart = $extractor->fileLength - (4 * ($numLanguages + 3));
  101.  
  102. // Loop over the file in chunks of 4 bytes (the file is int-aligned).
  103. for ($i = 0; $i < $lastStringStart; $i += 4) {
  104.   $ptr = $i;         // Temp pointer to read the data in, from $i onwards.
  105.   $descData = null;  // Data for the optional descriptor string.
  106.  
  107.   // Speed optimization: scan rapidly until we hit something that might be valid.
  108.   if (
  109.     ("\0" !== $extractor->data[$i+3]) ||
  110.     ("\0" !== $extractor->data[$i+2]) ||
  111.     (
  112.       ("\0" === $extractor->data[$i+1]) &&
  113.       ("\0" === $extractor->data[$i])
  114.     )
  115.   ) {
  116.     continue;
  117.   }
  118.  
  119.   $labelData = $extractor->readString($ptr); // Data for the initial label string.
  120.  
  121.   // Fail on error.
  122.   if (0 !== $labelData->ret) {
  123.     continue;
  124.   }
  125.  
  126.   // Fail if it's not a valid label.
  127.   if (!preg_match('#^Log\w+/[^/][-\'\w ]+$#', $labelData->str)) {
  128.     continue;
  129.   }
  130.  
  131.   // There are two optional strings here. I don't know what they are for.
  132.   $descData1 = $extractor->readString($ptr, true);
  133.   if (0 !== $labelData->ret) {
  134.     echo '!!0!!' . $labelData->str . ':' . $descData->ret .':' . $extractor->stringError($descData) . " at 0x".dechex($ptr)."\n";
  135.     continue;
  136.   }
  137.   $descData2 = $extractor->readString($ptr, true);
  138.   if (0 !== $labelData->ret) {
  139.     echo '!!1!!' . $labelData->str . ':' . $descData->ret .':' . $extractor->stringError($descData) . " at 0x".dechex($ptr)."\n";
  140.     continue;
  141.   }
  142.  
  143.   // Then one 0x0000000b, little-endian (so 0b 00 00 00).
  144.   if ($extractor->readInt($ptr) !== 0x0b) {
  145.     continue;
  146.   }
  147.  
  148.   // Read in one string per language. Some strings may be zero length.
  149.   $strings = [];
  150.   $labelBits = explode('/', $labelData->str);
  151.   for ($s = 0; $s < $numLanguages; $s++) {
  152.     // Add this string to our array.
  153.     $str = $extractor->readString($ptr, true);
  154.     if (0 !== $labelData->ret) {
  155.       echo '!!2!!' . $labelData->str .':' . $extractor->stringError($labelData) . " at 0x".dechex($ptr)."\n";
  156.       continue 2;
  157.     }
  158.     if ('Logs' === $labelBits[0]) {
  159.       $str->str = preg_replace('/\r?\n/', '<br>', $str->str);
  160.       $str->str = preg_replace('/\t/', ' &nbsp;&nbsp;', $str->str);
  161.     }
  162.  
  163.     $strings[$s] = trim($str->str);
  164.   }
  165.  
  166.   // We have found all strings for this label. Output!.
  167.   $output[$labelBits[1]][$labelBits[0]] = $strings;
  168.  
  169.   // We don't need to scan the strings we already found for the start of other strings!
  170.   // Subtract 4 because we'll add 4 in the loop increment.
  171.   $i = $ptr - 4;
  172. }
  173.  
  174. // If we're gonna be using a bot to output stuff, prepare it here.
  175. if ($USE_BOT) {
  176.   $bridge = new Standalone_Bridge ($logins['underwiki'], $bot_settings);
  177.   $numBotEdits = 0;
  178. }
  179.  
  180.  
  181. // Now we've gathered everything into one big array, we can parse out the data.
  182. foreach ($output as $name => $record) {
  183.   // Record contains two parts, 'logs' and 'logtitles'
  184.   $titleBits = preg_split('/\r?\n| - /', $record['LogTitles'][0], 2);
  185.   switch(trim($titleBits[0])) {
  186.     case 'Xefreyani of The Deep Elves':
  187.       $author = 'Xefreyani';
  188.       $icon = 'Elves_icon.png';
  189.     break;
  190.     case 'Khosnak, Mediator to The Shamblers':
  191.       $author = 'Khosnak';
  192.       $icon = 'Shamblers_icon.png';
  193.     break;
  194.     case 'Executor Rubric of The Expedition':
  195.       $author = 'Executor Rubric';
  196.       $icon = 'Expedition_icon.png';
  197.     break;
  198.     case 'Cabirus':
  199.       $author = 'Cabirus';
  200.       $icon = 'Cabirus_icon.png';
  201.     break;
  202.     default:
  203.       $author = 'Anon';
  204.       $icon = '';
  205.     break;
  206.   }
  207.   // Strip the author names off the beginnings of all titles, and convert to title case.
  208.   foreach ($record['LogTitles'] as $id => $title) {
  209.     if (!empty($title)) {
  210.       $bits = preg_split('/\r?\n| - /', $title, 2);
  211.       if (2 == count($bits)) {
  212.         $record['LogTitles'][$id] = trim($bits[1]);
  213.       }
  214.       $record['LogTitles'][$id] = mb_convert_case($record['LogTitles'][$id], MB_CASE_TITLE, "UTF-8");
  215.     }
  216.   }
  217.  
  218.   $keywords = [];
  219.   foreach ($keywordSearch as $keyword => $regex) {
  220.     if (preg_match($regex, $record['Logs'][0])) {
  221.       $keywords []= $keyword;
  222.     }
  223.   }
  224.   $memoraPage = "
  225. {{Infobox Memora
  226. |author=$author
  227. |icon=$icon
  228. |scene=?
  229. |location=?
  230. |keywords=" . implode(',', $keywords) . "\n"
  231. . (empty($record['LogTitles'][0]) ? '' : "|title={$record['LogTitles'][0]}\n")
  232. . (empty($record['LogTitles'][1]) ? '' : "|title_es={$record['LogTitles'][1]}\n")
  233. . (empty($record['LogTitles'][2]) ? '' : "|title_fr={$record['LogTitles'][2]}\n")
  234. . (empty($record['LogTitles'][3]) ? '' : "|title_it={$record['LogTitles'][3]}\n")
  235. . (empty($record['LogTitles'][4]) ? '' : "|title_de={$record['LogTitles'][4]}\n")
  236. . (empty($record['LogTitles'][5]) ? '' : "|title_ch={$record['LogTitles'][5]}\n")
  237. . (empty($record['LogTitles'][6]) ? '' : "|title_ru={$record['LogTitles'][6]}\n")
  238. . (empty($record['LogTitles'][7]) ? '' : "|title_kr={$record['LogTitles'][7]}\n")
  239. . (empty($record['LogTitles'][8]) ? '' : "|title_pt={$record['LogTitles'][8]}\n")
  240. . (empty($record['LogTitles'][9]) ? '' : "|title_jp={$record['LogTitles'][9]}\n")
  241. . (empty($record['Logs'][0]) ? '' : "|text={$record['Logs'][0]}\n")
  242. . (empty($record['Logs'][1]) ? '' : "|text_es={$record['Logs'][1]}\n")
  243. . (empty($record['Logs'][2]) ? '' : "|text_fr={$record['Logs'][2]}\n")
  244. . (empty($record['Logs'][3]) ? '' : "|text_it={$record['Logs'][3]}\n")
  245. . (empty($record['Logs'][4]) ? '' : "|text_de={$record['Logs'][4]}\n")
  246. . (empty($record['Logs'][5]) ? '' : "|text_ch={$record['Logs'][5]}\n")
  247. . (empty($record['Logs'][6]) ? '' : "|text_ru={$record['Logs'][6]}\n")
  248. . (empty($record['Logs'][7]) ? '' : "|text_kr={$record['Logs'][7]}\n")
  249. . (empty($record['Logs'][8]) ? '' : "|text_pt={$record['Logs'][8]}\n")
  250. . (empty($record['Logs'][9]) ? '' : "|text_jp={$record['Logs'][9]}\n")
  251. . "}}
  252. ";
  253.  
  254.   if (!$USE_BOT) {
  255.     echo "$memoraPage\n";
  256.     continue;
  257.   }
  258.  
  259.   // Here we use a bot to do the job!
  260.   $page = $bridge->fetch_editable($record['LogTitles'][0]);
  261.   if (false === $page) {
  262.     echo "Couldn't fetch page '{$record['LogTitles'][0]}' - skipping.\n";
  263.     continue;
  264.   }
  265.  
  266.   $page->text = $memoraPage;
  267.  
  268.   //echo "Here I would edit the page {$record['LogTitles'][0]} to say:\n{$page->text}\n";
  269.  
  270.   $result = $bridge->edit(
  271.     $page,
  272.     'Memora: update by bot',
  273.     true,  // Is a minor edit
  274.     true,  // Is a bot edit
  275.     true,  // Do watch the page
  276.     false, // Don't recreate if deleted.
  277.     false, // Don't only create. Editing is allowed/expected.
  278.     true   // Do avoid creating nonexistent pages.
  279.   );
  280.   $numBotEdits++;
  281.   echo "$numBotEdits edits made to wiki pages.\n";
  282. }
  283.  
  284.  
  285. /** Class to extract the string from file data. */
  286. class StringExtractor {
  287.   public $fileLength;
  288.   public $data;
  289.  
  290.   /** Constructor.
  291.   * @param $path The file to load in and extract strings from.
  292.   */
  293.   function __construct($path) {
  294.     $this->data = file_get_contents($path);
  295.     $this->fileLength = strlen($this->data);
  296.   }
  297.  
  298.   /** Read in a string from our data.
  299.   * @param $ptr         Pointer to read from, passed by ref and incremented.
  300.   * @param $allowEmpty  True if zero-length strings are acceptable.
  301.   * @return StringData  Object describing the string parsed.
  302.   */
  303.   function readString(&$ptr, $allowEmpty = false, $debug=false) {
  304.     $tmpPtr = $ptr; // Use a temp pointer until we're sure the string was valid.
  305.     $result = new StringData();
  306.  
  307.     // Get string length.
  308.     $result->len = $this->readInt($tmpPtr, $debug);
  309. if ($debug) { echo __LINE__ . ": length = {$result->len}\n"; }
  310.     // Check string length is valid.
  311.     if (($result->len < 0) || ($result->len > (2 ** 16))) {
  312.       $result->err = "Bad length: {$result->len}";
  313.       $result->ret = "1";
  314.       return $result;
  315.     }
  316.     if ((!$allowEmpty) && ($result->len === 0)) {
  317.       $result->err = "Zero length: {$result->len}";
  318.       $result->ret = "2";
  319.       return $result;
  320.     }
  321.     if ($tmpPtr + $result->len > $this->fileLength) {
  322.       $result->err = "String would pass EoF ({$result->len} bytes).";
  323.       $result->ret = "4";
  324.       return $result;
  325.     }
  326.  
  327.     // Now we're confident in its length, read the string into our struct.
  328.     $result->str = substr($this->data, $tmpPtr, $result->len);
  329.     $tmpPtr += $result->len;
  330.     $result->ptr = $tmpPtr;
  331.  
  332.     // Check for forbidden characters in the string.
  333.     if (preg_match('/[\0]/', $result->str)) {
  334.       $result->err = "String contains nulls.";
  335.       $result->ret = "5";
  336.       return $result;
  337.     }
  338.  
  339.     // And then there must be 0-3 nulls to the next 4-byte boundary.
  340.     $result->pad = (4 - ($result->len % 4)) % 4;
  341.     if ($tmpPtr + $result->pad > $this->fileLength) {
  342.       $result->err = "Padding would pass EoF.";
  343.       $result->ret = "6";
  344.       return $result;
  345.     }
  346.    
  347.     for ($i = $result->pad; $i > 0; $i--) {
  348.       if ("\0" !== $this->data[$tmpPtr+$result->pad-1]) {
  349.         $result->err .= "Bad padding at 0x".dechex($tmpPtr)." $i/{$result->pad}: data 0x".dechex(ord($this->data[$tmpPtr+$result->pad])).".";
  350.         $result->ret = "7";
  351.         return $result;
  352.       }
  353.     }
  354.    
  355.     // Update the pointer and return our result.
  356.     $tmpPtr += $result->pad;
  357.     $ptr = $tmpPtr;
  358.     return $result;
  359.   }
  360.  
  361.   /** Read in an integer and move the pointer past it.
  362.   * @param $ptr Pointer to read the int from, passed by ref and incremented.
  363.   * @return The value read, or -1 if no value could be read.
  364.   */
  365.   public function readInt(&$ptr, $debug=false) {
  366. if ($debug) {echo __LINE__ . ": ptr=$ptr; reading ints, '"
  367.   . dechex(ord($this->data[$ptr+0])) . ' '
  368.   . dechex(ord($this->data[$ptr+1])) . ' '
  369.   . dechex(ord($this->data[$ptr+2])) . ' '
  370.   . dechex(ord($this->data[$ptr+3])) . "'\n"
  371. ;
  372. }
  373.     if ($ptr < $this->fileLength - 4) {
  374.       $result =
  375.         (ord($this->data[$ptr+0]) <<  0) +
  376.         (ord($this->data[$ptr+1]) <<  8) +
  377.         (ord($this->data[$ptr+2]) << 16) +
  378.         (ord($this->data[$ptr+3]) << 24);
  379.       $ptr += 4;
  380.       return $result;
  381.     }
  382.     else {
  383.       $ptr += 4;
  384.       return -1;
  385.     }
  386.   }
  387.  
  388.   /** Return an error message that describes this string
  389.   * @param StringData $s  The struct that contains the errors.
  390.   * @return string        Description of the errors.
  391.   */
  392.   public function stringError(StringData $s) {
  393.     return "'{$s->err}' at 0x".dechex($s->ptr).'-0x'.dechex($s->ptr+$s->len+$s->pad)." ({$s->len}+{$s->pad} long: '{$s->str}')";
  394.   }
  395. }
  396.  
  397. /** POD struct to hold data about a string. */
  398. class StringData {
  399.   public $str='';
  400.   public $err='';
  401.   public $len=0;
  402.   public $ptr=0;
  403.   public $pad=0;
  404.   public $ret=0;
  405. }
Advertisement
Add Comment
Please, Sign In to add comment