ویکی‌نبشته fawikisource https://fa.wikisource.org/wiki/%D8%B5%D9%81%D8%AD%D9%87%D9%94_%D8%A7%D8%B5%D9%84%DB%8C MediaWiki 1.47.0-wmf.20 first-letter مدیا ویژه بحث کاربر بحث کاربر ویکی‌نبشته بحث ویکی‌نبشته پرونده بحث پرونده مدیاویکی بحث مدیاویکی الگو بحث الگو راهنما بحث راهنما رده بحث رده درگاه بحث درگاه پدیدآورنده بحث پدیدآورنده برگه گفتگوی برگه فهرست گفتگوی فهرست TimedText TimedText talk پودمان بحث پودمان Event Event talk کاربر:Hanooz/common.js 2 63731 299780 299731 2026-09-20T19:46:14Z Hanooz 17889 299780 javascript text/javascript window.charinsertCustom = { "کاربر": '{{em}} {{gap}} {{sc|+}} {{sp|+}} {{xl|+}} {{c|+}} {{c|{{xl|+}}}} {{c|{{xl|{{sp|+}}}}}} {{dhr|2}}' }; //<nowiki> /* Cat-a-lot - changes category of multiple files */ mw.loader.using(['jquery.ui', 'mediawiki.util'], function(){ mw.loader.load('//commons.wikimedia.org/w/load.php?modules=ext.gadget.Cat-a-lot'); }); ////////// Cat-a-lot user preferences ////////// window.catALotPrefs = {"watchlist":"preferences","minor":true,"editpages":true,"docleanup":false,"subcatcount":10}; ////////////////////////////////////catALotEnd// //</nowiki> mw.loader.load('//meta.wikimedia.org/w/index.php?title=User:Krinkle/Tools/WhatLeavesHere.js&action=raw&ctype=text/javascript'); mw.loader.load('//meta.wikimedia.org/w/index.php?title=User:Indic-TechCom/Script/massMover.js&action=raw&ctype=text/javascript'); mw.loader.load('//mediawiki.org/w/index.php?title=MediaWiki:Gadget-UTCLiveClock.js&action=raw&ctype=text/javascript&smaxage=21600&maxage=86400'); mw.loader.load('//mediawiki.org/w/index.php?title=User:PerfektesChaos/js/resultListSort/r.js&action=raw&bcache=1&ctype=text/javascript'); mw.loader.load('//fa.wikipedia.org/w/index.php?title=مدیاویکی:Gadget-quickLinker.js&action=raw&ctype=text/javascript'); mw.loader.load('//fa.wikipedia.org/w/index.php?title=مدیاویکی:Gadget-Extra-Editbuttons-sisters.js&action=raw&ctype=text/javascript'); mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/Gadget-charinsert-core.js&action=raw&ctype=text/javascript'); mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/page carousel.js&action=raw&ctype=text/javascript'); mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/NopInserter.js&action=raw&ctype=text/javascript'); mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/redirectmaker.js&action=raw&ctype=text/javascript'); mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/RunningHeader.js&action=raw&ctype=text/javascript'); mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/test.js&action=raw&ctype=text/javascript'); mw.loader.load('//en.wikisource.org/w/index.php?title=MediaWiki:Gadget-TemplatePreloader.js&action=raw&ctype=text/javascript'); mw.loader.load('//en.wikisource.org/w/index.php?title=MediaWiki:Gadget-Without text.js&action=raw&ctype=text/javascript'); mw.loader.load('//en.wikisource.org/w/index.php?title=MediaWiki:Gadget-transclusion-check.js&action=raw&ctype=text/javascript'); mw.loader.load('//en.wikisource.org/w/index.php?title=MediaWiki:Gadget-HotCat.js&action=raw&ctype=text/javascript'); mw.loader.load('//en.wikisource.org/w/index.php?title=MediaWiki:Gadget-Easy LST.js&action=raw&ctype=text/javascript'); mw.loader.load('//en.wikisource.org/w/index.php?title=MediaWiki:Gadget-Fill Index.js&action=raw&ctype=text/javascript'); mw.loader.load('//en.wikisource.org/w/index.php?title=User:Inductiveload/cleanup.js&action=raw&ctype=text/javascript'); mw.loader.load('//en.wikisource.org/w/index.php?title=User:Inductiveload/quick_access.js&action=raw&ctype=text/javascript'); mw.loader.load('//en.wikisource.org/w/index.php?title=User:Xover/cleanup.js&action=raw&ctype=text/javascript'); mw.loader.load('//en.wikisource.org/w/index.php?title=User:Xover/unwrap.js&action=raw&ctype=text/javascript'); mw.loader.load('//en.wikisource.org/w/index.php?title=User:Xover/copySource.js&action=raw&ctype=text/javascript'); mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Nightdevil/indexFiller.js&action=raw&ctype=text/javascript'); 5n4arv3iz62zwryzww1462l41hwl09x 299781 299780 2026-09-20T19:49:23Z Hanooz 17889 299781 javascript text/javascript window.charinsertCustom = { "کاربر": '{{em}} {{gap}} {{sc|+}} {{sp|+}} {{xl|+}} {{c|+}} {{c|{{xl|+}}}} {{c|{{xl|{{sp|+}}}}}} {{dhr|2}}' }; //<nowiki> /* Cat-a-lot - changes category of multiple files */ mw.loader.using(['jquery.ui', 'mediawiki.util'], function(){ mw.loader.load('//commons.wikimedia.org/w/load.php?modules=ext.gadget.Cat-a-lot'); }); ////////// Cat-a-lot user preferences ////////// window.catALotPrefs = {"watchlist":"preferences","minor":true,"editpages":true,"docleanup":false,"subcatcount":10}; ////////////////////////////////////catALotEnd// //</nowiki> mw.loader.load('//meta.wikimedia.org/w/index.php?title=User:Krinkle/Tools/WhatLeavesHere.js&action=raw&ctype=text/javascript'); mw.loader.load('//meta.wikimedia.org/w/index.php?title=User:Indic-TechCom/Script/massMover.js&action=raw&ctype=text/javascript'); mw.loader.load('//mediawiki.org/w/index.php?title=MediaWiki:Gadget-UTCLiveClock.js&action=raw&ctype=text/javascript&smaxage=21600&maxage=86400'); mw.loader.load('//mediawiki.org/w/index.php?title=User:PerfektesChaos/js/resultListSort/r.js&action=raw&bcache=1&ctype=text/javascript'); mw.loader.load('//fa.wikipedia.org/w/index.php?title=مدیاویکی:Gadget-quickLinker.js&action=raw&ctype=text/javascript'); mw.loader.load('//fa.wikipedia.org/w/index.php?title=مدیاویکی:Gadget-Extra-Editbuttons-sisters.js&action=raw&ctype=text/javascript'); mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/Gadget-charinsert-core.js&action=raw&ctype=text/javascript'); mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/page carousel.js&action=raw&ctype=text/javascript'); mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/NopInserter.js&action=raw&ctype=text/javascript'); mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/redirectmaker.js&action=raw&ctype=text/javascript'); mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/RunningHeader.js&action=raw&ctype=text/javascript'); mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/test.js&action=raw&ctype=text/javascript'); mw.loader.load('//en.wikisource.org/w/index.php?title=MediaWiki:Gadget-TemplatePreloader.js&action=raw&ctype=text/javascript'); mw.loader.load('//en.wikisource.org/w/index.php?title=MediaWiki:Gadget-Without text.js&action=raw&ctype=text/javascript'); mw.loader.load('//en.wikisource.org/w/index.php?title=MediaWiki:Gadget-HotCat.js&action=raw&ctype=text/javascript'); mw.loader.load('//en.wikisource.org/w/index.php?title=MediaWiki:Gadget-Easy LST.js&action=raw&ctype=text/javascript'); mw.loader.load('//en.wikisource.org/w/index.php?title=MediaWiki:Gadget-Fill Index.js&action=raw&ctype=text/javascript'); mw.loader.load('//en.wikisource.org/w/index.php?title=User:Xover/unwrap.js&action=raw&ctype=text/javascript'); mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Nightdevil/indexFiller.js&action=raw&ctype=text/javascript'); rqal0ndesu6lolk5knvgx5p0zizpmce کاربر:Hanooz/test.js 2 86084 299769 299758 2026-09-20T14:41:39Z Hanooz 17889 299769 javascript text/javascript /* This page defines a TemplateScript library for fa.wikisource.org (ویکی‌نبشتهٔ فارسی). It's not meant to be referenced directly - install it the same way as the upstream English version described at [[Wikisource:TemplateScript]]. This is a Persian Wikisource adaptation of Pathoschild's proofreading and typography libraries, pared down to a focused four-action toolset. It draws on the fa.wikipedia/ fa.wikisource "Persian text style improvement tools" gadget (Gadget-Extra-Editbuttons- persianwikitools.js) for several of the character/orthography helpers, since that's the community's own tested implementation rather than something invented for this file. */ /* global $, pathoschild */ /** * TemplateScript adds configurable templates and scripts to the sidebar, and adds an example regex editor. * @see https://meta.wikimedia.org/wiki/TemplateScript * * The upstream proofreading library's @update-token has been removed on purpose: Wikisource's * maintenance bot uses that token to overwrite local copies with the vanilla English script, * which would silently wipe out every change made here for fa.wikisource. See [[Wikisource:Tools * and scripts]] for how the update-bot category works. */ // <nowiki> $.ajax('//tools-static.wmflabs.org/meta/scripts/pathoschild.templatescript.js', { dataType:'script', cache:true }).then(function() { /********* ** Define library *********/ pathoschild.TemplateScript.library.define({ key: 'wikisource.fa.tools', name: 'ابزارهای ویکی‌نبشتهٔ فارسی', url: '//fa.wikisource.org/wiki/راهنما:نمونه‌خوانی', description: 'مجموعه‌ای از ابزارها برای <a href="/wiki/راهنما:نمونه‌خوانی">نمونه‌خوانی آثار در فضای نام <tt>برگه:</tt></a>: پاک‌سازی OCR (با اصلاح نویسه‌ها، کشیده، ارقام، جای اعراب، نیم‌فاصله، نویسه‌های نامرئی بی‌اثر، و نشانه‌گذاری) و اصلاح نیم‌فاصلهٔ افعال.', categories: [ { name: 'ابزارهای برگه', scripts: [ { key: 'cleanup-ocr', name: 'پاک‌سازی OCR', script: function(editor) { pageCleanup(editor); }, forNamespaces: 'page' }, { key: 'smart-zwnj', name: 'اصلاح نیم‌فاصلهٔ افعال', script: function(editor) { smartZwnj(editor); }, forNamespaces: 'page' } ] } ] }); /********* ** Private methods *********/ // Digit tables used to convert between Latin, Persian and Arabic-Indic numerals. var _persianDigits = '۰۱۲۳۴۵۶۷۸۹'; var _arabicIndicDigits = '٠١٢٣٤٥٦٧٨٩'; var _latinDigits = '0123456789'; // Character-class tables used by the Persian-specific cleanup helpers below. Ported from the // fa.wikipedia/fa.wikisource "Persian text style improvement tools" gadget (Gadget-Extra- // Editbuttons-persianwikitools.js) - the community's own tested implementation of this // normalisation, rather than something invented for this file. var _hamza = '\u0654'; var _persianVowels = '\u064B-\u0650\u0652\u0670'; // non-standard letterforms (Arabic ي/ك, Urdu, Pashto, Uyghur presentation forms, etc.) that // should still count as "a Persian-ish letter" for context-matching, even before // _toStandardPersianCharacters() below has normalised them var _similarPersianCharacters = '\u0643\uFB91\uFB90\uFB8F\uFB8E\uFEDC\uFEDB\uFEDA\uFED9\u0649\uFEEF\u064A\u06C1\u06D5\u06BE\uFEF0-\uFEF4'; var _persianCharacters = '\u0621-\u0655\u067E\u0686\u0698\u06AF\u06A9\u0643\u06AA\uFED9\uFEDA\u06CC\uFEF1\uFEF2' + _similarPersianCharacters; /** * Convert any mix of Latin or Arabic-Indic digits in a string to Persian digits, for display in wikitext. * @param {string} text The text to convert. */ var _digitsToPersian = function(text) { var result = ''; for (var i = 0; i < text.length; i++) { var ch = text.charAt(i); var idx = _latinDigits.indexOf(ch); if (idx === -1) idx = _arabicIndicDigits.indexOf(ch); result += (idx !== -1) ? _persianDigits.charAt(idx) : ch; } return result; }; /** * OCR/PDF presentation-form and script-variant glyphs that extraction sometimes leaves as * literal characters instead of the standard Persian letter - a common problem specifically on * PDF-sourced works, where each glyph can get saved as its own shaped codepoint. Keys are the * standard letter (or letter + a real ZWNJ, for the two glyphs that visually look joined); * values are a character class of the presentation-form/variant glyphs that should fold into it. */ var _persianGlyphs = { '\u200cه': 'ﻫ', 'ی\u200c': 'ﻰﻲ', 'أ': 'ﺄﺃﺃ', 'آ': 'ﺁﺁﺂ', 'إ': 'ﺇﺈﺇ', 'ا': 'ﺍﺎ', 'ب': 'ﺏﺐﺑﺒ', 'پ': 'ﭖﭗﭘﭙ', 'ت': 'ﺕﺖﺗﺘ', 'ث': 'ﺙﺚﺛﺜ', 'ج': 'ﺝﺞﺟﺠ', 'چ': 'ﭺﭻﭼﭽ', 'ح': 'ﺡﺢﺣﺤ', 'خ': 'ﺥﺦﺧﺨ', 'د': 'ﺩﺪ', 'ذ': 'ﺫﺬ', 'ر': 'ﺭﺮ', 'ز': 'ﺯﺰ', 'ژ': 'ﮊﮋ', 'س': 'ﺱﺲﺳﺴ', 'ش': 'ﺵﺶﺷﺸ', 'ص': 'ﺹﺺﺻﺼ', 'ض': 'ﺽﺾﺿﻀ', 'ط': 'ﻁﻂﻃﻄ', 'ظ': 'ﻅﻆﻇﻈ', 'ع': 'ﻉﻊﻋﻌ', 'غ': 'ﻍﻎﻏﻐ', 'ف': 'ﻑﻒﻓﻔ', 'ق': 'ﻕﻖﻗﻘ', 'ک': 'ﮎﮏﮐﮑﻙﻚﻛﻜ', 'گ': 'ﮒﮓﮔﮕ', 'ل': 'ﻝﻞﻟﻠ', 'م': 'ﻡﻢﻣﻤ', 'ن': 'ﻥﻦﻧﻨ', 'ه': 'ﻩﻪﻫﻬ', 'هٔ': 'ﮤﮥ', 'و': 'ﻭﻮ', 'ؤ': 'ﺅﺅﺆ', 'ی': 'ﯼﯽﯾﯿﻯﻰﻱﻲﻳﻴ', 'ئ': 'ﺉﺊﺋﺌ', 'لا': 'ﻻﻼ', 'لإ': 'ﻹﻺ', 'لأ': 'ﻸﻷ', 'لآ': 'ﻵﻶ' }; /** * Remove stray/duplicate ZWNJ-family characters and OCR artifacts that stand in for a ZWNJ, * without trying to guess where a NEW ZWNJ should be inserted between words - that needs the * verb word-lists in smartZwnj() below, since blindly inserting one is what risks corrupting * unrelated text (see that function's comment for why). * @param {string} text The text to clean. */ var _cleanupZwnjArtifacts = function(text) { return text // a ZWNJ (or stray zero-width/BOM/bidi-mark character) with nothing before or after it // to join or not-join has no possible effect on the rendered output - most commonly a // ZWNJ that ended up with a plain space next to it instead of a letter .replace(/^[\u200B-\u200D\uFEFF]+/, '') .replace(/[\u200B-\u200D\uFEFF]+$/, '') // a stray LRM/RLM between two Persian letters was almost always meant to be a ZWNJ .replace(new RegExp('([' + _persianCharacters + '] *)[\u200F\u200E]+( *[' + _persianCharacters + '])', 'g'), '$1\u200c$2') // collapse repeated invisible-joiner/mark characters down to one .replace(/([\u200B-\u200D\uFEFF\u200E\u200F]){2,}/g, '$1') // ¬ between two Persian letters is almost always a misrecognised ZWNJ, not a real character .replace(new RegExp('([' + _persianCharacters + '])¬(?=[' + _persianCharacters + '])', 'g'), '$1\u200c') // a ZWNJ has no business after a digit or most punctuation/brackets, or next to a Latin word .replace(/([۰-۹0-9إأةؤورزژاآدذ،؛,:«»\\\/@#$٪×*()ـ\-=|ء])\u200c/g, '$1') .replace(/[\u200B-\u200D\uFEFF]([\w])/g, '$1') .replace(/([\w])[\u200B-\u200D\uFEFF]/g, '$1') .replace(new RegExp('[\\u200B-\\u200D\\uFEFF]([' + _persianVowels + _arabicIndicDigits + _persianDigits + _latinDigits + _hamza + '])', 'g'), '$1') .replace(new RegExp('([' + _arabicIndicDigits + '])[\\u200B-\\u200D\\uFEFF]', 'g'), '$1') .replace(/[\u200B\u200C\uFEFF]([ء\n\s\[\].،«»:()؛؟?;$!@\-=+\\|])/g, '$1') .replace(/([\n\s\[.،«»:()؛؟?;$!@\-=+\\|])[\u200B-\u200D\uFEFF]/g, '$1') .replace(/[\u200B-\u200D\uFEFF](\]\][\s\n])/g, '$1') .replace(/([\n\s]\[\[)[\u200B-\u200D\uFEFF]/g, '$1'); }; /** * Fold OCR/PDF presentation-form glyphs and non-Persian script variants (Arabic ي/ك, Urdu, * Pashto, Uyghur, Kurdish) back to standard Persian characters, and canonicalise the hamza- * bearing forms per ISIRI 6219 (Iran's national character-encoding standard). * @param {string} text The text to clean. */ var _toStandardPersianCharacters = function(text) { for (var standard in _persianGlyphs) { if (_persianGlyphs.hasOwnProperty(standard)) { text = text.replace(new RegExp('[' + _persianGlyphs[standard] + ']', 'g'), standard); } } return _cleanupZwnjArtifacts(text) // needed because of the two ZWNJ-bearing keys above .replace(/ك/g, 'ک') // Arabic .replace(/ڪ/g, 'ک') // Urdu .replace(/ﻙ/g, 'ک') // Pushtu .replace(/ﻚ/g, 'ک') // Uyghur .replace(/ي/g, 'ی') // Arabic .replace(/ى/g, 'ی') // Urdu .replace(/ے/g, 'ی') // Urdu .replace(/ۍ/g, 'ی') // Pushtu .replace(/ې/g, 'ی') // Uyghur .replace(/ہ/g, 'ه') // Urdu .replace(/ە/g, 'ه\u200c') // Kurdish .replace(/ھ/g, 'ه') // Kurdish .replace(/أ/g, 'أ') // canonical alef+hamza form per ISIRI 6219 .replace(/آ/g, 'آ'); }; /** * Strip tatweel/kashide (ـ, U+0640) characters from a string. Tatweel is a pure justification * stretch-mark used in decorative Persian typesetting (common on title pages); OCR frequently * captures it as literal characters, but it has no place in a wikilink target or in wikitext. * @param {string} text The text to clean. */ var _stripTatweel = function(text) { return text.replace(/\u0640+/g, ''); }; /********* ** Script methods *********/ /** * Clean up OCR errors in the text, and push <noinclude> content at the top * & bottom of the page into the header & footer boxes respectively. * @param {object} editor The script helpers for the page. */ var pageCleanup = function(editor) { // Strip characters that have no effect on the rendered output at all, before anything // else runs: control characters, BOM and soft hyphen have no legitimate place in body // text; \r is a Windows line-ending artifact once \n is already there; a run of exotic // Unicode space characters (NBSP, thin space, ideographic space, etc.) right before a // line break just disappears, a tab/NBSP right after one disappears too (plain space is // deliberately left alone there, since a leading space triggers MediaWiki's <pre> // formatting), and any exotic space left in running text becomes a plain space. Ported // from the fa.wikipedia/fa.wikisource gadget's applyOrthography(), and cross-checked // against Virastar (github.com/aziz/virastar and its ports, a widely used standalone // Persian text cleaner) - both do this same normalisation as a first step. var text = editor.get() .replace(/\r/g, '') .replace(/[\x00-\x08\x0B\x0C\x0E-\x1F\x7F-\x9F\uFEFF\u00AD\u0085]+/g, '') .replace(/[ \xA0\u1680\u180E\u2000-\u200A\u202F\u205F\u3000]+\n/g, '\n') .replace(/\n[\t\u00A0]+/g, '\n') .replace(/[\xA0\u1680\u180E\u2000-\u200A\u202F\u205F\u3000]/g, ' '); // Normalise the raw text next: fold OCR/PDF presentation-form glyphs and non-Persian // letter variants to standard Persian characters (this also mops up stray ZWNJ-family // characters left over from OCR - see _cleanupZwnjArtifacts, including a ZWNJ that's now // got a plain space next to it instead of a letter, which can no longer do anything), // strip decorative tatweel, convert Latin/Arabic-Indic digits to Persian, and fix vowel- // mark (harakat) placement - a mark separated from its letter by a stray space moves back // next to the letter, and duplicate marks on the same letter collapse to one. The digit // conversion runs unconditionally on the whole page - if a work legitimately needs // Western digits somewhere (an ISBN, a citation in another language), those get // converted too; there's no separate opt-in step for that anymore. text = _toStandardPersianCharacters(text); text = _stripTatweel(text); text = _digitsToPersian(text); text = text.replace(new RegExp('([' + _persianCharacters + _persianVowels + _hamza + '])(\\s)([' + _persianVowels + _hamza + '])', 'g'), '$1$3$2'); text = text.replace(new RegExp('([' + _persianVowels + _hamza + ']){2,}', 'g'), '$1'); editor.set(text); // Strip leading lines that are pure page-number "running header" junk - the digits-only // case of the en.wikisource WsCleanup gadget's running-header patterns. Dropped that // gadget's other pattern (digits plus an ALL-CAPS title fragment) since Persian has no // letter case to signal "this is header text, not body text" the way English capitals // do. Caveat carried over from the source: a work with numbered stanzas/verses starting // right at the top of a page could have its number line mistaken for header junk here - // worth a glance after running this on poetry. (function() { var lines = editor.get().split(/\r?\n/); var pageNumberOnly = /^[\s0-9۰-۹٠-٩.,\-–—]+$/; var start = 0; for (var li = 0; li < lines.length; li++) { if (lines[li].trim().length === 0 || (pageNumberOnly.test(lines[li]) && lines[li].trim().length > 0)) { start++; continue; } break; } if (start > 0) { editor.set(lines.slice(start).join('\n')); } })(); // push <noinclude> content at the top & bottom into the header & footer if (editor.get().match(/^<noinclude\>/)) { var text = editor.get(); var e = text.indexOf("</noinclude>"); $('#wpHeaderTextbox').val(function(i, val) { return $.trim(val + "\n" + text.substr(11, e-11).replace(/^\s+|\s+$/g, '')); }); editor.set(text.substr(e+12)); } if (editor.get().match(/<\/noinclude\>$/)) { var text = editor.get(); var s = text.lastIndexOf("<noinclude>"); $('#wpFooterTextbox').val(function(i, val) { return $.trim(text.substr(s+11, text.length-s-11-12).replace(/^\s+|\s+$/g, '') + "\n" + val); }); editor.set(text.substr(0, s)); } // clean up text editor // remove trailing spaces at the end of each line .replace(/ +\n/g, '\n') // remove trailing whitespace preceding a hard line break .replace(/ +<br *\/?>/g, '<br />') // remove trailing whitespace and numerals at the end of page text - matches // Latin, Persian and Arabic-Indic digits, since page-number artifacts at the // bottom of a scanned fa.wikisource page can OCR as any of the three .replace(/[\s0-9۰-۹٠-٩]+$/g, '') // remove trailing spaces at the end of refs .replace(/ +<\/ref>/g, '</ref>') // remove trailing spaces at the end of template calls .replace(/ +}}/g, '}}') // convert double-hyphen to mdash (avoiding breaking HTML comment syntax) .replace(/([^\!])--([^>])/g, '$1—$2') // remove spacing around mdash, but only if it has spaces on both sides .replace(/ +— +/g, '—') // "Digitized by Google" scan-watermark text and stray OCR of the word "Google" - // the watermark itself is always in English regardless of the scanned work's // language, so this is just as relevant to a Google Books-sourced Persian scan .replace(/\s?D[il]g[il]t[il][sz][eco]d\s+by[^\n]*\s+([6G][Oo0]{2}g[lI]e)?/g, '') .replace(/\bG[oO0]{2}gle\b/g, '') // highly suspicious OCR noise characters .replace(/[■•]/g, '') // a line containing nothing but a stray punctuation mark is likely scan noise .replace(/^[.,^،؛]$/gm, '') // underscores (often OCR of underlined text or a form field) aren't meaningful in wikitext .replace(/_/g, ' ') // number ranges get an en dash, not a hyphen, per fa.wikisource's وپ:خط تیره .replace(/([۰-۹]+)[ \t]?-[ \t]?(?=[۰-۹])/g, '$1–') // <section begin=x/> needs to be <section begin="x"/> to parse .replace(/<section (begin|end)=(\w[^/]+)\/>/g, '<section $1="$2"/>') // a <ref> shouldn't have a space glued before it .replace(/ <ref/g, '<ref') // {{hws}}/{{hwe}} shorthand for a word split across a page boundary, expanded to the // full template names (same English names used across Wikisource language versions, // fa.wikisource included, for this kind of utility template) .replace(/{{hws\|/g, '{{hyphenated word start|') .replace(/{{hwe\|/g, '{{hyphenated word end|') // {{c|...}} shorthand for {{center|...}}, if used .replace(/{{c\|/g, '{{center|'); // clean up pages if they don't have <poem> if (!editor.contains('<poem>') && !editor.contains('{{' + 'ppoem')) { editor // remove single line breaks; preserve multiple. // but not if there's a tag, template or table syntax either side of the line break .replace(/([^>}\|\n])\n([^:#\*<{\|\n])/g, '$1 $2') // collapse sequences of spaces into a single space .replace(/ +/g, ' '); } // more page cleanup editor // dump spurious hard breaks at the end of paragraphs .replace(/<br *\/?>\n\n/g, '\n\n') // remove unwanted spaces before punctuation marks - both Latin (;:?!,.)) and // Persian (،؛؟) forms, since either can turn up depending on the OCR engine .replace(/ ([;:\?!,.)،؛؟])/g, '$1') // and no space after an opening parenthesis .replace(/\( +/g, '(') // convert ASCII look-alikes to proper Persian punctuation when they follow Persian // text. Properly printed Persian sources always used ؟/؛/، - an ASCII ?/;/, next to // Persian text is an OCR/encoding artifact to restore, not an authorial style choice .replace(new RegExp('([' + _persianCharacters + '])\\?', 'g'), '$1؟') .replace(new RegExp('([' + _persianCharacters + ']);', 'g'), '$1؛') .replace(new RegExp('([' + _persianCharacters + '])(\\]\\]|»|)\\,', 'g'), '$1$2،') // ensure a space after punctuation when it's glued directly to the next word... .replace(/([;:\?!,.)،؛؟])([^\s0-9۰-۹٠-٩\n}|"«»'’”])/g, '$1 $2') // ...but consecutive punctuation marks (or end of line) don't get a space between them .replace(/([;:\?!,.)،؛؟]) +([\n;:\?!,.)،؛؟\]]|$)/g, '$1$2') // a double period after a Persian letter is almost always meant to be one .replace(new RegExp('([' + _persianCharacters + '])\\.\\. (?=[' + _persianCharacters + '])', 'g'), '$1. ') // three-or-more dots after Persian text is an ellipsis, not literal periods .replace(new RegExp('([' + _persianCharacters + '])( *)\\.{3,}', 'g'), '$1$2…') // canonicalise ۀ / هٴ / هٓ to ISIRI 6219's ه + combining hamza above (هٔ) - this is a // same-character encoding choice, not a wording change, so it's safe to automate .replace(/[ۂۀ]/g, 'هٔ') .replace(/هٴ/g, 'هٔ') .replace(/هٓ/g, 'هٔ') // unicodify .replace(/&mdash;/g, '—') .replace(/&ndash;/g, '–') .replace(/&quot;/g, '"') .replace(/<center>\s*([.\n]*?)\s*<\/center>/g, '{{center|$1}}'); // Run the invisible-ZWNJ-artifact cleanup once more: the punctuation and digit changes // above can newly strand a ZWNJ next to something it can no longer affect (e.g. a ZWNJ // that's now right before a converted Persian digit or a newly-inserted punctuation mark). editor.set(_cleanupZwnjArtifacts(editor.get())); }; // Verb stems used by smartZwnj() below to decide when "می"/"نمی" is a verb prefix that should // join to what follows with a ZWNJ, rather than something else (most importantly, the classical // word «می» meaning "wine" - extremely common in the poetry Wikisource hosts a lot of, e.g. // Hafez or Khayyam). Matching only against a real verb stem from this list, with a real person/ // tense suffix, is what makes this safe enough to automate at all; a blind "می" + space match // would wrongly rewrite «می ناب» (pure wine) in a ghazal. Ported verbatim from the fa.wikipedia/ // fa.wikisource "Persian text style improvement tools" gadget, since retyping ~500 verb stems by // hand is exactly how new bugs get introduced. var _persianPastVerbs = '(' + 'ارزید|افتاد|افراشت|افروخت|افزود|افسرد|افشاند|افکند|انباشت|انجامید|انداخت|اندوخت|اندود|اندیشید|انگاشت|انگیخت|انگیزاند|اوباشت|ایستاد' + '|آراست|آراماند|آرامید|آرمید|آزرد|آزمود|آسود|آشامید|آشفت|آشوبید|آغازید|آغشت|آفرید|آکند|آگند|آلود|آمد|آمرزید|آموخت|آموزاند' + '|آمیخت|آهیخت|آورد|آویخت|باخت|باراند|بارید|بافت|بالید|باوراند|بایست|بخشود|بخشید|برازید|برد|برید|بست|بسود|بسیجید|بلعید' + '|بود|بوسید|بویید|بیخت|پاشاند|پاشید|پالود|پایید|پخت|پذیراند|پذیرفت|پراکند|پراند|پرداخت|پرستید|پرسید|پرهیزید|پروراند|پرورد|پرید' + '|پژمرد|پژوهید|پسندید|پلاسید|پلکید|پناهید|پنداشت|پوسید|پوشاند|پوشید|پویید|پیچاند|پیچانید|پیچید|پیراست|پیمود|پیوست|تاباند|تابید|تاخت' + '|تاراند|تازاند|تازید|تافت|تپاند|تپید|تراشاند|تراشید|تراوید|ترساند|ترسید|ترشید|ترکاند|ترکید|تکاند|تکانید|تنید|توانست|جَست|جُست' + '|جست|جنباند|جنبید|جنگید|جهاند|جهید|جوشاند|جوشید|جوید|چاپید|چایید|چپاند|چپید|چراند|چربید|چرخاند|چرخید|چرید|چسباند|چسبید' + '|چشاند|چشید|چکاند|چکید|چلاند|چلانید|چمید|چید|خاراند|خارید|خاست|خایید|خراشاند|خراشید|خرامید|خروشید|خرید|خزید|خشکاند' + '|خشکید|خفت|خلید|خمید|خنداند|خندانید|خندید|خواباند|خوابانید|خوابید|خواست|خواند|خوراند|خورد|خوفید|خیساند|خیسید|داد|داشت|دانست' + '|درخشانید|درخشید|دروید|درید|دزدید|دمید|دواند|دوخت|دوشید|دوید|دید|دیدم|راند|ربود|رخشید|رساند|رسانید|رست|رَست|رُست' + '|رسید|رشت|رفت|رُفت|رقصاند|رقصید|رمید|رنجاند|رنجید|رندید|رهاند|رهانید|رهید|روبید|روفت|رویاند|رویید|ریخت|رید|ریسید' + '|زاد|زارید|زایید|زد|زدود|زیست|سابید|ساخت|سپارد|سپرد|سپوخت|ستاند|ستد|سترد|ستود|ستیزید|سرایید|سرشت|سرود|سرید' + '|سزید|سفت|سگالید|سنجید|سوخت|سود|سوزاند|شاشید|شایست|شتافت|شد|شست|شکافت|شکست|شکفت|شکیفت|شگفت|شمارد|شمرد|شناخت' + '|شناساند|شنید|شوراند|شورید|طپید|طلبید|طوفید|غارتید|غرید|غلتاند|غلتانید|غلتید|غلطاند|غلطانید|غلطید|غنود|فرستاد|فرسود|فرمود|فروخت' + '|فریفت|فشاند|فشرد|فهماند|فهمید|قاپید|قبولاند|کاست|کاشت|کاوید|کرد|کشاند|کشانید|کشت|کشید|کفت|کفید|کند|کوبید|کوچید' + '|کوشید|کوفت|گَزید|گُزید|گایید|گداخت|گذارد|گذاشت|گذراند|گذشت|گرازید|گرایید|گرداند|گردانید|گردید|گرفت|گروید|گریاند|گریخت|گریست' + '|گزارد|گزید|گسارد|گستراند|گسترد|گسست|گسیخت|گشت|گشود|گفت|گمارد|گماشت|گنجاند|گنجانید|گنجید|گندید|گوارید|گوزید|لرزاند|لرزید' + '|لغزاند|لغزید|لمباند|لمدنی|لمید|لندید|لنگید|لهید|لولید|لیسید|ماسید|مالاند|مالید|ماند|مانست|مرد|مکشید|مکید|مولید|مویید' + '|نازید|نالید|نامید|نشاند|نشست|نکوهید|نگاشت|نگریست|نمایاند|نمود|نهاد|نهفت|نواخت|نوردید|نوشاند|نوشت|نوشید|نیوشید|هراسید|هشت' + '|ورزید|وزاند|وزید|یارست|یازید|یافت' + ')'; var _persianPresentVerbs = '(' + 'ارز|افت|افراز|افروز|افزا|افزای|افسر|افشان|افکن|انبار|انباز|انجام|انداز|اندای|اندوز|اندیش|انگار|انگیز|انگیزان' + '|اوبار|ایست|آرا|آرام|آرامان|آرای|آزار|آزما|آزمای|آسا|آسای|آشام|آشوب|آغار|آغاز|آفرین|آکن|آگن|آلا|آلای' + '|آمرز|آموز|آموزان|آمیز|آهنج|آور|آویز|آی|بار|باران|باز|باش|باف|بال|باوران|بای|باید|بخش|بخشا|بخشای' + '|بر|بَر|بُر|براز|بساو|بسیج|بلع|بند|بو|بوس|بوی|بیز|بین|پا|پاش|پاشان|پالا|پالای|پذیر|پذیران' + '|پر|پراکن|پران|پرداز|پرس|پرست|پرهیز|پرور|پروران|پز|پژمر|پژوه|پسند|پلاس|پلک|پناه|پندار|پوس|پوش|پوشان' + '|پوی|پیچ|پیچان|پیرا|پیرای|پیما|پیمای|پیوند|تاب|تابان|تاران|تاز|تازان|تپ|تپان|تراش|تراشان|تراو|ترس|ترسان' + '|ترش|ترک|ترکان|تکان|تن|توان|توپ|جنب|جنبان|جنگ|جه|جهان|جو|جوش|جوشان|جوی|چاپ|چای|چپ|چپان' + '|چر|چران|چرب|چرخ|چرخان|چسب|چسبان|چش|چشان|چک|چکان|چل|چلان|چم|چین|خار|خاران|خای|خر|خراش' + '|خراشان|خرام|خروش|خز|خشک|خشکان|خل|خم|خند|خندان|خواب|خوابان|خوان|خواه|خور|خوران|خوف|خیز|خیس' + '|خیسان|دار|درخش|درخشان|درو|دزد|دم|ده|دو|دوان|دوز|دوش|ران|ربا|ربای|رخش|رس|رسان' + '|رشت|رقص|رقصان|رم|رنج|رنجان|رند|ره|رهان|رو|روب|روی|رویان|ریز|ریس|رین|زا|زار|زای|زدا' + '|زدای|زن|زی|ساب|ساز|سای|سپار|سپر|سپوز|ستا|ستان|ستر|ستیز|سر|سرا|سرای|سرشت|سز|سگال|سنب' + '|سنج|سوز|سوزان|شاش|شای|شتاب|شکاف|شکف|شکن|شکوف|شکیب|شمار|شمر|شناس|شناسان|شنو|شو|شور|شوران|شوی' + '|طپ|طلب|طوف|غارت|غر|غلت|غلتان|غلط|غلطان|غنو|فرسا|فرسای|فرست|فرما|فرمای|فروش|فریب|فشار|فشان|فشر' + '|فهم|فهمان|قاپ|قبولان|کار|کاه|کاو|کش|کَش|کُش|کِش|کشان|کف|کن|کوب|کوچ|کوش|گا|گای|گداز' + '|گذار|گذر|گذران|گرا|گراز|گرای|گرد|گردان|گرو|گری|گریان|گریز|گز|گزار|گزین|گسار|گستر|گستران|گسل|گشا' + '|گشای|گمار|گنج|گنجان|گند|گو|گوار|گوز|گوی|گیر|لرز|لرزان|لغز|لغزان|لم|لمبان|لند|لنگ|له|لول' + '|لیس|ماس|مال|مان|مک|مول|موی|میر|ناز|نال|نام|نشان|نشین|نکوه|نگار|نگر|نما|نمای|نمایان|نه' + '|نهنب|نواز|نورد|نوش|نوشان|نویس|نیوش|هراس|هست|هل|ورز|وز|وزان|یاب|یار|یاز' + ')'; var _persianComplexPastVerbs = { 'باز': 'آفرید|آمد|آموخت|آورد|ایستاد|تابید|جست|خواند|داشت|رساند|ستاند|شمرد|ماند|نمایاند|نهاد|نگریست|پرسید|گذارد' + '|گرداند|گردید|گرفت|گشت|گشود|گفت|یافت', 'در': 'بر ?داشت|بر ?گرفت|آمد|آمیخت|آورد|آویخت|افتاد|افکند|انداخت|رفت|ماند|نوردید|کشید|گرفت', 'بر': 'آشفت|آمد|آورد|افتاد|افراشت|افروخت|افشاند|افکند|انداخت|انگیخت|تاباند|تابید|تافت|تنید|جهید|خاست|خواست|خورد' + '|داشت|دمید|شمرد|نهاد|چید|کرد|کشید|گرداند|گردانید|گردید|گزید|گشت|گشود|گمارد|گماشت', 'فرو': 'آمد|خورد|داد|رفت|نشاند|کرد|گذارد|گذاشت', 'وا': 'داشت|رهاند|ماند|نهاد|کرد', 'ور': 'آمد|افتاد|رفت', 'یاد': 'گرفت', 'پراکنده': 'ساخت', 'زمین': 'خورد', 'گول': 'زد', 'لخت': 'کرد' }; var _persianComplexPresentVerbs = { 'باز': 'آفرین|آموز|آور|ایست|تاب|جو|خوان|دار|رس|ستان|شمار|مان|نمایان|نه|نگر|پرس|گذار|گردان|گرد|گشا|گو|گیر|یاب', 'در': 'بر ?دار|بر ?گیر|آمیز|آور|آویز|افت|افکن|انداز|مان|نورد|کش|گذر|گیر', 'بر': 'آشوب|آور|افت|افراز|افروز|افشان|افکن|انداز|انگیز|تابان|تاب|تن|جه|خواه|خور|خیز|دار|دم|شمار|نه|چین|کش|کن' + '|گردان|گزین|گشا|گمار', 'فرو': 'خور|ده|رو|نشین|کن|گذار', 'وا': 'دار|رهان|مان|نه|کن', 'ور': 'افت|رو', 'یاد': 'گیر', 'پراکنده': 'ساز', 'زمین': 'خور', 'گول': 'زن', 'لخت': 'کن' }; /** * Join compound verbs (باز آفرید, در آمد, بر داشت, etc.) with a ZWNJ instead of a space, using * the curated prefix/stem pairs above so only real compound verbs match. * @param {string} text The text to process. */ var _applyComplexVerbZwnj = function(text) { var prefix, stems; for (prefix in _persianComplexPastVerbs) { if (!_persianComplexPastVerbs.hasOwnProperty(prefix)) continue; stems = _persianComplexPastVerbs[prefix]; text = text.replace(new RegExp( '(^|[^' + _persianCharacters + '])(' + prefix + ') ?(می|نمی|)( |\u200c|)(ن|)(' + stems + ')(م|ی|یم|ید|ند|ه|ن|)($|[^' + _persianCharacters + '])', 'g'), '$1$2\u200c$3\u200c$5$6$7$8'); } for (prefix in _persianComplexPresentVerbs) { if (!_persianComplexPresentVerbs.hasOwnProperty(prefix)) continue; stems = _persianComplexPresentVerbs[prefix]; text = text.replace(new RegExp( '(^|[^' + _persianCharacters + '])(' + prefix + ') ?(می|نمی|)( |\u200c|)(ن|)(' + stems + ')(م|ی|د|یم|ید|ند|ن)($|[^' + _persianCharacters + '])', 'g'), '$1$2\u200c$3\u200c$5$6$7$8'); } return text; }; /** * Join می/نمی to a following verb stem with a ZWNJ (می‌رود), join the plural/superlative * suffixes ها/ترین with a ZWNJ, and a handful of specific, well-tested exceptions (میدانی, * میتوان, میگوی دریایی, میدوی) that would otherwise be wrongly split by the general rules. * Ported from the same gadget's applyZwnj(); see the smartZwnj() comment for the caveat this * still doesn't (and can't) fully resolve. * @param {string} text The text to process. */ var _applyZwnj = function(text) { text = _applyComplexVerbZwnj(text); return _cleanupZwnjArtifacts(text) .replace( new RegExp('(^|[^' + _persianCharacters + '])(می|نمی) ?' + _persianPastVerbs + '(م|ی|یم|ید|ند|ه|)($|[^' + _persianCharacters + '])', 'g'), '$1$2\u200c$3$4$5' ) .replace( new RegExp('(^|[^' + _persianCharacters + '])(می|نمی) ?' + _persianPresentVerbs + '(م|ی|د|یم|ید|ند)($|[^' + _persianCharacters + '])', 'g'), '$1$2\u200c$3$4$5' ) // ماضی نقلی: خورده‌ام, رفته‌اید, ... .replace( new RegExp('(^|[^' + _persianCharacters + '])(ن|)' + _persianPastVerbs + 'ه (ام|ای|ایم|اید|اند)($|[^' + _persianCharacters + '])', 'g'), '$1$2$3ه\u200c$4$5' ) .replace(new RegExp('([' + _persianCharacters + '])ه‌است($|[^' + _persianCharacters + '])', 'g'), '$1ه است$2') // «دان» handled separately: its «ی» suffix collides with the verb «دانی»/میدانی .replace( new RegExp('(^|[^' + _persianCharacters + '])(می|نمی) ?(دان)(م|د|یم|ید|ند)($|[^' + _persianCharacters + '])', 'g'), '$1$2\u200c$3$4$5' ) .replace(/(\s)(می|نمی) ?توان/g, '$1$2\u200cتوان') // «ها» and «ترین» always join with a ZWNJ, never a full space .replace(/ ها([\]\.،\:»\)\s]|\'{2,3}|\={2,})/g, '\u200cها$1') .replace(/ ها(ی|یی|یم|یت|یش|ی?مان|ی?تان|ی?شان)([\]\.،\:»\)\s])/g, '\u200cها$1$2') .replace(/هها/g, 'ه‌ها') .replace(/ ترین([\]\.،\:»\)\s]|\'{2,3}|\={2,})/g, '\u200cترین$1') // a few specific words that would otherwise be caught by the general rules above .replace(new RegExp('می\u200cگوی($|[^' + _persianCharacters + '\u200c])', 'g'), 'میگوی$1') // میگوی دریایی .replace(new RegExp('می\u200cدوی($|[^' + _persianCharacters + '\u200c])', 'g'), 'میدوی$1'); // میدوی (ابهام‌زدایی) }; /** * Join می/نمی verb prefixes (and a few common suffixes) to what follows with a ZWNJ instead of * a full space, using curated Persian verb-stem lists so it only fires on real verbs. * * This is scoped to your selection rather than the whole page on purpose: even with the word * lists, «می» on its own is also the classical word for "wine" (می ناب، می‌گلگون، ساقی می ده…), * and Wikisource hosts a lot of exactly the poetry where that comes up. The word-list approach * makes this far safer than a blind "می" + space replace, but it can't tell «می» the prefix from * «می» the noun by context alone - so select the passage you've checked, rather than running it * across a whole ghazal unread. * @param {object} editor The script helpers for the page. */ var smartZwnj = function(editor) { editor.replaceSelection(_applyZwnj); }; }); // </nowiki> jjhu7s091hapt3gxqwwnr3hbjsr50kn 299772 299769 2026-09-20T15:03:27Z Hanooz 17889 299772 javascript text/javascript /* This page defines a TemplateScript library for fa.wikisource.org (ویکی‌نبشتهٔ فارسی). It's not meant to be referenced directly - install it the same way as the upstream English version described at [[Wikisource:TemplateScript]]. This is a Persian Wikisource adaptation of Pathoschild's proofreading and typography libraries, pared down to a focused four-action toolset. It draws on the fa.wikipedia/ fa.wikisource "Persian text style improvement tools" gadget (Gadget-Extra-Editbuttons- persianwikitools.js) for several of the character/orthography helpers, since that's the community's own tested implementation rather than something invented for this file. */ /* global $, pathoschild */ /** * TemplateScript adds configurable templates and scripts to the sidebar, and adds an example regex editor. * @see https://meta.wikimedia.org/wiki/TemplateScript * * The upstream proofreading library's @update-token has been removed on purpose: Wikisource's * maintenance bot uses that token to overwrite local copies with the vanilla English script, * which would silently wipe out every change made here for fa.wikisource. See [[Wikisource:Tools * and scripts]] for how the update-bot category works. */ // <nowiki> $.ajax('//tools-static.wmflabs.org/meta/scripts/pathoschild.templatescript.js', { dataType:'script', cache:true }).then(function() { /********* ** Define library *********/ pathoschild.TemplateScript.library.define({ key: 'wikisource.fa.tools', name: 'ابزارهای ویکی‌نبشتهٔ فارسی', url: '//fa.wikisource.org/wiki/راهنما:نمونه‌خوانی', description: 'مجموعه‌ای از ابزارها برای <a href="/wiki/راهنما:نمونه‌خوانی">نمونه‌خوانی آثار در فضای نام <tt>برگه:</tt></a>: پاک‌سازی OCR (با اصلاح نویسه‌ها، کشیده، ارقام، جای اعراب، نیم‌فاصله، نویسه‌های نامرئی بی‌اثر، و نشانه‌گذاری) و اصلاح نیم‌فاصلهٔ افعال.', categories: [ { name: 'ابزارهای برگه', scripts: [ { key: 'cleanup-ocr', name: 'پاک‌سازی OCR', script: function(editor) { pageCleanup(editor); }, forNamespaces: 'page' }, { key: 'smart-zwnj', name: 'اصلاح نیم‌فاصلهٔ افعال', script: function(editor) { smartZwnj(editor); }, forNamespaces: 'page' } ] } ] }); /********* ** Private methods *********/ // Digit tables used to convert between Latin, Persian and Arabic-Indic numerals. var _persianDigits = '۰۱۲۳۴۵۶۷۸۹'; var _arabicIndicDigits = '٠١٢٣٤٥٦٧٨٩'; var _latinDigits = '0123456789'; // Character-class tables used by the Persian-specific cleanup helpers below. Ported from the // fa.wikipedia/fa.wikisource "Persian text style improvement tools" gadget (Gadget-Extra- // Editbuttons-persianwikitools.js) - the community's own tested implementation of this // normalisation, rather than something invented for this file. var _hamza = '\u0654'; var _persianVowels = '\u064B-\u0650\u0652\u0670'; // non-standard letterforms (Arabic ي/ك, Urdu, Pashto, Uyghur presentation forms, etc.) that // should still count as "a Persian-ish letter" for context-matching, even before // _toStandardPersianCharacters() below has normalised them var _similarPersianCharacters = '\u0643\uFB91\uFB90\uFB8F\uFB8E\uFEDC\uFEDB\uFEDA\uFED9\u0649\uFEEF\u064A\u06C1\u06D5\u06BE\uFEF0-\uFEF4'; var _persianCharacters = '\u0621-\u0655\u067E\u0686\u0698\u06AF\u06A9\u0643\u06AA\uFED9\uFEDA\u06CC\uFEF1\uFEF2' + _similarPersianCharacters; /** * Convert any mix of Latin or Arabic-Indic digits in a string to Persian digits, for display in wikitext. * @param {string} text The text to convert. */ var _digitsToPersian = function(text) { var result = ''; for (var i = 0; i < text.length; i++) { var ch = text.charAt(i); var idx = _latinDigits.indexOf(ch); if (idx === -1) idx = _arabicIndicDigits.indexOf(ch); result += (idx !== -1) ? _persianDigits.charAt(idx) : ch; } return result; }; /** * OCR/PDF presentation-form and script-variant glyphs that extraction sometimes leaves as * literal characters instead of the standard Persian letter - a common problem specifically on * PDF-sourced works, where each glyph can get saved as its own shaped codepoint. Keys are the * standard letter (or letter + a real ZWNJ, for the two glyphs that visually look joined); * values are a character class of the presentation-form/variant glyphs that should fold into it. */ var _persianGlyphs = { '\u200cه': 'ﻫ', 'ی\u200c': 'ﻰﻲ', 'أ': 'ﺄﺃﺃ', 'آ': 'ﺁﺁﺂ', 'إ': 'ﺇﺈﺇ', 'ا': 'ﺍﺎ', 'ب': 'ﺏﺐﺑﺒ', 'پ': 'ﭖﭗﭘﭙ', 'ت': 'ﺕﺖﺗﺘ', 'ث': 'ﺙﺚﺛﺜ', 'ج': 'ﺝﺞﺟﺠ', 'چ': 'ﭺﭻﭼﭽ', 'ح': 'ﺡﺢﺣﺤ', 'خ': 'ﺥﺦﺧﺨ', 'د': 'ﺩﺪ', 'ذ': 'ﺫﺬ', 'ر': 'ﺭﺮ', 'ز': 'ﺯﺰ', 'ژ': 'ﮊﮋ', 'س': 'ﺱﺲﺳﺴ', 'ش': 'ﺵﺶﺷﺸ', 'ص': 'ﺹﺺﺻﺼ', 'ض': 'ﺽﺾﺿﻀ', 'ط': 'ﻁﻂﻃﻄ', 'ظ': 'ﻅﻆﻇﻈ', 'ع': 'ﻉﻊﻋﻌ', 'غ': 'ﻍﻎﻏﻐ', 'ف': 'ﻑﻒﻓﻔ', 'ق': 'ﻕﻖﻗﻘ', 'ک': 'ﮎﮏﮐﮑﻙﻚﻛﻜ', 'گ': 'ﮒﮓﮔﮕ', 'ل': 'ﻝﻞﻟﻠ', 'م': 'ﻡﻢﻣﻤ', 'ن': 'ﻥﻦﻧﻨ', 'ه': 'ﻩﻪﻫﻬ', 'هٔ': 'ﮤﮥ', 'و': 'ﻭﻮ', 'ؤ': 'ﺅﺅﺆ', 'ی': 'ﯼﯽﯾﯿﻯﻰﻱﻲﻳﻴ', 'ئ': 'ﺉﺊﺋﺌ', 'لا': 'ﻻﻼ', 'لإ': 'ﻹﻺ', 'لأ': 'ﻸﻷ', 'لآ': 'ﻵﻶ' }; /** * Remove stray/duplicate ZWNJ-family characters and OCR artifacts that stand in for a ZWNJ, * without trying to guess where a NEW ZWNJ should be inserted between words - that needs the * verb word-lists in smartZwnj() below, since blindly inserting one is what risks corrupting * unrelated text (see that function's comment for why). * @param {string} text The text to clean. */ var _cleanupZwnjArtifacts = function(text) { return text // a ZWNJ (or stray zero-width/BOM/bidi-mark character) with nothing before or after it // to join or not-join has no possible effect on the rendered output - most commonly a // ZWNJ that ended up with a plain space next to it instead of a letter .replace(/^[\u200B-\u200D\uFEFF]+/, '') .replace(/[\u200B-\u200D\uFEFF]+$/, '') // a stray LRM/RLM between two Persian letters was almost always meant to be a ZWNJ .replace(new RegExp('([' + _persianCharacters + '] *)[\u200F\u200E]+( *[' + _persianCharacters + '])', 'g'), '$1\u200c$2') // collapse repeated invisible-joiner/mark characters down to one .replace(/([\u200B-\u200D\uFEFF\u200E\u200F]){2,}/g, '$1') // ¬ between two Persian letters is almost always a misrecognised ZWNJ, not a real character .replace(new RegExp('([' + _persianCharacters + '])¬(?=[' + _persianCharacters + '])', 'g'), '$1\u200c') // a ZWNJ has no business after a digit or most punctuation/brackets, or next to a Latin word .replace(/([۰-۹0-9إأةؤورزژاآدذ،؛,:«»\\\/@#$٪×*()ـ\-=|ء])\u200c/g, '$1') .replace(/[\u200B-\u200D\uFEFF]([\w])/g, '$1') .replace(/([\w])[\u200B-\u200D\uFEFF]/g, '$1') .replace(new RegExp('[\\u200B-\\u200D\\uFEFF]([' + _persianVowels + _arabicIndicDigits + _persianDigits + _latinDigits + _hamza + '])', 'g'), '$1') .replace(new RegExp('([' + _arabicIndicDigits + '])[\\u200B-\\u200D\\uFEFF]', 'g'), '$1') .replace(/[\u200B\u200C\uFEFF]([ء\n\s\[\].،«»:()؛؟?;$!@\-=+\\|])/g, '$1') .replace(/([\n\s\[.،«»:()؛؟?;$!@\-=+\\|])[\u200B-\u200D\uFEFF]/g, '$1') .replace(/[\u200B-\u200D\uFEFF](\]\][\s\n])/g, '$1') .replace(/([\n\s]\[\[)[\u200B-\u200D\uFEFF]/g, '$1'); }; /** * Fold OCR/PDF presentation-form glyphs and non-Persian script variants (Arabic ي/ك, Urdu, * Pashto, Uyghur, Kurdish) back to standard Persian characters, and canonicalise the hamza- * bearing forms per ISIRI 6219 (Iran's national character-encoding standard). * @param {string} text The text to clean. */ var _toStandardPersianCharacters = function(text) { for (var standard in _persianGlyphs) { if (_persianGlyphs.hasOwnProperty(standard)) { text = text.replace(new RegExp('[' + _persianGlyphs[standard] + ']', 'g'), standard); } } return _cleanupZwnjArtifacts(text) // needed because of the two ZWNJ-bearing keys above .replace(/ك/g, 'ک') // Arabic .replace(/ڪ/g, 'ک') // Urdu .replace(/ﻙ/g, 'ک') // Pushtu .replace(/ﻚ/g, 'ک') // Uyghur .replace(/ي/g, 'ی') // Arabic .replace(/ى/g, 'ی') // Urdu .replace(/ے/g, 'ی') // Urdu .replace(/ۍ/g, 'ی') // Pushtu .replace(/ې/g, 'ی') // Uyghur .replace(/ہ/g, 'ه') // Urdu .replace(/ە/g, 'ه\u200c') // Kurdish .replace(/ھ/g, 'ه') // Kurdish .replace(/أ/g, 'أ') // canonical alef+hamza form per ISIRI 6219 .replace(/آ/g, 'آ'); }; /** * Strip tatweel/kashide (ـ, U+0640) characters from a string. Tatweel is a pure justification * stretch-mark used in decorative Persian typesetting (common on title pages); OCR frequently * captures it as literal characters, but it has no place in a wikilink target or in wikitext. * @param {string} text The text to clean. */ var _stripTatweel = function(text) { return text.replace(/\u0640+/g, ''); }; /********* ** Script methods *********/ /** * Clean up OCR errors in the text, and push <noinclude> content at the top * & bottom of the page into the header & footer boxes respectively. * @param {object} editor The script helpers for the page. */ var pageCleanup = function(editor) { // Strip characters that have no effect on the rendered output at all, before anything // else runs: control characters, BOM and soft hyphen have no legitimate place in body // text; \r is a Windows line-ending artifact once \n is already there; a run of exotic // Unicode space characters (NBSP, thin space, ideographic space, etc.) right before a // line break just disappears, a tab/NBSP right after one disappears too (plain space is // deliberately left alone there, since a leading space triggers MediaWiki's <pre> // formatting), and any exotic space left in running text becomes a plain space. Ported // from the fa.wikipedia/fa.wikisource gadget's applyOrthography(), and cross-checked // against Virastar (github.com/aziz/virastar and its ports, a widely used standalone // Persian text cleaner) - both do this same normalisation as a first step. var text = editor.get() .replace(/\r/g, '') .replace(/[\x00-\x08\x0B\x0C\x0E-\x1F\x7F-\x9F\uFEFF\u00AD\u0085]+/g, '') .replace(/[ \xA0\u1680\u180E\u2000-\u200A\u202F\u205F\u3000]+\n/g, '\n') .replace(/\n[\t\u00A0]+/g, '\n') .replace(/[\xA0\u1680\u180E\u2000-\u200A\u202F\u205F\u3000]/g, ' '); // Normalise the raw text next: fold OCR/PDF presentation-form glyphs and non-Persian // letter variants to standard Persian characters (this also mops up stray ZWNJ-family // characters left over from OCR - see _cleanupZwnjArtifacts, including a ZWNJ that's now // got a plain space next to it instead of a letter, which can no longer do anything), // strip decorative tatweel, convert Latin/Arabic-Indic digits to Persian, and fix vowel- // mark (harakat) placement - a mark separated from its letter by a stray space moves back // next to the letter, and duplicate marks on the same letter collapse to one. The digit // conversion runs unconditionally on the whole page - if a work legitimately needs // Western digits somewhere (an ISBN, a citation in another language), those get // converted too; there's no separate opt-in step for that anymore. text = _toStandardPersianCharacters(text); text = _stripTatweel(text); text = _digitsToPersian(text); text = text.replace(new RegExp('([' + _persianCharacters + _persianVowels + _hamza + '])(\\s)([' + _persianVowels + _hamza + '])', 'g'), '$1$3$2'); text = text.replace(new RegExp('([' + _persianVowels + _hamza + ']){2,}', 'g'), '$1'); editor.set(text); // Strip leading lines that are pure page-number "running header" junk - the digits-only // case of the en.wikisource WsCleanup gadget's running-header patterns. Dropped that // gadget's other pattern (digits plus an ALL-CAPS title fragment) since Persian has no // letter case to signal "this is header text, not body text" the way English capitals // do. Caveat carried over from the source: a work with numbered stanzas/verses starting // right at the top of a page could have its number line mistaken for header junk here - // worth a glance after running this on poetry. (function() { var lines = editor.get().split(/\r?\n/); var pageNumberOnly = /^[\s0-9۰-۹٠-٩.,\-–—]+$/; var start = 0; for (var li = 0; li < lines.length; li++) { if (lines[li].trim().length === 0 || (pageNumberOnly.test(lines[li]) && lines[li].trim().length > 0)) { start++; continue; } break; } if (start > 0) { editor.set(lines.slice(start).join('\n')); } })(); // push <noinclude> content at the top & bottom into the header & footer if (editor.get().match(/^<noinclude\>/)) { var text = editor.get(); var e = text.indexOf("</noinclude>"); $('#wpHeaderTextbox').val(function(i, val) { return $.trim(val + "\n" + text.substr(11, e-11).replace(/^\s+|\s+$/g, '')); }); editor.set(text.substr(e+12)); } if (editor.get().match(/<\/noinclude\>$/)) { var text = editor.get(); var s = text.lastIndexOf("<noinclude>"); $('#wpFooterTextbox').val(function(i, val) { return $.trim(text.substr(s+11, text.length-s-11-12).replace(/^\s+|\s+$/g, '') + "\n" + val); }); editor.set(text.substr(0, s)); } // clean up text editor // remove trailing spaces at the end of each line .replace(/ +\n/g, '\n') // remove trailing whitespace preceding a hard line break .replace(/ +<br *\/?>/g, '<br />') // remove trailing whitespace and numerals at the end of page text - matches // Latin, Persian and Arabic-Indic digits, since page-number artifacts at the // bottom of a scanned fa.wikisource page can OCR as any of the three .replace(/[\s0-9۰-۹٠-٩]+$/g, '') // remove trailing spaces at the end of refs .replace(/ +<\/ref>/g, '</ref>') // remove trailing spaces at the end of template calls .replace(/ +}}/g, '}}') // convert double-hyphen to mdash (avoiding breaking HTML comment syntax) .replace(/([^\!])--([^>])/g, '$1—$2') // remove spacing around mdash, but only if it has spaces on both sides .replace(/ +— +/g, '—') // "Digitized by Google" scan-watermark text and stray OCR of the word "Google" - // the watermark itself is always in English regardless of the scanned work's // language, so this is just as relevant to a Google Books-sourced Persian scan .replace(/\s?D[il]g[il]t[il][sz][eco]d\s+by[^\n]*\s+([6G][Oo0]{2}g[lI]e)?/g, '') .replace(/\bG[oO0]{2}gle\b/g, '') // highly suspicious OCR noise characters .replace(/[■•]/g, '') // a line containing nothing but a stray punctuation mark is likely scan noise .replace(/^[.,^،؛]$/gm, '') // underscores (often OCR of underlined text or a form field) aren't meaningful in wikitext .replace(/_/g, ' ') // number ranges get an en dash, not a hyphen, per fa.wikisource's وپ:خط تیره .replace(/([۰-۹]+)[ \t]?-[ \t]?(?=[۰-۹])/g, '$1–') // <section begin=x/> needs to be <section begin="x"/> to parse .replace(/<section (begin|end)=(\w[^/]+)\/>/g, '<section $1="$2"/>') // a <ref> shouldn't have a space glued before it .replace(/ <ref/g, '<ref') // {{hws}}/{{hwe}} shorthand for a word split across a page boundary, expanded to the // full template names (same English names used across Wikisource language versions, // fa.wikisource included, for this kind of utility template) .replace(/{{hws\|/g, '{{hyphenated word start|') .replace(/{{hwe\|/g, '{{hyphenated word end|') // {{c|...}} shorthand for {{center|...}}, if used .replace(/{{c\|/g, '{{center|'); // clean up pages if they don't have <poem> if (!editor.contains('<poem>') && !editor.contains('{{' + 'ppoem')) { editor // remove single line breaks; preserve multiple. // but not if there's a tag, template or table syntax either side of the line break .replace(/([^>}\|\n])\n([^:#\*<{\|\n])/g, '$1 $2') // collapse sequences of spaces into a single space .replace(/ +/g, ' '); } // more page cleanup editor // dump spurious hard breaks at the end of paragraphs .replace(/<br *\/?>\n\n/g, '\n\n') // remove unwanted spaces before punctuation marks - both Latin (;:?!,.)) and // Persian (،؛؟) forms, since either can turn up depending on the OCR engine .replace(/ ([;:\?!,.)،؛؟])/g, '$1') // and no space after an opening parenthesis .replace(/\( +/g, '(') // convert ASCII look-alikes to proper Persian punctuation when they follow Persian // text. Properly printed Persian sources always used ؟/؛/، - an ASCII ?/;/, next to // Persian text is an OCR/encoding artifact to restore, not an authorial style choice .replace(new RegExp('([' + _persianCharacters + '])\\?', 'g'), '$1؟') .replace(new RegExp('([' + _persianCharacters + ']);', 'g'), '$1؛') .replace(new RegExp('([' + _persianCharacters + '])(\\]\\]|»|)\\,', 'g'), '$1$2،') // ensure a space after punctuation when it's glued directly to the next word... .replace(/([;:\?!,.)،؛؟])([^\s0-9۰-۹٠-٩\n}|"«»'’”])/g, '$1 $2') // ...but consecutive punctuation marks (or end of line) don't get a space between them .replace(/([;:\?!,.)،؛؟]) +([\n;:\?!,.)،؛؟\]]|$)/g, '$1$2') // a double period after a Persian letter is almost always meant to be one .replace(new RegExp('([' + _persianCharacters + '])\\.\\. (?=[' + _persianCharacters + '])', 'g'), '$1. ') // three-or-more dots after Persian text is an ellipsis, not literal periods .replace(new RegExp('([' + _persianCharacters + '])( *)\\.{3,}', 'g'), '$1$2…') // canonicalise ۀ / هٴ / هٓ to ISIRI 6219's ه + combining hamza above (هٔ) - this is a // same-character encoding choice, not a wording change, so it's safe to automate .replace(/[ۂۀ]/g, 'هٔ') .replace(/هٴ/g, 'هٔ') .replace(/هٓ/g, 'هٔ') // unicodify .replace(/&mdash;/g, '—') .replace(/&ndash;/g, '–') .replace(/&quot;/g, '"') .replace(/<center>\s*([.\n]*?)\s*<\/center>/g, '{{center|$1}}'); // Run the invisible-ZWNJ-artifact cleanup once more: the punctuation and digit changes // above can newly strand a ZWNJ next to something it can no longer affect (e.g. a ZWNJ // that's now right before a converted Persian digit or a newly-inserted punctuation mark). editor.set(_cleanupZwnjArtifacts(editor.get())); }; // Verb stems used by smartZwnj() below to decide when "می"/"نمی" is a verb prefix that should // join to what follows with a ZWNJ, rather than something else (most importantly, the classical // word «می» meaning "wine" - extremely common in the poetry Wikisource hosts a lot of, e.g. // Hafez or Khayyam). Matching only against a real verb stem from this list, with a real person/ // tense suffix, is what makes this safe enough to automate at all; a blind "می" + space match // would wrongly rewrite «می ناب» (pure wine) in a ghazal. Ported verbatim from the fa.wikipedia/ // fa.wikisource "Persian text style improvement tools" gadget, since retyping ~500 verb stems by // hand is exactly how new bugs get introduced. var _persianPastVerbs = '(' + 'ارزید|افتاد|افراشت|افروخت|افزود|افسرد|افشاند|افکند|انباشت|انجامید|انداخت|اندوخت|اندود|اندیشید|انگاشت|انگیخت|انگیزاند|اوباشت|ایستاد' + '|آراست|آراماند|آرامید|آرمید|آزرد|آزمود|آسود|آشامید|آشفت|آشوبید|آغازید|آغشت|آفرید|آکند|آگند|آلود|آمد|آمرزید|آموخت|آموزاند' + '|آمیخت|آهیخت|آورد|آویخت|باخت|باراند|بارید|بافت|بالید|باوراند|بایست|بخشود|بخشید|برازید|برد|برید|بست|بسود|بسیجید|بلعید' + '|بود|بوسید|بویید|بیخت|پاشاند|پاشید|پالود|پایید|پخت|پذیراند|پذیرفت|پراکند|پراند|پرداخت|پرستید|پرسید|پرهیزید|پروراند|پرورد|پرید' + '|پژمرد|پژوهید|پسندید|پلاسید|پلکید|پناهید|پنداشت|پوسید|پوشاند|پوشید|پویید|پیچاند|پیچانید|پیچید|پیراست|پیمود|پیوست|تاباند|تابید|تاخت' + '|تاراند|تازاند|تازید|تافت|تپاند|تپید|تراشاند|تراشید|تراوید|ترساند|ترسید|ترشید|ترکاند|ترکید|تکاند|تکانید|تنید|توانست|جَست|جُست' + '|جست|جنباند|جنبید|جنگید|جهاند|جهید|جوشاند|جوشید|جوید|چاپید|چایید|چپاند|چپید|چراند|چربید|چرخاند|چرخید|چرید|چسباند|چسبید' + '|چشاند|چشید|چکاند|چکید|چلاند|چلانید|چمید|چید|خاراند|خارید|خاست|خایید|خراشاند|خراشید|خرامید|خروشید|خرید|خزید|خشکاند' + '|خشکید|خفت|خلید|خمید|خنداند|خندانید|خندید|خواباند|خوابانید|خوابید|خواست|خواند|خوراند|خورد|خوفید|خیساند|خیسید|داد|داشت|دانست' + '|درخشانید|درخشید|دروید|درید|دزدید|دمید|دواند|دوخت|دوشید|دوید|دید|دیدم|راند|ربود|رخشید|رساند|رسانید|رست|رَست|رُست' + '|رسید|رشت|رفت|رُفت|رقصاند|رقصید|رمید|رنجاند|رنجید|رندید|رهاند|رهانید|رهید|روبید|روفت|رویاند|رویید|ریخت|رید|ریسید' + '|زاد|زارید|زایید|زد|زدود|زیست|سابید|ساخت|سپارد|سپرد|سپوخت|ستاند|ستد|سترد|ستود|ستیزید|سرایید|سرشت|سرود|سرید' + '|سزید|سفت|سگالید|سنجید|سوخت|سود|سوزاند|شاشید|شایست|شتافت|شد|شست|شکافت|شکست|شکفت|شکیفت|شگفت|شمارد|شمرد|شناخت' + '|شناساند|شنید|شوراند|شورید|طپید|طلبید|طوفید|غارتید|غرید|غلتاند|غلتانید|غلتید|غلطاند|غلطانید|غلطید|غنود|فرستاد|فرسود|فرمود|فروخت' + '|فریفت|فشاند|فشرد|فهماند|فهمید|قاپید|قبولاند|کاست|کاشت|کاوید|کرد|کشاند|کشانید|کشت|کشید|کفت|کفید|کند|کوبید|کوچید' + '|کوشید|کوفت|گَزید|گُزید|گایید|گداخت|گذارد|گذاشت|گذراند|گذشت|گرازید|گرایید|گرداند|گردانید|گردید|گرفت|گروید|گریاند|گریخت|گریست' + '|گزارد|گزید|گسارد|گستراند|گسترد|گسست|گسیخت|گشت|گشود|گفت|گمارد|گماشت|گنجاند|گنجانید|گنجید|گندید|گوارید|گوزید|لرزاند|لرزید' + '|لغزاند|لغزید|لمباند|لمدنی|لمید|لندید|لنگید|لهید|لولید|لیسید|ماسید|مالاند|مالید|ماند|مانست|مرد|مکشید|مکید|مولید|مویید' + '|نازید|نالید|نامید|نشاند|نشست|نکوهید|نگاشت|نگریست|نمایاند|نمود|نهاد|نهفت|نواخت|نوردید|نوشاند|نوشت|نوشید|نیوشید|هراسید|هشت' + '|ورزید|وزاند|وزید|یارست|یازید|یافت' + ')'; var _persianPresentVerbs = '(' + 'ارز|افت|افراز|افروز|افزا|افزای|افسر|افشان|افکن|انبار|انباز|انجام|انداز|اندای|اندوز|اندیش|انگار|انگیز|انگیزان' + '|اوبار|ایست|آرا|آرام|آرامان|آرای|آزار|آزما|آزمای|آسا|آسای|آشام|آشوب|آغار|آغاز|آفرین|آکن|آگن|آلا|آلای' + '|آمرز|آموز|آموزان|آمیز|آهنج|آور|آویز|آی|بار|باران|باز|باش|باف|بال|باوران|بای|باید|بخش|بخشا|بخشای' + '|بر|بَر|بُر|براز|بساو|بسیج|بلع|بند|بو|بوس|بوی|بیز|بین|پا|پاش|پاشان|پالا|پالای|پذیر|پذیران' + '|پر|پراکن|پران|پرداز|پرس|پرست|پرهیز|پرور|پروران|پز|پژمر|پژوه|پسند|پلاس|پلک|پناه|پندار|پوس|پوش|پوشان' + '|پوی|پیچ|پیچان|پیرا|پیرای|پیما|پیمای|پیوند|تاب|تابان|تاران|تاز|تازان|تپ|تپان|تراش|تراشان|تراو|ترس|ترسان' + '|ترش|ترک|ترکان|تکان|تن|توان|توپ|جنب|جنبان|جنگ|جه|جهان|جو|جوش|جوشان|جوی|چاپ|چای|چپ|چپان' + '|چر|چران|چرب|چرخ|چرخان|چسب|چسبان|چش|چشان|چک|چکان|چل|چلان|چم|چین|خار|خاران|خای|خر|خراش' + '|خراشان|خرام|خروش|خز|خشک|خشکان|خل|خم|خند|خندان|خواب|خوابان|خوان|خواه|خور|خوران|خوف|خیز|خیس' + '|خیسان|دار|درخش|درخشان|درو|دزد|دم|ده|دو|دوان|دوز|دوش|ران|ربا|ربای|رخش|رس|رسان' + '|رشت|رقص|رقصان|رم|رنج|رنجان|رند|ره|رهان|رو|روب|روی|رویان|ریز|ریس|رین|زا|زار|زای|زدا' + '|زدای|زن|زی|ساب|ساز|سای|سپار|سپر|سپوز|ستا|ستان|ستر|ستیز|سر|سرا|سرای|سرشت|سز|سگال|سنب' + '|سنج|سوز|سوزان|شاش|شای|شتاب|شکاف|شکف|شکن|شکوف|شکیب|شمار|شمر|شناس|شناسان|شنو|شو|شور|شوران|شوی' + '|طپ|طلب|طوف|غارت|غر|غلت|غلتان|غلط|غلطان|غنو|فرسا|فرسای|فرست|فرما|فرمای|فروش|فریب|فشار|فشان|فشر' + '|فهم|فهمان|قاپ|قبولان|کار|کاه|کاو|کش|کَش|کُش|کِش|کشان|کف|کن|کوب|کوچ|کوش|گا|گای|گداز' + '|گذار|گذر|گذران|گرا|گراز|گرای|گرد|گردان|گرو|گری|گریان|گریز|گز|گزار|گزین|گسار|گستر|گستران|گسل|گشا' + '|گشای|گمار|گنج|گنجان|گند|گو|گوار|گوز|گوی|گیر|لرز|لرزان|لغز|لغزان|لم|لمبان|لند|لنگ|له|لول' + '|لیس|ماس|مال|مان|مک|مول|موی|میر|ناز|نال|نام|نشان|نشین|نکوه|نگار|نگر|نما|نمای|نمایان|نه' + '|نهنب|نواز|نورد|نوش|نوشان|نویس|نیوش|هراس|هست|هل|ورز|وز|وزان|یاب|یار|یاز' + ')'; var _persianComplexPastVerbs = { 'باز': 'آفرید|آمد|آموخت|آورد|ایستاد|تابید|جست|خواند|داشت|رساند|ستاند|شمرد|ماند|نمایاند|نهاد|نگریست|پرسید|گذارد' + '|گرداند|گردید|گرفت|گشت|گشود|گفت|یافت', 'در': 'بر ?داشت|بر ?گرفت|آمد|آمیخت|آورد|آویخت|افتاد|افکند|انداخت|رفت|ماند|نوردید|کشید|گرفت', 'بر': 'آشفت|آمد|آورد|افتاد|افراشت|افروخت|افشاند|افکند|انداخت|انگیخت|تاباند|تابید|تافت|تنید|جهید|خاست|خواست|خورد' + '|داشت|دمید|شمرد|نهاد|چید|کرد|کشید|گرداند|گردانید|گردید|گزید|گشت|گشود|گمارد|گماشت', 'فرو': 'آمد|خورد|داد|رفت|نشاند|کرد|گذارد|گذاشت', 'وا': 'داشت|رهاند|ماند|نهاد|کرد', 'ور': 'آمد|افتاد|رفت', 'یاد': 'گرفت', 'پراکنده': 'ساخت', 'زمین': 'خورد', 'گول': 'زد', 'لخت': 'کرد' }; var _persianComplexPresentVerbs = { 'باز': 'آفرین|آموز|آور|ایست|تاب|جو|خوان|دار|رس|ستان|شمار|مان|نمایان|نه|نگر|پرس|گذار|گردان|گرد|گشا|گو|گیر|یاب', 'در': 'بر ?دار|بر ?گیر|آمیز|آور|آویز|افت|افکن|انداز|مان|نورد|کش|گذر|گیر', 'بر': 'آشوب|آور|افت|افراز|افروز|افشان|افکن|انداز|انگیز|تابان|تاب|تن|جه|خواه|خور|خیز|دار|دم|شمار|نه|چین|کش|کن' + '|گردان|گزین|گشا|گمار', 'فرو': 'خور|ده|رو|نشین|کن|گذار', 'وا': 'دار|رهان|مان|نه|کن', 'ور': 'افت|رو', 'یاد': 'گیر', 'پراکنده': 'ساز', 'زمین': 'خور', 'گول': 'زن', 'لخت': 'کن' }; /** * Join compound verbs (باز آفرید, در آمد, بر داشت, etc.) with a ZWNJ instead of a space, using * the curated prefix/stem pairs above so only real compound verbs match. * @param {string} text The text to process. */ var _applyComplexVerbZwnj = function(text) { var prefix, stems; for (prefix in _persianComplexPastVerbs) { if (!_persianComplexPastVerbs.hasOwnProperty(prefix)) continue; stems = _persianComplexPastVerbs[prefix]; text = text.replace(new RegExp( '(^|[^' + _persianCharacters + ']) *(' + prefix + ') +(می|نمی|)( |\u200c|)(ن|)(' + stems + ')(م|ی|یم|ید|ند|ه|ن|)($|[^' + _persianCharacters + '])', 'g'), '$1$2\u200c$3\u200c$5$6$7$8'); } for (prefix in _persianComplexPresentVerbs) { if (!_persianComplexPresentVerbs.hasOwnProperty(prefix)) continue; stems = _persianComplexPresentVerbs[prefix]; text = text.replace(new RegExp( '(^|[^' + _persianCharacters + ']) *(' + prefix + ') +(می|نمی|)( |\u200c|)(ن|)(' + stems + ')(م|ی|د|یم|ید|ند|ن)($|[^' + _persianCharacters + '])', 'g'), '$1$2\u200c$3\u200c$5$6$7$8'); } return text; }; /** * Join می/نمی to a following verb stem with a ZWNJ (می‌رود), join the plural/superlative * suffixes ها/ترین with a ZWNJ, and a handful of specific, well-tested exceptions (میدانی, * میتوان, میگوی دریایی, میدوی) that would otherwise be wrongly split by the general rules. * * Every pattern below requires an ACTUAL existing space to convert - می and the verb stem * must already be written as two separate space-divided words. An already-solid word like * «میشود» or «برداشت», with no space anywhere, is left exactly as-is: there is nothing there * to "fix", and inserting a ZWNJ that wasn't in the source at all is exactly the kind of edit * that changes what's actually on the page without a real reason to. * * Ported from the same gadget's applyZwnj(); see the smartZwnj() comment for the caveat this * still doesn't (and can't) fully resolve. * @param {string} text The text to process. */ var _applyZwnj = function(text) { text = _applyComplexVerbZwnj(text); return _cleanupZwnjArtifacts(text) .replace( new RegExp('(^|[^' + _persianCharacters + '])(می|نمی) +' + _persianPastVerbs + '(م|ی|یم|ید|ند|ه|)($|[^' + _persianCharacters + '])', 'g'), '$1$2\u200c$3$4$5' ) .replace( new RegExp('(^|[^' + _persianCharacters + '])(می|نمی) +' + _persianPresentVerbs + '(م|ی|د|یم|ید|ند)($|[^' + _persianCharacters + '])', 'g'), '$1$2\u200c$3$4$5' ) // ماضی نقلی: خورده‌ام, رفته‌اید, ... .replace( new RegExp('(^|[^' + _persianCharacters + '])(ن|)' + _persianPastVerbs + 'ه (ام|ای|ایم|اید|اند)($|[^' + _persianCharacters + '])', 'g'), '$1$2$3ه\u200c$4$5' ) .replace(new RegExp('([' + _persianCharacters + '])ه‌است($|[^' + _persianCharacters + '])', 'g'), '$1ه است$2') // «دان» handled separately: its «ی» suffix collides with the verb «دانی»/میدانی .replace( new RegExp('(^|[^' + _persianCharacters + '])(می|نمی) +(دان)(م|د|یم|ید|ند)($|[^' + _persianCharacters + '])', 'g'), '$1$2\u200c$3$4$5' ) .replace(/(\s)(می|نمی) +توان/g, '$1$2\u200cتوان') // «ها» and «ترین» always join with a ZWNJ, never a full space .replace(/ ها([\]\.،\:»\)\s]|\'{2,3}|\={2,})/g, '\u200cها$1') .replace(/ ها(ی|یی|یم|یت|یش|ی?مان|ی?تان|ی?شان)([\]\.،\:»\)\s])/g, '\u200cها$1$2') .replace(/ ترین([\]\.،\:»\)\s]|\'{2,3}|\={2,})/g, '\u200cترین$1') // a few specific words that would otherwise be caught by the general rules above .replace(new RegExp('می\u200cگوی($|[^' + _persianCharacters + '\u200c])', 'g'), 'میگوی$1') // میگوی دریایی .replace(new RegExp('می\u200cدوی($|[^' + _persianCharacters + '\u200c])', 'g'), 'میدوی$1'); // میدوی (ابهام‌زدایی) }; /** * Run a text transform on the user's selection if they have one, or on the whole page * otherwise. General-purpose - not specific to any one action - so the same one-button-does- * both pattern can be reused for other linguistically-risky transforms later, not just * smartZwnj below. * @param {object} editor The script helpers for the page. * @param {function} transform A function that takes a string and returns the transformed string. */ var _runOnSelectionOrWholePage = function(editor, transform) { var box = $('#wpTextbox1').get(0); if (box && box.selectionStart !== box.selectionEnd) { editor.replaceSelection(transform); } else { editor.set(transform(editor.get())); } }; /** * Join می/نمی verb prefixes (and a few common suffixes) to what follows with a ZWNJ instead of * a full space, using curated Persian verb-stem lists so it only fires on real verbs. * * Runs on your selection if you have one, or the whole page if you don't (see * _runOnSelectionOrWholePage above) - deliberately your choice rather than always-automatic, * for two separate reasons: * * 1. Even with the word lists, «می» on its own is also the classical word for "wine" (می ناب، * می‌گلگون، ساقی می ده…), and Wikisource hosts a lot of exactly the poetry where that comes * up. The word-list approach makes this far safer than a blind "می" + space replace, but it * can't tell «می» the prefix from «می» the noun by context alone. * 2. More fundamentally: on Wikisource, "می شود" written with a full space isn't necessarily an * error to begin with. Older typesetting predates the ZWNJ/half-space convention entirely, * so a full space between می and the verb may be exactly what the original book printed, not * an OCR artifact - "fixing" it would mean silently changing the source's actual typesetting * to match a modern convention it never used, which the whole point of a transcription * project is to avoid. * * Given both of these, review before running this on a whole page - it's not something to * reach for by default the way cleanup-ocr's encoding-level fixes are. * @param {object} editor The script helpers for the page. */ var smartZwnj = function(editor) { _runOnSelectionOrWholePage(editor, _applyZwnj); }; }); // </nowiki> t1saqvpte9qykkos8a1bkkw1czr7bbn برگه:تاریخ سی ساله ایران - بیژن جزنی (جلد اول).pdf/۷۱ 104 95540 299773 299534 2026-09-20T15:43:36Z Hanooz 17889 299773 proofread-page text/x-wiki <noinclude><pagequality level="1" user="Hanooz" /> {{سصم|||تاریخ سی ساله ایران}}</noinclude>دوم - جبهه ملی و مصدق در جنبش ملی کردن نفت محمد مصدق در یک خانواده اشرافی بزرگ شده و در فرانسه در رشته حقوق دکترا گرفته بود. مصدق پس از بازگشت به ایران تمایلات ملی از خود نشان داد و در پایان جنگ جهانی اول در جریان قرار داد ۱۹۱۹ چهره او آشکار شد ولی در این دوره جناح، ملیون چهره های سرشناس تری از مصدق دارد. در کودتای ۱۲۹۹ هنگامی که سید ضیاالدین نخست وزیر میشود و دستور بازداشت رجال قاجار را صادر میکند، مصدق والی فارس بود و هنگامی که به تهران احضار شد در اصفهان مسیر خود را عوض کرد و به چهار محال بختیاری رفت. پس از سقوط کابینه صد روزه و تشکیل کابینه قوام، مصدق پست وزارت دارائی را بعهده گرفت. با توجه به موضعی که ملیون در مقابل جنبشهای انقلابی آن دوره داشتند مصدق در مقابل جنبش<noinclude></noinclude> tqjwk2zldwonvawnbfn08zp0lqwtggx 299775 299773 2026-09-20T18:25:42Z Hanooz 17889 /* نمونه‌خوانی‌شده */ 299775 proofread-page text/x-wiki <noinclude><pagequality level="3" user="Hanooz" />{{dhr|12em}}</noinclude>{{زیرخط|'''''دوم - جبهه ملی و مصدق در جنبش ملی کردن نفت'''''}} محمد مصدق در یک خانواده اشرافی بزرگ شده و در فرانسه در رشته حقوق دکترا گرفته بود. مصدق پس از بازگشت به ایران تمایلات ملی از خود نشان داد و در پایان {{w|جنگ جهانی اول}} در جریان {{w|قرارداد ۱۹۱۹}} چهرهٔ او آشکار شد ولی در این دورهٔ جناح ملیون چهره های سرشناس تری از مصدق دارد. در {{w|کودتای ۳ اسفند ۱۲۹۹|کودتای ۱۲۹۹}} هنگامی که {{w|سید ضیاءالدین طباطبایی|سید ضیاالدین}} نخست وزیر میشود و دستور بازداشت رجال قاجار را صادر میکند، مصدق والی فارس بود و هنگامی که به تهران احضار شد در اصفهان مسیر خود را عوض کرد و به چهار محال بختیاری رفت. پس از سقوط {{w|کابینه سیاه|کابینه صد روزه}} و تشکیل کابینه قوام، مصدق پست وزارت دارائی را بعهده گرفت. با توجه به موضعی که ملیون در مقابل جنبشهای انقلابی آن دوره داشتند مصدق در مقابل جنبش<noinclude></noinclude> rvb35wz5g7bvp01zn47odd6zic47vjh برگه:تاریخ سی ساله ایران - بیژن جزنی (جلد اول).pdf/۷۲ 104 95541 299776 299535 2026-09-20T18:35:06Z Hanooz 17889 /* نمونه‌خوانی‌شده */ 299776 proofread-page text/x-wiki <noinclude><pagequality level="3" user="Hanooz" />{{سرصفحه‌مستمر|۷۰||تاریخ سی ساله ایران}}</noinclude>جنگل و قیام {{w|محمدتقی پسیان|کلنل}} و مانند آن روشی اعتدالی ولی محافظه کارانه داشت (در مقابل رضاخان، ملیون مدتها روش محافظه کارانه داشتند.) {{w|سید حسن مدرس|مدرس}} که سرشناس ترین چهرهٔ این دوره بود در ابتدای امر سعی کرد از تضادهای رضا خان با دربار استفاده کند. ملیون نمی توانستند عواقب سیاست خود را پیش بینی کنند و موقعی واقعیت را درک کردند که دیگر دور کردن رضا خان از قدرت، در توانائی آنان نبود. مدرس و همکارانش میخواستند رضا خان را با "پولیتیک" از میدان بدر کنند ولی دستهایی که رضاخان را روی کار آورده بودند در استفاده از مانورهای دیپلماتیک و برپا کردن صحنه های فریبنده بر شاگردان مکتب دیپلماسی "{{w|میرزا علی‌اصغر اتابک|امین السطان}}" و "{{w|حسین پیرنیا (مؤتمن‌الملک)|موتمن الملک}}"<ref>مقصود یکی کردن امین السطان با موتمن الملک نیست، بلکه توجه دادن به روشهای سیاسی و دیپلماسی کهنه و فرسوده‌ای بود که رجال آن دوره در هر سنی که بودند بکار میبستند. آثار این "دلقک بازی" تا سالهای پس از شهریور ۲۰ نیز در {{کذا|چهان|جهان}} سیاسی دیده میشود.</ref> استاد تر بودند. با همین معیارها بود که مصدق در مجلس سیاستمدارانه به مخالفت با سلطنت رضا خان میپردازد و اظهار میدارد که چون رضاخان (که سردار سپه و نخست وزیر بود) مرد لایق و کاردانی است و با نشستن او بر جایگاه سلطنت، ملت ایران نمیتواند از این لیاقت بهره مند شود. بهتر است رضاخان شاه شود و در همان سمتهای خود بماند سرانجام رضاخان حاکم شد و دیکتاتوری خود را برقرار کرد. مصدق خانه نشین شد و منتظر حوادث ماند. هنگامیکه متفقین وارد ایران شدند و رضا شاه سقوط کرد مصدق بیش از شصت سال داشت و این امتیاز را بر دیگران<noinclude>{{خطکش}} {{پانویس}}</noinclude> hknu5p9whjyyfii7mmbsis0q8qkc0xr برگه:تاریخ سی ساله ایران - بیژن جزنی (جلد اول).pdf/۷۳ 104 95542 299777 299536 2026-09-20T19:11:07Z Hanooz 17889 /* نمونه‌خوانی‌شده */ 299777 proofread-page text/x-wiki <noinclude><pagequality level="3" user="Hanooz" /> {{سصم|||تاریخ سی ساله ایران}}</noinclude>داشت که از دوره "{{w|مرتضی‌قلی صنیع‌الدوله|صنیع الدوله}}" و "موتمن الملک" تا مدرس در صف ملیون مبارزه کرده بود. برای مردم تهران بخصوص، برای قشرهای بازاری و کسبه که قدیمی تر بودند. (تهران در دوره رضاشاه رشد زیادی کرده بود و جمعیت زیادی از شهرستانها به تهران آمده بودند) باین ترتیب پس از شهریور ۲۰ مصدق دوباره پا به میدان گذاشت. در دوره چهاردهم مجلس شورای ملی مصدق از تهران نامزد شد و بعنوان اولین نماینده تهران انتخاب شد. در مجلس عده ای از نمایندگان منفرد و نمایندگان عضو حزب ایران گرد او جمع شده بودند و مصدق سخنگوی آنها بود. اولین مساله ای که در مجلس چهاردهم موضع مصدق را روشن کرد اعتراض به اعتبار نامهٔ سید ضیا الدین طباطبائی بود که مصدق طی سخنرانی بجای خود پرده از کودتای ۱۲۹۹ و نقش سید ضیا در آن برداشت. بر سر این مساله فراکسیون حزب توده، در کنار مصدق بود ولی مصدق مانورهائی در مجلس داد تا جدائی خود را از این فراکسیون آشکار سازد. در مجلس چهاردهم، امتیاز نامهٔ "استاندارد اویل کمپانی" مطرح شد که مصدق با آن مخالفت ورزید.<ref>سخنرانی معروف مصدق در مورد امتیازنامه استاندارد اویل کمپانی در ۷ آبان ۱۳۲۳ در مجلس {{کذا|چهاردم|چهاردهم}} انجام گرفته</ref> و نه فقط بر علیه آن بلکه بر علیه {{w|شرکت نفت ایران و انگلیس|شرکت نفت ایران و انگلیس}} و امتیاز نفت جنوب نیز صحبت کرد. هنوز مدتی از این ماجرا نگذشته بود که مساله امتیاز نفت شمال پیش آمد. در اینجا مصدق به روش خود ادامه داد. اما فراکسیون خود ادامه داد. اما فراکسیون حزب توده خطائی بزرگ مرتکب شد و از امتیاز نفت شمال دربست حمایت کرد. در اینجا مصدق و حزب توده در<noinclude>{{خطکش}} {{پانویس}}</noinclude> mv0ggsdj74ml7x9zubtccyd4kluvcoq 299778 299777 2026-09-20T19:11:25Z Hanooz 17889 299778 proofread-page text/x-wiki <noinclude><pagequality level="3" user="Hanooz" />{{سرصفحه‌مستمر|تاریخ سی ساله ایران||۷۱}}</noinclude>داشت که از دوره "{{w|مرتضی‌قلی صنیع‌الدوله|صنیع الدوله}}" و "موتمن الملک" تا مدرس در صف ملیون مبارزه کرده بود. برای مردم تهران بخصوص، برای قشرهای بازاری و کسبه که قدیمی تر بودند. (تهران در دوره رضاشاه رشد زیادی کرده بود و جمعیت زیادی از شهرستانها به تهران آمده بودند) باین ترتیب پس از شهریور ۲۰ مصدق دوباره پا به میدان گذاشت. در دوره چهاردهم مجلس شورای ملی مصدق از تهران نامزد شد و بعنوان اولین نماینده تهران انتخاب شد. در مجلس عده ای از نمایندگان منفرد و نمایندگان عضو حزب ایران گرد او جمع شده بودند و مصدق سخنگوی آنها بود. اولین مساله ای که در مجلس چهاردهم موضع مصدق را روشن کرد اعتراض به اعتبار نامهٔ سید ضیا الدین طباطبائی بود که مصدق طی سخنرانی بجای خود پرده از کودتای ۱۲۹۹ و نقش سید ضیا در آن برداشت. بر سر این مساله فراکسیون حزب توده، در کنار مصدق بود ولی مصدق مانورهائی در مجلس داد تا جدائی خود را از این فراکسیون آشکار سازد. در مجلس چهاردهم، امتیاز نامهٔ "استاندارد اویل کمپانی" مطرح شد که مصدق با آن مخالفت ورزید.<ref>سخنرانی معروف مصدق در مورد امتیازنامه استاندارد اویل کمپانی در ۷ آبان ۱۳۲۳ در مجلس {{کذا|چهاردم|چهاردهم}} انجام گرفته</ref> و نه فقط بر علیه آن بلکه بر علیه {{w|شرکت نفت ایران و انگلیس|شرکت نفت ایران و انگلیس}} و امتیاز نفت جنوب نیز صحبت کرد. هنوز مدتی از این ماجرا نگذشته بود که مساله امتیاز نفت شمال پیش آمد. در اینجا مصدق به روش خود ادامه داد. اما فراکسیون خود ادامه داد. اما فراکسیون حزب توده خطائی بزرگ مرتکب شد و از امتیاز نفت شمال دربست حمایت کرد. در اینجا مصدق و حزب توده در<noinclude>{{خطکش}} {{پانویس}}</noinclude> psj5bqzx5wp5iz57k3qnaobiymhydld برگه:تاریخ سی ساله ایران - بیژن جزنی (جلد اول).pdf/۷۴ 104 95543 299779 299537 2026-09-20T19:17:54Z Hanooz 17889 /* نمونه‌خوانی‌شده */ 299779 proofread-page text/x-wiki <noinclude><pagequality level="3" user="Hanooz" />{{سرصفحه‌مستمر|۷۲||تاریخ سی ساله ایران}}</noinclude>مقابل هم قرار گرفتند. ولی مصدق برای اینکه مخالفتش با قرارداد حمل بر مخالفت با شوروی که در مقابل آلمان هیتلری میجنگید نشود، طی نامه ای به سفیر شوروی در ایران موضع خود را درباره نفت روشن کرد. مصدق در این نامه پس از تشریح رابطه ایران با شوروی و تائید فعالیت یعنی همزیستی و کمک برادرانه شوروی جوان به تهران، صرفنظر کردن از کلیه مطالبات دولت تزاری و الغای قراردادهای یک جانبه، به دولت شوروی توصیه میکرد که پیشنهاد امتیاز نفت شمال را پس بگیرد و بجای آن پیشنهادی برای کمک به دولت ایران برای استخراج نفت در شمال بدهد. فشردهٔ پیشنهاد مصدق این است که دولت شوروی وام برای (که همه از شوروی تهیه شود.) با این وام شوروی نفت شمال استخراج و انحصراً به نرخ مبادله (بنابر بازار روز) به شوروی فروخته میشود. وام شوروی و بهرهٔ عادلانه آن از این طریق مستهلک خواهد شد. این پیشنهادها که امروز برای دولت شوروی حداکثر چیزی است که ممکن است در یک کشور به آن برسد (با توجه به انحصار فروش محصول در شوروی) و در اغلب موارد بدون توجه به ماهیت رژیم طرف معامله خود قراردادهائی به مراتب سودمندتر از نظر کشور طرف قرارداد (مثلاً ایران با مصر و مانند آن) امضاء می کند، با بی اعتنائی شوروی ها روبرو شد و انعکاس {{کذا|ایادی|زیادی}} هم نیافت. موضع مصدق در ایران در این مورد نشان دهندهٔ موضع ملی اوست. پس از تشکیل فرقهٔ دمکرات، مصدق خواسته های عمومی فرقه را در زمینه زبان مادری و مصرف مالیاتها و تشکیل انجمن ایالتی با روح قانون اساسی مطابق دانسته، خواستار رسیدگی به آنها شد. ولی هنگامیکه فرقه دمکرات حکومت را در دست گرفت و خودمختاری اعلام کرد و "توافق قوام"<noinclude></noinclude> g14enpsmfyl2ll6hv13cko0cdawcr08 ویکی‌نبشته:نام‌های پدیدآورندگان 4 95582 299771 299747 2026-09-20T14:59:16Z Hanooz 17889 299771 wikitext text/x-wiki {{process header | title = نام‌های پدیدآورندگان | section = | previous = [[ویکی‌نبشته:جستارها]] | next = | shortcut = | notes = خلاصه‌ای کوتاه از دیدگاه‌ها دربارهٔ نام پدیدآورندگان، قراردادهای نام‌گذاری، نام‌های مستعار و استفاده از حروف اول نام. }}{{جستار}} تعیین نام درستی که باید برای صفحهٔ یک پدیدآورنده در فضای نام پدیدآورنده به کار برد، در برخی موارد که پدیدآورنده‌ای با نام‌های متفاوتی شناخته می‌شود — خواه با نام مستعار یا با گونه‌های مختلفی از نام واقعی‌اش — می‌تواند چندان ساده نباشد. هیچ سیاست ثابتی دربارهٔ اینکه کدام نام باید استفاده شود وجود ندارد. در حال حاضر، به‌کارگیری قراردادها و شیوه‌های گوناگون به این معناست که در فضای پدیدآورندگانِ ویکی‌نبشته یکدستی وجود ندارد. این صفحه برخی از دیدگاه‌ها را دربارهٔ استفاده از نام پدیدآورندگان در ویکی‌نبشته خلاصه می‌کند. این بحث در نهایت به تقابل میان قراردادهای استاندارد کتابخانه‌ای و مسائل مرتب‌سازی، در برابر شهرت عمومی و خواست پدیدآورنده باز می‌گردد. ==نام‌های کامل== استفاده از حروف اول نام در نام یک پدیدآورنده گاهی در ویکی‌نبشته مورد مخالفت قرار می‌گیرد. استفاده از نام کامل پدیدآورندگان استاندارد شناخته‌شدهٔ بین‌المللی برای کتابخانه‌هاست. کتابخانه‌ها برای فهرست‌نویسیِ آثار خود باید تعداد زیادی از پدیدآورندگان را رده‌بندی کنند. اگر از حروف اول نام استفاده شود، ممکن است باعث رده‌بندی نادرست پدیدآورندگان شود (در مورد ویکی‌نبشته، ممکن است در صفحات رده و در [[ویکی‌نبشته:پدیدآورندگان]] به ترتیب نادرست فهرست شوند). همچنین ممکن است در میان همهٔ پدیدآورندگان، نام‌ها و حروف اول تکراری وجود داشته باشد که سردرگمی غیرضروری ایجاد می‌کند. در این مورد اخیر، ویکی‌نبشته می‌تواند از صفحات ابهام‌زدایی استفاده کند، هرچند این کار ممکن است غیرضروری باشد یا تنها به‌عنوان روشی پشتیبان، در کنار استفاده از نام درست، ترجیح داده شود. ویکی‌پدیا از این قرارداد نام‌گذاری استفاده نمی‌کند، چون یک دانشنامه است نه یک کتابخانه، و به‌طور استاندارد حروف اول نام را در عنوان مقاله‌های خود می‌گنجاند. این موضوع می‌تواند پیونددهی میان پروژه‌ها را با مشکل مواجه کند. برخی پدیدآورندگان ممکن است ترجیح داده باشند به شیوه‌ای خاص شناخته شوند و بیشتر با همان نام معروف‌اند. برخی از کاربران ویکی‌نبشته بر این باورند که پراستفاده‌ترین نام باید در اولویت نخست قرار گیرد و سایر نام‌ها به‌صورت تغییرمسیر درآیند. کاربران دیگر معتقدند خواستهٔ پدیدآورنده دربارهٔ چگونگی شناخته‌شدنش باید محترم شمرده شود. {| class=wikitable width="70%" {{ts|mc}} ! style="width:50%;" | نام کامل ! style="width:50%;" | نام رایج |- | [[پدیدآورنده:جیمز متیو بری]] | [[پدیدآورنده:جی. ام. بری]] |- | [[پدیدآورنده:توماس استرنز الیوت]] | [[پدیدآورنده:تی. اس. الیوت]] |- | [[پدیدآورنده:کلارنس مایکل جیمز استانیسلاوس دنیس]] | [[پدیدآورنده:سی. جی. دنیس]] |- | [[پدیدآورنده:ادیت نزبیت]] | [[پدیدآورنده:ای. نزبیت]] |- | [[پدیدآورنده:ویلیام متیو فلیندرز پیتری]] | [[پدیدآورنده:فلیندرز پیتری]] |- | [[پدیدآورنده:ادلین ویرجینیا وولف]] | [[پدیدآورنده:ویرجینیا وولف]] |- | | [[پدیدآورنده:هومر]] |} ===فنی=== الگوی {{tlx|حروف اول اسم}} را می‌توان در صفحات پدیدآورندگان قرار داد تا نشان دهد در عنوان صفحه از حرف اول نام استفاده شده و چنانچه بازکردن آن حرف اول ممکن شود، صفحه باید جابه‌جا شود. این‌گونه نمایش داده می‌شود: {{حروف اول اسم|nocat=yes}} اگر صفحهٔ یک پدیدآورنده با اطلاعات نام کامل به‌روزرسانی شده، لطفاً در [[ویکی‌نبشته:دفترخانه|دفترخانه]] اطلاع دهید تا جابه‌جا شود. ==نام‌های دیگر== بسیاری از پدیدآورندگان از نام مستعار، نام قلمی، لقب یا دیگر نام‌های جعلی استفاده می‌کنند. برخی به دلایل گوناگون بیش از یک نام جایگزین به کار می‌برند. برخی پدیدآورندگان به نام مستعار خود بیش از نام کاملشان شهرت دارند. این وضعیت شرایطی مشابه با استفاده از نام‌های کامل ایجاد می‌کند. {| class=wikitable width="70%" {{ts|mc}} ! style="width:50%;" | نام کامل ! style="width:50%;" | نام رایج |- | [[پدیدآورنده:ساموئل لنگهورن کلمنز]] | rowspan="2" | [[پدیدآورنده:مارک تواین]] |- | [[پدیدآورنده:ساموئل کلمنز]] |- | [[پدیدآورنده:اریک آرتور بلر]] | [[پدیدآورنده:جورج اورول]] |} در مواردی که یک پدیدآورنده چند نام مستعار دارد، می‌توان نام‌های دیگر را به نام کامل تغییرمسیر داد. در جایی که تنها یک نام مستعار وجود دارد، انتخاب همچنان میان نام کامل و نام مورداستناد محل بحث است. ویکی‌پدیا و دیگر پروژه‌های خواهر از نام رایج استفاده می‌کنند. بااین‌حال، نام‌های مستعار مشکل مرتب‌سازی را در ویکی‌نبشته حتی بزرگ‌تر می‌کنند، چون ممکن است نام خانوادگی یکسان نباشد و این کار نام مستعار و نام واقعی را — به‌جای صرفاً نامرتب‌بودن در یک صفحه — به‌طور کامل در صفحات جداگانه قرار می‌دهد. عنوان‌ها (مانند «پاپ»، «شاه»، «لرد»، «سر» و مانند آن) نباید در عنوان صفحه استفاده شوند، حتی اگر نام رایج پدیدآورنده باشند. در این موارد، عنوان همچنان می‌تواند در الگوی پدیدآورنده به کار رود یا در توضیحات ذکر شود. {| class=wikitable width="70%" {{ts|mc}} ! style="width:50%;" | نام بدون عنوان ! style="width:50%;" | نام رایج |- | [[پدیدآورنده:جورج گوردون بایرون]] | [[پدیدآورنده:لرد بایرون]] |- | [[پدیدآورنده:پیوس دوم]] | [[پدیدآورنده:پاپ پیوس دوم]] |- | [[پدیدآورنده:ویکتوریای بریتانیا]] | [[پدیدآورنده:ملکه ویکتوریا]] |} ==یادداشت‌های تکمیلی== * به‌صورت اختیاری، نامی که در سربرگ یک اثر استفاده می‌شود می‌تواند همان نامی باشد که در خودِ اثر آمده، حتی اگر با عنوان صفحهٔ پدیدآورنده یکسان نباشد. باید از یک تغییرمسیر برای پیونددادن نام پیوندشدهٔ ویکی به صفحهٔ درست استفاده کرد. * هر نامی که برای صفحهٔ پدیدآورنده انتخاب شود، همهٔ نام‌های دیگر باید به‌صورت صفحات تغییرمسیر به صفحهٔ اصلی پدیدآورنده ساخته شوند. * هنگام ابهام‌زدایی میان دو پدیدآورنده با نام کامل یکسان با استفاده از اطلاعات درون‌پرانتز، تاریخ‌های شناخته‌شدهٔ تولد و مرگ را بر دیگر اطلاعات ترجیح دهید؛ برای نمونه Author:William Ellery Channing ('''1818-1901''')، به‌جای Author:William Ellery Channing (poet). برای چنین عنوان‌هایی، میان تاریخ‌ها از یک خط‌تیرهٔ ساده استفاده کنید. ==همچنین ببینید== *[[ویکی‌نبشته:شیوه‌نامه#صفحات پدیدآورندگان]] *[[الگو:پدیدآورنده]] *[[:رده:پدیدآورندگان با حروف اول ناشناخته]] ===نمایه‌ها=== *[[ویکی‌نبشته:پدیدآورندگان]] *[[:رده:پدیدآورندگان]] [[رده:پدیدآورندگان با حروف اول ناشناخته| ]] dafnx4wgtwlqb2m8wix6ru1kpijxzze برگه:فرهنگ جغرافیایی ایران جلد دوم.pdf/۳۲۸ 104 95583 299774 299749 2026-09-20T18:04:30Z Hanooz 17889 299774 proofread-page text/x-wiki <noinclude><pagequality level="1" user="204.18.193.198" /></noinclude>{{حس|خالی}}<noinclude><references/></noinclude> qq4jnc9gglbflvflyaz44jtydryqjk7 الگو:جستار 10 95585 299767 2026-09-20T14:34:44Z Hanooz 17889 صفحه‌ای تازه حاوی «{{ambox | image = [[پرونده:Text-x-generic_with_pencil.svg|40px]] | text = این یک [[ویکی‌نبشته:جستارها|جستار]] است؛ شامل توصیه‌ها و/یا دیدگاه‌های یک یا چند مشارکت‌کنندهٔ ویکی‌نبشته است. این صفحه یک [[ویکی‌نبشته:سیاست‌ها و رهنمودها|سیاست یا رهنمود]] '''نیست''' و ویرایشگرا...» ایجاد کرد 299767 wikitext text/x-wiki {{ambox | image = [[پرونده:Text-x-generic_with_pencil.svg|40px]] | text = این یک [[ویکی‌نبشته:جستارها|جستار]] است؛ شامل توصیه‌ها و/یا دیدگاه‌های یک یا چند مشارکت‌کنندهٔ ویکی‌نبشته است. این صفحه یک [[ویکی‌نبشته:سیاست‌ها و رهنمودها|سیاست یا رهنمود]] '''نیست''' و ویرایشگران ملزم به پیروی از آن نیستند. {{#if:{{{1|}}}{{{2|}}}{{{3|}}}{{{4|}}}{{{5|}}} | <br />{{shortcut |{{{1|}}}|{{{2|}}}|{{{3|}}}|{{{4|}}}|{{{5|}}} }} }} }}<includeonly>{{#ifeq:{{NAMESPACE}}|{{ns:4}}|{{{category|[[رده:جستارهای ویکی‌نبشته|{{PAGENAME}}]]}}}}}{{#ifeq:{{NAMESPACE}}|{{ns:2}}|{{{category|[[رده:جستارهای کاربران|{{PAGENAME}}]]}}}}}</includeonly><noinclude>{{documentation}}</noinclude> rp6gogwmrx1we78i60jvapc5az5o8o4 299768 299767 2026-09-20T14:36:15Z Hanooz 17889 ترجمه هوش مصنوعی 299768 wikitext text/x-wiki {{ambox | image = [[پرونده:Text-x-generic_with_pencil.svg|40px]] | text = این یک [[ویکی‌نبشته:جستارها|جستار]] است؛ شامل توصیه‌ها و/یا دیدگاه‌های یک یا چند مشارکت‌کنندهٔ ویکی‌نبشته است. این صفحه یک [[ویکی‌نبشته:سیاست‌ها و رهنمودها|سیاست یا رهنمود]] '''نیست''' و ویرایشگران ملزم به پیروی از آن نیستند. {{#if:{{{1|}}}{{{2|}}}{{{3|}}}{{{4|}}}{{{5|}}} | <br />{{shortcut |{{{1|}}}|{{{2|}}}|{{{3|}}}|{{{4|}}}|{{{5|}}} }} }} }}<includeonly>{{#ifeq:{{NAMESPACE}}|{{ns:4}}|{{{category|[[رده:جستارهای ویکی‌نبشته|{{PAGENAME}}]]}}}}}{{#ifeq:{{NAMESPACE}}|{{ns:2}}|{{{category|[[رده:جستارهای کاربران|{{PAGENAME}}]]}}}}}</includeonly><noinclude>{{documentation}}</noinclude> i530a4rykch1ux5x1qypcw8lagrv0n8 الگو:حروف اول اسم 10 95586 299770 2026-09-20T14:52:27Z Hanooz 17889 ترجمه هوش مصنوعی 299770 wikitext text/x-wiki {{#if: {{{justification|}}} | {{ombox |type=notice |text='''این صفحهٔ پدیدآورنده در نام موضوع خود از حروف اختصاری استفاده می‌کند. این کار قابل‌قبول دانسته شده است، زیرا:''' <br /> {{smaller block|{{{justification}}}}} }}<includeonly>{{category handler | all = [[رده:پدیدآورندگانی که حروف اختصاری نامشان نیازی به شناسایی ندارد]] | nocat = {{{nocat|}}} }}</includeonly> | {{ombox |type=content |text='''این صفحهٔ پدیدآورنده در نام موضوع خود، به‌جای نوشتن کامل هر یک از نام‌ها، از حروف اختصاری استفاده می‌کند. ''' <br /> {{smaller block|برای انتقال این صفحهٔ پدیدآورنده به عنوان کامل و درست آن، پژوهش بیشتری لازم است؛ پس از آن می‌توان این اعلان را حذف کرد.}} }}<includeonly>{{category handler | all = [[رده:پدیدآورندگان دارای حروف اختصاری شناسایی‌نشده]] | nocat = {{{nocat|}}} }}</includeonly> }}<noinclude>{{documentation}}</noinclude> 5rcpk8j5mph2k6u7653tfepst0tugdb