ویکینبشته
fawikisource
https://fa.wikisource.org/wiki/%D8%B5%D9%81%D8%AD%D9%87%D9%94_%D8%A7%D8%B5%D9%84%DB%8C
MediaWiki 1.47.0-wmf.20
first-letter
مدیا
ویژه
بحث
کاربر
بحث کاربر
ویکینبشته
بحث ویکینبشته
پرونده
بحث پرونده
مدیاویکی
بحث مدیاویکی
الگو
بحث الگو
راهنما
بحث راهنما
رده
بحث رده
درگاه
بحث درگاه
پدیدآورنده
بحث پدیدآورنده
برگه
گفتگوی برگه
فهرست
گفتگوی فهرست
TimedText
TimedText talk
پودمان
بحث پودمان
Event
Event talk
کاربر:Hanooz/common.js
2
63731
299780
299731
2026-09-20T19:46:14Z
Hanooz
17889
299780
javascript
text/javascript
window.charinsertCustom = {
"کاربر": '{{em}} {{gap}} {{sc|+}} {{sp|+}} {{xl|+}} {{c|+}} {{c|{{xl|+}}}} {{c|{{xl|{{sp|+}}}}}} {{dhr|2}}'
};
//<nowiki>
/* Cat-a-lot - changes category of multiple files */
mw.loader.using(['jquery.ui', 'mediawiki.util'], function(){
mw.loader.load('//commons.wikimedia.org/w/load.php?modules=ext.gadget.Cat-a-lot');
});
////////// Cat-a-lot user preferences //////////
window.catALotPrefs = {"watchlist":"preferences","minor":true,"editpages":true,"docleanup":false,"subcatcount":10};
////////////////////////////////////catALotEnd//
//</nowiki>
mw.loader.load('//meta.wikimedia.org/w/index.php?title=User:Krinkle/Tools/WhatLeavesHere.js&action=raw&ctype=text/javascript');
mw.loader.load('//meta.wikimedia.org/w/index.php?title=User:Indic-TechCom/Script/massMover.js&action=raw&ctype=text/javascript');
mw.loader.load('//mediawiki.org/w/index.php?title=MediaWiki:Gadget-UTCLiveClock.js&action=raw&ctype=text/javascript&smaxage=21600&maxage=86400');
mw.loader.load('//mediawiki.org/w/index.php?title=User:PerfektesChaos/js/resultListSort/r.js&action=raw&bcache=1&ctype=text/javascript');
mw.loader.load('//fa.wikipedia.org/w/index.php?title=مدیاویکی:Gadget-quickLinker.js&action=raw&ctype=text/javascript');
mw.loader.load('//fa.wikipedia.org/w/index.php?title=مدیاویکی:Gadget-Extra-Editbuttons-sisters.js&action=raw&ctype=text/javascript');
mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/Gadget-charinsert-core.js&action=raw&ctype=text/javascript');
mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/page carousel.js&action=raw&ctype=text/javascript');
mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/NopInserter.js&action=raw&ctype=text/javascript');
mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/redirectmaker.js&action=raw&ctype=text/javascript');
mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/RunningHeader.js&action=raw&ctype=text/javascript');
mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/test.js&action=raw&ctype=text/javascript');
mw.loader.load('//en.wikisource.org/w/index.php?title=MediaWiki:Gadget-TemplatePreloader.js&action=raw&ctype=text/javascript');
mw.loader.load('//en.wikisource.org/w/index.php?title=MediaWiki:Gadget-Without text.js&action=raw&ctype=text/javascript');
mw.loader.load('//en.wikisource.org/w/index.php?title=MediaWiki:Gadget-transclusion-check.js&action=raw&ctype=text/javascript');
mw.loader.load('//en.wikisource.org/w/index.php?title=MediaWiki:Gadget-HotCat.js&action=raw&ctype=text/javascript');
mw.loader.load('//en.wikisource.org/w/index.php?title=MediaWiki:Gadget-Easy LST.js&action=raw&ctype=text/javascript');
mw.loader.load('//en.wikisource.org/w/index.php?title=MediaWiki:Gadget-Fill Index.js&action=raw&ctype=text/javascript');
mw.loader.load('//en.wikisource.org/w/index.php?title=User:Inductiveload/cleanup.js&action=raw&ctype=text/javascript');
mw.loader.load('//en.wikisource.org/w/index.php?title=User:Inductiveload/quick_access.js&action=raw&ctype=text/javascript');
mw.loader.load('//en.wikisource.org/w/index.php?title=User:Xover/cleanup.js&action=raw&ctype=text/javascript');
mw.loader.load('//en.wikisource.org/w/index.php?title=User:Xover/unwrap.js&action=raw&ctype=text/javascript');
mw.loader.load('//en.wikisource.org/w/index.php?title=User:Xover/copySource.js&action=raw&ctype=text/javascript');
mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Nightdevil/indexFiller.js&action=raw&ctype=text/javascript');
5n4arv3iz62zwryzww1462l41hwl09x
299781
299780
2026-09-20T19:49:23Z
Hanooz
17889
299781
javascript
text/javascript
window.charinsertCustom = {
"کاربر": '{{em}} {{gap}} {{sc|+}} {{sp|+}} {{xl|+}} {{c|+}} {{c|{{xl|+}}}} {{c|{{xl|{{sp|+}}}}}} {{dhr|2}}'
};
//<nowiki>
/* Cat-a-lot - changes category of multiple files */
mw.loader.using(['jquery.ui', 'mediawiki.util'], function(){
mw.loader.load('//commons.wikimedia.org/w/load.php?modules=ext.gadget.Cat-a-lot');
});
////////// Cat-a-lot user preferences //////////
window.catALotPrefs = {"watchlist":"preferences","minor":true,"editpages":true,"docleanup":false,"subcatcount":10};
////////////////////////////////////catALotEnd//
//</nowiki>
mw.loader.load('//meta.wikimedia.org/w/index.php?title=User:Krinkle/Tools/WhatLeavesHere.js&action=raw&ctype=text/javascript');
mw.loader.load('//meta.wikimedia.org/w/index.php?title=User:Indic-TechCom/Script/massMover.js&action=raw&ctype=text/javascript');
mw.loader.load('//mediawiki.org/w/index.php?title=MediaWiki:Gadget-UTCLiveClock.js&action=raw&ctype=text/javascript&smaxage=21600&maxage=86400');
mw.loader.load('//mediawiki.org/w/index.php?title=User:PerfektesChaos/js/resultListSort/r.js&action=raw&bcache=1&ctype=text/javascript');
mw.loader.load('//fa.wikipedia.org/w/index.php?title=مدیاویکی:Gadget-quickLinker.js&action=raw&ctype=text/javascript');
mw.loader.load('//fa.wikipedia.org/w/index.php?title=مدیاویکی:Gadget-Extra-Editbuttons-sisters.js&action=raw&ctype=text/javascript');
mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/Gadget-charinsert-core.js&action=raw&ctype=text/javascript');
mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/page carousel.js&action=raw&ctype=text/javascript');
mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/NopInserter.js&action=raw&ctype=text/javascript');
mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/redirectmaker.js&action=raw&ctype=text/javascript');
mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/RunningHeader.js&action=raw&ctype=text/javascript');
mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Hanooz/test.js&action=raw&ctype=text/javascript');
mw.loader.load('//en.wikisource.org/w/index.php?title=MediaWiki:Gadget-TemplatePreloader.js&action=raw&ctype=text/javascript');
mw.loader.load('//en.wikisource.org/w/index.php?title=MediaWiki:Gadget-Without text.js&action=raw&ctype=text/javascript');
mw.loader.load('//en.wikisource.org/w/index.php?title=MediaWiki:Gadget-HotCat.js&action=raw&ctype=text/javascript');
mw.loader.load('//en.wikisource.org/w/index.php?title=MediaWiki:Gadget-Easy LST.js&action=raw&ctype=text/javascript');
mw.loader.load('//en.wikisource.org/w/index.php?title=MediaWiki:Gadget-Fill Index.js&action=raw&ctype=text/javascript');
mw.loader.load('//en.wikisource.org/w/index.php?title=User:Xover/unwrap.js&action=raw&ctype=text/javascript');
mw.loader.load('//fa.wikisource.org/w/index.php?title=کاربر:Nightdevil/indexFiller.js&action=raw&ctype=text/javascript');
rqal0ndesu6lolk5knvgx5p0zizpmce
کاربر:Hanooz/test.js
2
86084
299769
299758
2026-09-20T14:41:39Z
Hanooz
17889
299769
javascript
text/javascript
/*
This page defines a TemplateScript library for fa.wikisource.org (ویکینبشتهٔ فارسی).
It's not meant to be referenced directly - install it the same way as the upstream
English version described at [[Wikisource:TemplateScript]].
This is a Persian Wikisource adaptation of Pathoschild's proofreading and typography
libraries, pared down to a focused four-action toolset. It draws on the fa.wikipedia/
fa.wikisource "Persian text style improvement tools" gadget (Gadget-Extra-Editbuttons-
persianwikitools.js) for several of the character/orthography helpers, since that's the
community's own tested implementation rather than something invented for this file.
*/
/* global $, pathoschild */
/**
* TemplateScript adds configurable templates and scripts to the sidebar, and adds an example regex editor.
* @see https://meta.wikimedia.org/wiki/TemplateScript
*
* The upstream proofreading library's @update-token has been removed on purpose: Wikisource's
* maintenance bot uses that token to overwrite local copies with the vanilla English script,
* which would silently wipe out every change made here for fa.wikisource. See [[Wikisource:Tools
* and scripts]] for how the update-bot category works.
*/
// <nowiki>
$.ajax('//tools-static.wmflabs.org/meta/scripts/pathoschild.templatescript.js', { dataType:'script', cache:true }).then(function() {
/*********
** Define library
*********/
pathoschild.TemplateScript.library.define({
key: 'wikisource.fa.tools',
name: 'ابزارهای ویکینبشتهٔ فارسی',
url: '//fa.wikisource.org/wiki/راهنما:نمونهخوانی',
description: 'مجموعهای از ابزارها برای <a href="/wiki/راهنما:نمونهخوانی">نمونهخوانی آثار در فضای نام <tt>برگه:</tt></a>: پاکسازی OCR (با اصلاح نویسهها، کشیده، ارقام، جای اعراب، نیمفاصله، نویسههای نامرئی بیاثر، و نشانهگذاری) و اصلاح نیمفاصلهٔ افعال.',
categories: [
{
name: 'ابزارهای برگه',
scripts: [
{ key: 'cleanup-ocr', name: 'پاکسازی OCR', script: function(editor) { pageCleanup(editor); }, forNamespaces: 'page' },
{ key: 'smart-zwnj', name: 'اصلاح نیمفاصلهٔ افعال', script: function(editor) { smartZwnj(editor); }, forNamespaces: 'page' }
]
}
]
});
/*********
** Private methods
*********/
// Digit tables used to convert between Latin, Persian and Arabic-Indic numerals.
var _persianDigits = '۰۱۲۳۴۵۶۷۸۹';
var _arabicIndicDigits = '٠١٢٣٤٥٦٧٨٩';
var _latinDigits = '0123456789';
// Character-class tables used by the Persian-specific cleanup helpers below. Ported from the
// fa.wikipedia/fa.wikisource "Persian text style improvement tools" gadget (Gadget-Extra-
// Editbuttons-persianwikitools.js) - the community's own tested implementation of this
// normalisation, rather than something invented for this file.
var _hamza = '\u0654';
var _persianVowels = '\u064B-\u0650\u0652\u0670';
// non-standard letterforms (Arabic ي/ك, Urdu, Pashto, Uyghur presentation forms, etc.) that
// should still count as "a Persian-ish letter" for context-matching, even before
// _toStandardPersianCharacters() below has normalised them
var _similarPersianCharacters = '\u0643\uFB91\uFB90\uFB8F\uFB8E\uFEDC\uFEDB\uFEDA\uFED9\u0649\uFEEF\u064A\u06C1\u06D5\u06BE\uFEF0-\uFEF4';
var _persianCharacters = '\u0621-\u0655\u067E\u0686\u0698\u06AF\u06A9\u0643\u06AA\uFED9\uFEDA\u06CC\uFEF1\uFEF2' + _similarPersianCharacters;
/**
* Convert any mix of Latin or Arabic-Indic digits in a string to Persian digits, for display in wikitext.
* @param {string} text The text to convert.
*/
var _digitsToPersian = function(text) {
var result = '';
for (var i = 0; i < text.length; i++) {
var ch = text.charAt(i);
var idx = _latinDigits.indexOf(ch);
if (idx === -1) idx = _arabicIndicDigits.indexOf(ch);
result += (idx !== -1) ? _persianDigits.charAt(idx) : ch;
}
return result;
};
/**
* OCR/PDF presentation-form and script-variant glyphs that extraction sometimes leaves as
* literal characters instead of the standard Persian letter - a common problem specifically on
* PDF-sourced works, where each glyph can get saved as its own shaped codepoint. Keys are the
* standard letter (or letter + a real ZWNJ, for the two glyphs that visually look joined);
* values are a character class of the presentation-form/variant glyphs that should fold into it.
*/
var _persianGlyphs = {
'\u200cه': 'ﻫ',
'ی\u200c': 'ﻰﻲ',
'أ': 'ﺄﺃﺃ',
'آ': 'ﺁﺁﺂ',
'إ': 'ﺇﺈﺇ',
'ا': 'ﺍﺎ',
'ب': 'ﺏﺐﺑﺒ',
'پ': 'ﭖﭗﭘﭙ',
'ت': 'ﺕﺖﺗﺘ',
'ث': 'ﺙﺚﺛﺜ',
'ج': 'ﺝﺞﺟﺠ',
'چ': 'ﭺﭻﭼﭽ',
'ح': 'ﺡﺢﺣﺤ',
'خ': 'ﺥﺦﺧﺨ',
'د': 'ﺩﺪ',
'ذ': 'ﺫﺬ',
'ر': 'ﺭﺮ',
'ز': 'ﺯﺰ',
'ژ': 'ﮊﮋ',
'س': 'ﺱﺲﺳﺴ',
'ش': 'ﺵﺶﺷﺸ',
'ص': 'ﺹﺺﺻﺼ',
'ض': 'ﺽﺾﺿﻀ',
'ط': 'ﻁﻂﻃﻄ',
'ظ': 'ﻅﻆﻇﻈ',
'ع': 'ﻉﻊﻋﻌ',
'غ': 'ﻍﻎﻏﻐ',
'ف': 'ﻑﻒﻓﻔ',
'ق': 'ﻕﻖﻗﻘ',
'ک': 'ﮎﮏﮐﮑﻙﻚﻛﻜ',
'گ': 'ﮒﮓﮔﮕ',
'ل': 'ﻝﻞﻟﻠ',
'م': 'ﻡﻢﻣﻤ',
'ن': 'ﻥﻦﻧﻨ',
'ه': 'ﻩﻪﻫﻬ',
'هٔ': 'ﮤﮥ',
'و': 'ﻭﻮ',
'ؤ': 'ﺅﺅﺆ',
'ی': 'ﯼﯽﯾﯿﻯﻰﻱﻲﻳﻴ',
'ئ': 'ﺉﺊﺋﺌ',
'لا': 'ﻻﻼ',
'لإ': 'ﻹﻺ',
'لأ': 'ﻸﻷ',
'لآ': 'ﻵﻶ'
};
/**
* Remove stray/duplicate ZWNJ-family characters and OCR artifacts that stand in for a ZWNJ,
* without trying to guess where a NEW ZWNJ should be inserted between words - that needs the
* verb word-lists in smartZwnj() below, since blindly inserting one is what risks corrupting
* unrelated text (see that function's comment for why).
* @param {string} text The text to clean.
*/
var _cleanupZwnjArtifacts = function(text) {
return text
// a ZWNJ (or stray zero-width/BOM/bidi-mark character) with nothing before or after it
// to join or not-join has no possible effect on the rendered output - most commonly a
// ZWNJ that ended up with a plain space next to it instead of a letter
.replace(/^[\u200B-\u200D\uFEFF]+/, '')
.replace(/[\u200B-\u200D\uFEFF]+$/, '')
// a stray LRM/RLM between two Persian letters was almost always meant to be a ZWNJ
.replace(new RegExp('([' + _persianCharacters + '] *)[\u200F\u200E]+( *[' + _persianCharacters + '])', 'g'), '$1\u200c$2')
// collapse repeated invisible-joiner/mark characters down to one
.replace(/([\u200B-\u200D\uFEFF\u200E\u200F]){2,}/g, '$1')
// ¬ between two Persian letters is almost always a misrecognised ZWNJ, not a real character
.replace(new RegExp('([' + _persianCharacters + '])¬(?=[' + _persianCharacters + '])', 'g'), '$1\u200c')
// a ZWNJ has no business after a digit or most punctuation/brackets, or next to a Latin word
.replace(/([۰-۹0-9إأةؤورزژاآدذ،؛,:«»\\\/@#$٪×*()ـ\-=|ء])\u200c/g, '$1')
.replace(/[\u200B-\u200D\uFEFF]([\w])/g, '$1')
.replace(/([\w])[\u200B-\u200D\uFEFF]/g, '$1')
.replace(new RegExp('[\\u200B-\\u200D\\uFEFF]([' + _persianVowels + _arabicIndicDigits + _persianDigits + _latinDigits + _hamza + '])', 'g'), '$1')
.replace(new RegExp('([' + _arabicIndicDigits + '])[\\u200B-\\u200D\\uFEFF]', 'g'), '$1')
.replace(/[\u200B\u200C\uFEFF]([ء\n\s\[\].،«»:()؛؟?;$!@\-=+\\|])/g, '$1')
.replace(/([\n\s\[.،«»:()؛؟?;$!@\-=+\\|])[\u200B-\u200D\uFEFF]/g, '$1')
.replace(/[\u200B-\u200D\uFEFF](\]\][\s\n])/g, '$1')
.replace(/([\n\s]\[\[)[\u200B-\u200D\uFEFF]/g, '$1');
};
/**
* Fold OCR/PDF presentation-form glyphs and non-Persian script variants (Arabic ي/ك, Urdu,
* Pashto, Uyghur, Kurdish) back to standard Persian characters, and canonicalise the hamza-
* bearing forms per ISIRI 6219 (Iran's national character-encoding standard).
* @param {string} text The text to clean.
*/
var _toStandardPersianCharacters = function(text) {
for (var standard in _persianGlyphs) {
if (_persianGlyphs.hasOwnProperty(standard)) {
text = text.replace(new RegExp('[' + _persianGlyphs[standard] + ']', 'g'), standard);
}
}
return _cleanupZwnjArtifacts(text) // needed because of the two ZWNJ-bearing keys above
.replace(/ك/g, 'ک') // Arabic
.replace(/ڪ/g, 'ک') // Urdu
.replace(/ﻙ/g, 'ک') // Pushtu
.replace(/ﻚ/g, 'ک') // Uyghur
.replace(/ي/g, 'ی') // Arabic
.replace(/ى/g, 'ی') // Urdu
.replace(/ے/g, 'ی') // Urdu
.replace(/ۍ/g, 'ی') // Pushtu
.replace(/ې/g, 'ی') // Uyghur
.replace(/ہ/g, 'ه') // Urdu
.replace(/ە/g, 'ه\u200c') // Kurdish
.replace(/ھ/g, 'ه') // Kurdish
.replace(/أ/g, 'أ') // canonical alef+hamza form per ISIRI 6219
.replace(/آ/g, 'آ');
};
/**
* Strip tatweel/kashide (ـ, U+0640) characters from a string. Tatweel is a pure justification
* stretch-mark used in decorative Persian typesetting (common on title pages); OCR frequently
* captures it as literal characters, but it has no place in a wikilink target or in wikitext.
* @param {string} text The text to clean.
*/
var _stripTatweel = function(text) {
return text.replace(/\u0640+/g, '');
};
/*********
** Script methods
*********/
/**
* Clean up OCR errors in the text, and push <noinclude> content at the top
* & bottom of the page into the header & footer boxes respectively.
* @param {object} editor The script helpers for the page.
*/
var pageCleanup = function(editor) {
// Strip characters that have no effect on the rendered output at all, before anything
// else runs: control characters, BOM and soft hyphen have no legitimate place in body
// text; \r is a Windows line-ending artifact once \n is already there; a run of exotic
// Unicode space characters (NBSP, thin space, ideographic space, etc.) right before a
// line break just disappears, a tab/NBSP right after one disappears too (plain space is
// deliberately left alone there, since a leading space triggers MediaWiki's <pre>
// formatting), and any exotic space left in running text becomes a plain space. Ported
// from the fa.wikipedia/fa.wikisource gadget's applyOrthography(), and cross-checked
// against Virastar (github.com/aziz/virastar and its ports, a widely used standalone
// Persian text cleaner) - both do this same normalisation as a first step.
var text = editor.get()
.replace(/\r/g, '')
.replace(/[\x00-\x08\x0B\x0C\x0E-\x1F\x7F-\x9F\uFEFF\u00AD\u0085]+/g, '')
.replace(/[ \xA0\u1680\u180E\u2000-\u200A\u202F\u205F\u3000]+\n/g, '\n')
.replace(/\n[\t\u00A0]+/g, '\n')
.replace(/[\xA0\u1680\u180E\u2000-\u200A\u202F\u205F\u3000]/g, ' ');
// Normalise the raw text next: fold OCR/PDF presentation-form glyphs and non-Persian
// letter variants to standard Persian characters (this also mops up stray ZWNJ-family
// characters left over from OCR - see _cleanupZwnjArtifacts, including a ZWNJ that's now
// got a plain space next to it instead of a letter, which can no longer do anything),
// strip decorative tatweel, convert Latin/Arabic-Indic digits to Persian, and fix vowel-
// mark (harakat) placement - a mark separated from its letter by a stray space moves back
// next to the letter, and duplicate marks on the same letter collapse to one. The digit
// conversion runs unconditionally on the whole page - if a work legitimately needs
// Western digits somewhere (an ISBN, a citation in another language), those get
// converted too; there's no separate opt-in step for that anymore.
text = _toStandardPersianCharacters(text);
text = _stripTatweel(text);
text = _digitsToPersian(text);
text = text.replace(new RegExp('([' + _persianCharacters + _persianVowels + _hamza + '])(\\s)([' + _persianVowels + _hamza + '])', 'g'), '$1$3$2');
text = text.replace(new RegExp('([' + _persianVowels + _hamza + ']){2,}', 'g'), '$1');
editor.set(text);
// Strip leading lines that are pure page-number "running header" junk - the digits-only
// case of the en.wikisource WsCleanup gadget's running-header patterns. Dropped that
// gadget's other pattern (digits plus an ALL-CAPS title fragment) since Persian has no
// letter case to signal "this is header text, not body text" the way English capitals
// do. Caveat carried over from the source: a work with numbered stanzas/verses starting
// right at the top of a page could have its number line mistaken for header junk here -
// worth a glance after running this on poetry.
(function() {
var lines = editor.get().split(/\r?\n/);
var pageNumberOnly = /^[\s0-9۰-۹٠-٩.,\-–—]+$/;
var start = 0;
for (var li = 0; li < lines.length; li++) {
if (lines[li].trim().length === 0 || (pageNumberOnly.test(lines[li]) && lines[li].trim().length > 0)) {
start++;
continue;
}
break;
}
if (start > 0) {
editor.set(lines.slice(start).join('\n'));
}
})();
// push <noinclude> content at the top & bottom into the header & footer
if (editor.get().match(/^<noinclude\>/)) {
var text = editor.get();
var e = text.indexOf("</noinclude>");
$('#wpHeaderTextbox').val(function(i, val) {
return $.trim(val + "\n" + text.substr(11, e-11).replace(/^\s+|\s+$/g, ''));
});
editor.set(text.substr(e+12));
}
if (editor.get().match(/<\/noinclude\>$/)) {
var text = editor.get();
var s = text.lastIndexOf("<noinclude>");
$('#wpFooterTextbox').val(function(i, val) {
return $.trim(text.substr(s+11, text.length-s-11-12).replace(/^\s+|\s+$/g, '') + "\n" + val);
});
editor.set(text.substr(0, s));
}
// clean up text
editor
// remove trailing spaces at the end of each line
.replace(/ +\n/g, '\n')
// remove trailing whitespace preceding a hard line break
.replace(/ +<br *\/?>/g, '<br />')
// remove trailing whitespace and numerals at the end of page text - matches
// Latin, Persian and Arabic-Indic digits, since page-number artifacts at the
// bottom of a scanned fa.wikisource page can OCR as any of the three
.replace(/[\s0-9۰-۹٠-٩]+$/g, '')
// remove trailing spaces at the end of refs
.replace(/ +<\/ref>/g, '</ref>')
// remove trailing spaces at the end of template calls
.replace(/ +}}/g, '}}')
// convert double-hyphen to mdash (avoiding breaking HTML comment syntax)
.replace(/([^\!])--([^>])/g, '$1—$2')
// remove spacing around mdash, but only if it has spaces on both sides
.replace(/ +— +/g, '—')
// "Digitized by Google" scan-watermark text and stray OCR of the word "Google" -
// the watermark itself is always in English regardless of the scanned work's
// language, so this is just as relevant to a Google Books-sourced Persian scan
.replace(/\s?D[il]g[il]t[il][sz][eco]d\s+by[^\n]*\s+([6G][Oo0]{2}g[lI]e)?/g, '')
.replace(/\bG[oO0]{2}gle\b/g, '')
// highly suspicious OCR noise characters
.replace(/[■•]/g, '')
// a line containing nothing but a stray punctuation mark is likely scan noise
.replace(/^[.,^،؛]$/gm, '')
// underscores (often OCR of underlined text or a form field) aren't meaningful in wikitext
.replace(/_/g, ' ')
// number ranges get an en dash, not a hyphen, per fa.wikisource's وپ:خط تیره
.replace(/([۰-۹]+)[ \t]?-[ \t]?(?=[۰-۹])/g, '$1–')
// <section begin=x/> needs to be <section begin="x"/> to parse
.replace(/<section (begin|end)=(\w[^/]+)\/>/g, '<section $1="$2"/>')
// a <ref> shouldn't have a space glued before it
.replace(/ <ref/g, '<ref')
// {{hws}}/{{hwe}} shorthand for a word split across a page boundary, expanded to the
// full template names (same English names used across Wikisource language versions,
// fa.wikisource included, for this kind of utility template)
.replace(/{{hws\|/g, '{{hyphenated word start|')
.replace(/{{hwe\|/g, '{{hyphenated word end|')
// {{c|...}} shorthand for {{center|...}}, if used
.replace(/{{c\|/g, '{{center|');
// clean up pages if they don't have <poem>
if (!editor.contains('<poem>') && !editor.contains('{{' + 'ppoem')) {
editor
// remove single line breaks; preserve multiple.
// but not if there's a tag, template or table syntax either side of the line break
.replace(/([^>}\|\n])\n([^:#\*<{\|\n])/g, '$1 $2')
// collapse sequences of spaces into a single space
.replace(/ +/g, ' ');
}
// more page cleanup
editor
// dump spurious hard breaks at the end of paragraphs
.replace(/<br *\/?>\n\n/g, '\n\n')
// remove unwanted spaces before punctuation marks - both Latin (;:?!,.)) and
// Persian (،؛؟) forms, since either can turn up depending on the OCR engine
.replace(/ ([;:\?!,.)،؛؟])/g, '$1')
// and no space after an opening parenthesis
.replace(/\( +/g, '(')
// convert ASCII look-alikes to proper Persian punctuation when they follow Persian
// text. Properly printed Persian sources always used ؟/؛/، - an ASCII ?/;/, next to
// Persian text is an OCR/encoding artifact to restore, not an authorial style choice
.replace(new RegExp('([' + _persianCharacters + '])\\?', 'g'), '$1؟')
.replace(new RegExp('([' + _persianCharacters + ']);', 'g'), '$1؛')
.replace(new RegExp('([' + _persianCharacters + '])(\\]\\]|»|)\\,', 'g'), '$1$2،')
// ensure a space after punctuation when it's glued directly to the next word...
.replace(/([;:\?!,.)،؛؟])([^\s0-9۰-۹٠-٩\n}|"«»'’”])/g, '$1 $2')
// ...but consecutive punctuation marks (or end of line) don't get a space between them
.replace(/([;:\?!,.)،؛؟]) +([\n;:\?!,.)،؛؟\]]|$)/g, '$1$2')
// a double period after a Persian letter is almost always meant to be one
.replace(new RegExp('([' + _persianCharacters + '])\\.\\. (?=[' + _persianCharacters + '])', 'g'), '$1. ')
// three-or-more dots after Persian text is an ellipsis, not literal periods
.replace(new RegExp('([' + _persianCharacters + '])( *)\\.{3,}', 'g'), '$1$2…')
// canonicalise ۀ / هٴ / هٓ to ISIRI 6219's ه + combining hamza above (هٔ) - this is a
// same-character encoding choice, not a wording change, so it's safe to automate
.replace(/[ۂۀ]/g, 'هٔ')
.replace(/هٴ/g, 'هٔ')
.replace(/هٓ/g, 'هٔ')
// unicodify
.replace(/—/g, '—')
.replace(/–/g, '–')
.replace(/"/g, '"')
.replace(/<center>\s*([.\n]*?)\s*<\/center>/g, '{{center|$1}}');
// Run the invisible-ZWNJ-artifact cleanup once more: the punctuation and digit changes
// above can newly strand a ZWNJ next to something it can no longer affect (e.g. a ZWNJ
// that's now right before a converted Persian digit or a newly-inserted punctuation mark).
editor.set(_cleanupZwnjArtifacts(editor.get()));
};
// Verb stems used by smartZwnj() below to decide when "می"/"نمی" is a verb prefix that should
// join to what follows with a ZWNJ, rather than something else (most importantly, the classical
// word «می» meaning "wine" - extremely common in the poetry Wikisource hosts a lot of, e.g.
// Hafez or Khayyam). Matching only against a real verb stem from this list, with a real person/
// tense suffix, is what makes this safe enough to automate at all; a blind "می" + space match
// would wrongly rewrite «می ناب» (pure wine) in a ghazal. Ported verbatim from the fa.wikipedia/
// fa.wikisource "Persian text style improvement tools" gadget, since retyping ~500 verb stems by
// hand is exactly how new bugs get introduced.
var _persianPastVerbs = '(' +
'ارزید|افتاد|افراشت|افروخت|افزود|افسرد|افشاند|افکند|انباشت|انجامید|انداخت|اندوخت|اندود|اندیشید|انگاشت|انگیخت|انگیزاند|اوباشت|ایستاد' +
'|آراست|آراماند|آرامید|آرمید|آزرد|آزمود|آسود|آشامید|آشفت|آشوبید|آغازید|آغشت|آفرید|آکند|آگند|آلود|آمد|آمرزید|آموخت|آموزاند' +
'|آمیخت|آهیخت|آورد|آویخت|باخت|باراند|بارید|بافت|بالید|باوراند|بایست|بخشود|بخشید|برازید|برد|برید|بست|بسود|بسیجید|بلعید' +
'|بود|بوسید|بویید|بیخت|پاشاند|پاشید|پالود|پایید|پخت|پذیراند|پذیرفت|پراکند|پراند|پرداخت|پرستید|پرسید|پرهیزید|پروراند|پرورد|پرید' +
'|پژمرد|پژوهید|پسندید|پلاسید|پلکید|پناهید|پنداشت|پوسید|پوشاند|پوشید|پویید|پیچاند|پیچانید|پیچید|پیراست|پیمود|پیوست|تاباند|تابید|تاخت' +
'|تاراند|تازاند|تازید|تافت|تپاند|تپید|تراشاند|تراشید|تراوید|ترساند|ترسید|ترشید|ترکاند|ترکید|تکاند|تکانید|تنید|توانست|جَست|جُست' +
'|جست|جنباند|جنبید|جنگید|جهاند|جهید|جوشاند|جوشید|جوید|چاپید|چایید|چپاند|چپید|چراند|چربید|چرخاند|چرخید|چرید|چسباند|چسبید' +
'|چشاند|چشید|چکاند|چکید|چلاند|چلانید|چمید|چید|خاراند|خارید|خاست|خایید|خراشاند|خراشید|خرامید|خروشید|خرید|خزید|خشکاند' +
'|خشکید|خفت|خلید|خمید|خنداند|خندانید|خندید|خواباند|خوابانید|خوابید|خواست|خواند|خوراند|خورد|خوفید|خیساند|خیسید|داد|داشت|دانست' +
'|درخشانید|درخشید|دروید|درید|دزدید|دمید|دواند|دوخت|دوشید|دوید|دید|دیدم|راند|ربود|رخشید|رساند|رسانید|رست|رَست|رُست' +
'|رسید|رشت|رفت|رُفت|رقصاند|رقصید|رمید|رنجاند|رنجید|رندید|رهاند|رهانید|رهید|روبید|روفت|رویاند|رویید|ریخت|رید|ریسید' +
'|زاد|زارید|زایید|زد|زدود|زیست|سابید|ساخت|سپارد|سپرد|سپوخت|ستاند|ستد|سترد|ستود|ستیزید|سرایید|سرشت|سرود|سرید' +
'|سزید|سفت|سگالید|سنجید|سوخت|سود|سوزاند|شاشید|شایست|شتافت|شد|شست|شکافت|شکست|شکفت|شکیفت|شگفت|شمارد|شمرد|شناخت' +
'|شناساند|شنید|شوراند|شورید|طپید|طلبید|طوفید|غارتید|غرید|غلتاند|غلتانید|غلتید|غلطاند|غلطانید|غلطید|غنود|فرستاد|فرسود|فرمود|فروخت' +
'|فریفت|فشاند|فشرد|فهماند|فهمید|قاپید|قبولاند|کاست|کاشت|کاوید|کرد|کشاند|کشانید|کشت|کشید|کفت|کفید|کند|کوبید|کوچید' +
'|کوشید|کوفت|گَزید|گُزید|گایید|گداخت|گذارد|گذاشت|گذراند|گذشت|گرازید|گرایید|گرداند|گردانید|گردید|گرفت|گروید|گریاند|گریخت|گریست' +
'|گزارد|گزید|گسارد|گستراند|گسترد|گسست|گسیخت|گشت|گشود|گفت|گمارد|گماشت|گنجاند|گنجانید|گنجید|گندید|گوارید|گوزید|لرزاند|لرزید' +
'|لغزاند|لغزید|لمباند|لمدنی|لمید|لندید|لنگید|لهید|لولید|لیسید|ماسید|مالاند|مالید|ماند|مانست|مرد|مکشید|مکید|مولید|مویید' +
'|نازید|نالید|نامید|نشاند|نشست|نکوهید|نگاشت|نگریست|نمایاند|نمود|نهاد|نهفت|نواخت|نوردید|نوشاند|نوشت|نوشید|نیوشید|هراسید|هشت' +
'|ورزید|وزاند|وزید|یارست|یازید|یافت' +
')';
var _persianPresentVerbs = '(' +
'ارز|افت|افراز|افروز|افزا|افزای|افسر|افشان|افکن|انبار|انباز|انجام|انداز|اندای|اندوز|اندیش|انگار|انگیز|انگیزان' +
'|اوبار|ایست|آرا|آرام|آرامان|آرای|آزار|آزما|آزمای|آسا|آسای|آشام|آشوب|آغار|آغاز|آفرین|آکن|آگن|آلا|آلای' +
'|آمرز|آموز|آموزان|آمیز|آهنج|آور|آویز|آی|بار|باران|باز|باش|باف|بال|باوران|بای|باید|بخش|بخشا|بخشای' +
'|بر|بَر|بُر|براز|بساو|بسیج|بلع|بند|بو|بوس|بوی|بیز|بین|پا|پاش|پاشان|پالا|پالای|پذیر|پذیران' +
'|پر|پراکن|پران|پرداز|پرس|پرست|پرهیز|پرور|پروران|پز|پژمر|پژوه|پسند|پلاس|پلک|پناه|پندار|پوس|پوش|پوشان' +
'|پوی|پیچ|پیچان|پیرا|پیرای|پیما|پیمای|پیوند|تاب|تابان|تاران|تاز|تازان|تپ|تپان|تراش|تراشان|تراو|ترس|ترسان' +
'|ترش|ترک|ترکان|تکان|تن|توان|توپ|جنب|جنبان|جنگ|جه|جهان|جو|جوش|جوشان|جوی|چاپ|چای|چپ|چپان' +
'|چر|چران|چرب|چرخ|چرخان|چسب|چسبان|چش|چشان|چک|چکان|چل|چلان|چم|چین|خار|خاران|خای|خر|خراش' +
'|خراشان|خرام|خروش|خز|خشک|خشکان|خل|خم|خند|خندان|خواب|خوابان|خوان|خواه|خور|خوران|خوف|خیز|خیس' +
'|خیسان|دار|درخش|درخشان|درو|دزد|دم|ده|دو|دوان|دوز|دوش|ران|ربا|ربای|رخش|رس|رسان' +
'|رشت|رقص|رقصان|رم|رنج|رنجان|رند|ره|رهان|رو|روب|روی|رویان|ریز|ریس|رین|زا|زار|زای|زدا' +
'|زدای|زن|زی|ساب|ساز|سای|سپار|سپر|سپوز|ستا|ستان|ستر|ستیز|سر|سرا|سرای|سرشت|سز|سگال|سنب' +
'|سنج|سوز|سوزان|شاش|شای|شتاب|شکاف|شکف|شکن|شکوف|شکیب|شمار|شمر|شناس|شناسان|شنو|شو|شور|شوران|شوی' +
'|طپ|طلب|طوف|غارت|غر|غلت|غلتان|غلط|غلطان|غنو|فرسا|فرسای|فرست|فرما|فرمای|فروش|فریب|فشار|فشان|فشر' +
'|فهم|فهمان|قاپ|قبولان|کار|کاه|کاو|کش|کَش|کُش|کِش|کشان|کف|کن|کوب|کوچ|کوش|گا|گای|گداز' +
'|گذار|گذر|گذران|گرا|گراز|گرای|گرد|گردان|گرو|گری|گریان|گریز|گز|گزار|گزین|گسار|گستر|گستران|گسل|گشا' +
'|گشای|گمار|گنج|گنجان|گند|گو|گوار|گوز|گوی|گیر|لرز|لرزان|لغز|لغزان|لم|لمبان|لند|لنگ|له|لول' +
'|لیس|ماس|مال|مان|مک|مول|موی|میر|ناز|نال|نام|نشان|نشین|نکوه|نگار|نگر|نما|نمای|نمایان|نه' +
'|نهنب|نواز|نورد|نوش|نوشان|نویس|نیوش|هراس|هست|هل|ورز|وز|وزان|یاب|یار|یاز' +
')';
var _persianComplexPastVerbs = {
'باز': 'آفرید|آمد|آموخت|آورد|ایستاد|تابید|جست|خواند|داشت|رساند|ستاند|شمرد|ماند|نمایاند|نهاد|نگریست|پرسید|گذارد' +
'|گرداند|گردید|گرفت|گشت|گشود|گفت|یافت',
'در': 'بر ?داشت|بر ?گرفت|آمد|آمیخت|آورد|آویخت|افتاد|افکند|انداخت|رفت|ماند|نوردید|کشید|گرفت',
'بر': 'آشفت|آمد|آورد|افتاد|افراشت|افروخت|افشاند|افکند|انداخت|انگیخت|تاباند|تابید|تافت|تنید|جهید|خاست|خواست|خورد' +
'|داشت|دمید|شمرد|نهاد|چید|کرد|کشید|گرداند|گردانید|گردید|گزید|گشت|گشود|گمارد|گماشت',
'فرو': 'آمد|خورد|داد|رفت|نشاند|کرد|گذارد|گذاشت',
'وا': 'داشت|رهاند|ماند|نهاد|کرد',
'ور': 'آمد|افتاد|رفت',
'یاد': 'گرفت',
'پراکنده': 'ساخت',
'زمین': 'خورد',
'گول': 'زد',
'لخت': 'کرد'
};
var _persianComplexPresentVerbs = {
'باز': 'آفرین|آموز|آور|ایست|تاب|جو|خوان|دار|رس|ستان|شمار|مان|نمایان|نه|نگر|پرس|گذار|گردان|گرد|گشا|گو|گیر|یاب',
'در': 'بر ?دار|بر ?گیر|آمیز|آور|آویز|افت|افکن|انداز|مان|نورد|کش|گذر|گیر',
'بر': 'آشوب|آور|افت|افراز|افروز|افشان|افکن|انداز|انگیز|تابان|تاب|تن|جه|خواه|خور|خیز|دار|دم|شمار|نه|چین|کش|کن' +
'|گردان|گزین|گشا|گمار',
'فرو': 'خور|ده|رو|نشین|کن|گذار',
'وا': 'دار|رهان|مان|نه|کن',
'ور': 'افت|رو',
'یاد': 'گیر',
'پراکنده': 'ساز',
'زمین': 'خور',
'گول': 'زن',
'لخت': 'کن'
};
/**
* Join compound verbs (باز آفرید, در آمد, بر داشت, etc.) with a ZWNJ instead of a space, using
* the curated prefix/stem pairs above so only real compound verbs match.
* @param {string} text The text to process.
*/
var _applyComplexVerbZwnj = function(text) {
var prefix, stems;
for (prefix in _persianComplexPastVerbs) {
if (!_persianComplexPastVerbs.hasOwnProperty(prefix)) continue;
stems = _persianComplexPastVerbs[prefix];
text = text.replace(new RegExp(
'(^|[^' + _persianCharacters + '])(' + prefix + ') ?(می|نمی|)( |\u200c|)(ن|)(' +
stems + ')(م|ی|یم|ید|ند|ه|ن|)($|[^' + _persianCharacters + '])', 'g'),
'$1$2\u200c$3\u200c$5$6$7$8');
}
for (prefix in _persianComplexPresentVerbs) {
if (!_persianComplexPresentVerbs.hasOwnProperty(prefix)) continue;
stems = _persianComplexPresentVerbs[prefix];
text = text.replace(new RegExp(
'(^|[^' + _persianCharacters + '])(' + prefix + ') ?(می|نمی|)( |\u200c|)(ن|)(' +
stems + ')(م|ی|د|یم|ید|ند|ن)($|[^' + _persianCharacters + '])', 'g'),
'$1$2\u200c$3\u200c$5$6$7$8');
}
return text;
};
/**
* Join می/نمی to a following verb stem with a ZWNJ (میرود), join the plural/superlative
* suffixes ها/ترین with a ZWNJ, and a handful of specific, well-tested exceptions (میدانی,
* میتوان, میگوی دریایی, میدوی) that would otherwise be wrongly split by the general rules.
* Ported from the same gadget's applyZwnj(); see the smartZwnj() comment for the caveat this
* still doesn't (and can't) fully resolve.
* @param {string} text The text to process.
*/
var _applyZwnj = function(text) {
text = _applyComplexVerbZwnj(text);
return _cleanupZwnjArtifacts(text)
.replace(
new RegExp('(^|[^' + _persianCharacters + '])(می|نمی) ?' + _persianPastVerbs +
'(م|ی|یم|ید|ند|ه|)($|[^' + _persianCharacters + '])', 'g'),
'$1$2\u200c$3$4$5'
)
.replace(
new RegExp('(^|[^' + _persianCharacters + '])(می|نمی) ?' + _persianPresentVerbs +
'(م|ی|د|یم|ید|ند)($|[^' + _persianCharacters + '])', 'g'),
'$1$2\u200c$3$4$5'
)
// ماضی نقلی: خوردهام, رفتهاید, ...
.replace(
new RegExp('(^|[^' + _persianCharacters + '])(ن|)' + _persianPastVerbs +
'ه (ام|ای|ایم|اید|اند)($|[^' + _persianCharacters + '])', 'g'),
'$1$2$3ه\u200c$4$5'
)
.replace(new RegExp('([' + _persianCharacters + '])هاست($|[^' + _persianCharacters + '])', 'g'), '$1ه است$2')
// «دان» handled separately: its «ی» suffix collides with the verb «دانی»/میدانی
.replace(
new RegExp('(^|[^' + _persianCharacters + '])(می|نمی) ?(دان)(م|د|یم|ید|ند)($|[^' + _persianCharacters + '])', 'g'),
'$1$2\u200c$3$4$5'
)
.replace(/(\s)(می|نمی) ?توان/g, '$1$2\u200cتوان')
// «ها» and «ترین» always join with a ZWNJ, never a full space
.replace(/ ها([\]\.،\:»\)\s]|\'{2,3}|\={2,})/g, '\u200cها$1')
.replace(/ ها(ی|یی|یم|یت|یش|ی?مان|ی?تان|ی?شان)([\]\.،\:»\)\s])/g, '\u200cها$1$2')
.replace(/هها/g, 'هها')
.replace(/ ترین([\]\.،\:»\)\s]|\'{2,3}|\={2,})/g, '\u200cترین$1')
// a few specific words that would otherwise be caught by the general rules above
.replace(new RegExp('می\u200cگوی($|[^' + _persianCharacters + '\u200c])', 'g'), 'میگوی$1') // میگوی دریایی
.replace(new RegExp('می\u200cدوی($|[^' + _persianCharacters + '\u200c])', 'g'), 'میدوی$1'); // میدوی (ابهامزدایی)
};
/**
* Join می/نمی verb prefixes (and a few common suffixes) to what follows with a ZWNJ instead of
* a full space, using curated Persian verb-stem lists so it only fires on real verbs.
*
* This is scoped to your selection rather than the whole page on purpose: even with the word
* lists, «می» on its own is also the classical word for "wine" (می ناب، میگلگون، ساقی می ده…),
* and Wikisource hosts a lot of exactly the poetry where that comes up. The word-list approach
* makes this far safer than a blind "می" + space replace, but it can't tell «می» the prefix from
* «می» the noun by context alone - so select the passage you've checked, rather than running it
* across a whole ghazal unread.
* @param {object} editor The script helpers for the page.
*/
var smartZwnj = function(editor) {
editor.replaceSelection(_applyZwnj);
};
});
// </nowiki>
jjhu7s091hapt3gxqwwnr3hbjsr50kn
299772
299769
2026-09-20T15:03:27Z
Hanooz
17889
299772
javascript
text/javascript
/*
This page defines a TemplateScript library for fa.wikisource.org (ویکینبشتهٔ فارسی).
It's not meant to be referenced directly - install it the same way as the upstream
English version described at [[Wikisource:TemplateScript]].
This is a Persian Wikisource adaptation of Pathoschild's proofreading and typography
libraries, pared down to a focused four-action toolset. It draws on the fa.wikipedia/
fa.wikisource "Persian text style improvement tools" gadget (Gadget-Extra-Editbuttons-
persianwikitools.js) for several of the character/orthography helpers, since that's the
community's own tested implementation rather than something invented for this file.
*/
/* global $, pathoschild */
/**
* TemplateScript adds configurable templates and scripts to the sidebar, and adds an example regex editor.
* @see https://meta.wikimedia.org/wiki/TemplateScript
*
* The upstream proofreading library's @update-token has been removed on purpose: Wikisource's
* maintenance bot uses that token to overwrite local copies with the vanilla English script,
* which would silently wipe out every change made here for fa.wikisource. See [[Wikisource:Tools
* and scripts]] for how the update-bot category works.
*/
// <nowiki>
$.ajax('//tools-static.wmflabs.org/meta/scripts/pathoschild.templatescript.js', { dataType:'script', cache:true }).then(function() {
/*********
** Define library
*********/
pathoschild.TemplateScript.library.define({
key: 'wikisource.fa.tools',
name: 'ابزارهای ویکینبشتهٔ فارسی',
url: '//fa.wikisource.org/wiki/راهنما:نمونهخوانی',
description: 'مجموعهای از ابزارها برای <a href="/wiki/راهنما:نمونهخوانی">نمونهخوانی آثار در فضای نام <tt>برگه:</tt></a>: پاکسازی OCR (با اصلاح نویسهها، کشیده، ارقام، جای اعراب، نیمفاصله، نویسههای نامرئی بیاثر، و نشانهگذاری) و اصلاح نیمفاصلهٔ افعال.',
categories: [
{
name: 'ابزارهای برگه',
scripts: [
{ key: 'cleanup-ocr', name: 'پاکسازی OCR', script: function(editor) { pageCleanup(editor); }, forNamespaces: 'page' },
{ key: 'smart-zwnj', name: 'اصلاح نیمفاصلهٔ افعال', script: function(editor) { smartZwnj(editor); }, forNamespaces: 'page' }
]
}
]
});
/*********
** Private methods
*********/
// Digit tables used to convert between Latin, Persian and Arabic-Indic numerals.
var _persianDigits = '۰۱۲۳۴۵۶۷۸۹';
var _arabicIndicDigits = '٠١٢٣٤٥٦٧٨٩';
var _latinDigits = '0123456789';
// Character-class tables used by the Persian-specific cleanup helpers below. Ported from the
// fa.wikipedia/fa.wikisource "Persian text style improvement tools" gadget (Gadget-Extra-
// Editbuttons-persianwikitools.js) - the community's own tested implementation of this
// normalisation, rather than something invented for this file.
var _hamza = '\u0654';
var _persianVowels = '\u064B-\u0650\u0652\u0670';
// non-standard letterforms (Arabic ي/ك, Urdu, Pashto, Uyghur presentation forms, etc.) that
// should still count as "a Persian-ish letter" for context-matching, even before
// _toStandardPersianCharacters() below has normalised them
var _similarPersianCharacters = '\u0643\uFB91\uFB90\uFB8F\uFB8E\uFEDC\uFEDB\uFEDA\uFED9\u0649\uFEEF\u064A\u06C1\u06D5\u06BE\uFEF0-\uFEF4';
var _persianCharacters = '\u0621-\u0655\u067E\u0686\u0698\u06AF\u06A9\u0643\u06AA\uFED9\uFEDA\u06CC\uFEF1\uFEF2' + _similarPersianCharacters;
/**
* Convert any mix of Latin or Arabic-Indic digits in a string to Persian digits, for display in wikitext.
* @param {string} text The text to convert.
*/
var _digitsToPersian = function(text) {
var result = '';
for (var i = 0; i < text.length; i++) {
var ch = text.charAt(i);
var idx = _latinDigits.indexOf(ch);
if (idx === -1) idx = _arabicIndicDigits.indexOf(ch);
result += (idx !== -1) ? _persianDigits.charAt(idx) : ch;
}
return result;
};
/**
* OCR/PDF presentation-form and script-variant glyphs that extraction sometimes leaves as
* literal characters instead of the standard Persian letter - a common problem specifically on
* PDF-sourced works, where each glyph can get saved as its own shaped codepoint. Keys are the
* standard letter (or letter + a real ZWNJ, for the two glyphs that visually look joined);
* values are a character class of the presentation-form/variant glyphs that should fold into it.
*/
var _persianGlyphs = {
'\u200cه': 'ﻫ',
'ی\u200c': 'ﻰﻲ',
'أ': 'ﺄﺃﺃ',
'آ': 'ﺁﺁﺂ',
'إ': 'ﺇﺈﺇ',
'ا': 'ﺍﺎ',
'ب': 'ﺏﺐﺑﺒ',
'پ': 'ﭖﭗﭘﭙ',
'ت': 'ﺕﺖﺗﺘ',
'ث': 'ﺙﺚﺛﺜ',
'ج': 'ﺝﺞﺟﺠ',
'چ': 'ﭺﭻﭼﭽ',
'ح': 'ﺡﺢﺣﺤ',
'خ': 'ﺥﺦﺧﺨ',
'د': 'ﺩﺪ',
'ذ': 'ﺫﺬ',
'ر': 'ﺭﺮ',
'ز': 'ﺯﺰ',
'ژ': 'ﮊﮋ',
'س': 'ﺱﺲﺳﺴ',
'ش': 'ﺵﺶﺷﺸ',
'ص': 'ﺹﺺﺻﺼ',
'ض': 'ﺽﺾﺿﻀ',
'ط': 'ﻁﻂﻃﻄ',
'ظ': 'ﻅﻆﻇﻈ',
'ع': 'ﻉﻊﻋﻌ',
'غ': 'ﻍﻎﻏﻐ',
'ف': 'ﻑﻒﻓﻔ',
'ق': 'ﻕﻖﻗﻘ',
'ک': 'ﮎﮏﮐﮑﻙﻚﻛﻜ',
'گ': 'ﮒﮓﮔﮕ',
'ل': 'ﻝﻞﻟﻠ',
'م': 'ﻡﻢﻣﻤ',
'ن': 'ﻥﻦﻧﻨ',
'ه': 'ﻩﻪﻫﻬ',
'هٔ': 'ﮤﮥ',
'و': 'ﻭﻮ',
'ؤ': 'ﺅﺅﺆ',
'ی': 'ﯼﯽﯾﯿﻯﻰﻱﻲﻳﻴ',
'ئ': 'ﺉﺊﺋﺌ',
'لا': 'ﻻﻼ',
'لإ': 'ﻹﻺ',
'لأ': 'ﻸﻷ',
'لآ': 'ﻵﻶ'
};
/**
* Remove stray/duplicate ZWNJ-family characters and OCR artifacts that stand in for a ZWNJ,
* without trying to guess where a NEW ZWNJ should be inserted between words - that needs the
* verb word-lists in smartZwnj() below, since blindly inserting one is what risks corrupting
* unrelated text (see that function's comment for why).
* @param {string} text The text to clean.
*/
var _cleanupZwnjArtifacts = function(text) {
return text
// a ZWNJ (or stray zero-width/BOM/bidi-mark character) with nothing before or after it
// to join or not-join has no possible effect on the rendered output - most commonly a
// ZWNJ that ended up with a plain space next to it instead of a letter
.replace(/^[\u200B-\u200D\uFEFF]+/, '')
.replace(/[\u200B-\u200D\uFEFF]+$/, '')
// a stray LRM/RLM between two Persian letters was almost always meant to be a ZWNJ
.replace(new RegExp('([' + _persianCharacters + '] *)[\u200F\u200E]+( *[' + _persianCharacters + '])', 'g'), '$1\u200c$2')
// collapse repeated invisible-joiner/mark characters down to one
.replace(/([\u200B-\u200D\uFEFF\u200E\u200F]){2,}/g, '$1')
// ¬ between two Persian letters is almost always a misrecognised ZWNJ, not a real character
.replace(new RegExp('([' + _persianCharacters + '])¬(?=[' + _persianCharacters + '])', 'g'), '$1\u200c')
// a ZWNJ has no business after a digit or most punctuation/brackets, or next to a Latin word
.replace(/([۰-۹0-9إأةؤورزژاآدذ،؛,:«»\\\/@#$٪×*()ـ\-=|ء])\u200c/g, '$1')
.replace(/[\u200B-\u200D\uFEFF]([\w])/g, '$1')
.replace(/([\w])[\u200B-\u200D\uFEFF]/g, '$1')
.replace(new RegExp('[\\u200B-\\u200D\\uFEFF]([' + _persianVowels + _arabicIndicDigits + _persianDigits + _latinDigits + _hamza + '])', 'g'), '$1')
.replace(new RegExp('([' + _arabicIndicDigits + '])[\\u200B-\\u200D\\uFEFF]', 'g'), '$1')
.replace(/[\u200B\u200C\uFEFF]([ء\n\s\[\].،«»:()؛؟?;$!@\-=+\\|])/g, '$1')
.replace(/([\n\s\[.،«»:()؛؟?;$!@\-=+\\|])[\u200B-\u200D\uFEFF]/g, '$1')
.replace(/[\u200B-\u200D\uFEFF](\]\][\s\n])/g, '$1')
.replace(/([\n\s]\[\[)[\u200B-\u200D\uFEFF]/g, '$1');
};
/**
* Fold OCR/PDF presentation-form glyphs and non-Persian script variants (Arabic ي/ك, Urdu,
* Pashto, Uyghur, Kurdish) back to standard Persian characters, and canonicalise the hamza-
* bearing forms per ISIRI 6219 (Iran's national character-encoding standard).
* @param {string} text The text to clean.
*/
var _toStandardPersianCharacters = function(text) {
for (var standard in _persianGlyphs) {
if (_persianGlyphs.hasOwnProperty(standard)) {
text = text.replace(new RegExp('[' + _persianGlyphs[standard] + ']', 'g'), standard);
}
}
return _cleanupZwnjArtifacts(text) // needed because of the two ZWNJ-bearing keys above
.replace(/ك/g, 'ک') // Arabic
.replace(/ڪ/g, 'ک') // Urdu
.replace(/ﻙ/g, 'ک') // Pushtu
.replace(/ﻚ/g, 'ک') // Uyghur
.replace(/ي/g, 'ی') // Arabic
.replace(/ى/g, 'ی') // Urdu
.replace(/ے/g, 'ی') // Urdu
.replace(/ۍ/g, 'ی') // Pushtu
.replace(/ې/g, 'ی') // Uyghur
.replace(/ہ/g, 'ه') // Urdu
.replace(/ە/g, 'ه\u200c') // Kurdish
.replace(/ھ/g, 'ه') // Kurdish
.replace(/أ/g, 'أ') // canonical alef+hamza form per ISIRI 6219
.replace(/آ/g, 'آ');
};
/**
* Strip tatweel/kashide (ـ, U+0640) characters from a string. Tatweel is a pure justification
* stretch-mark used in decorative Persian typesetting (common on title pages); OCR frequently
* captures it as literal characters, but it has no place in a wikilink target or in wikitext.
* @param {string} text The text to clean.
*/
var _stripTatweel = function(text) {
return text.replace(/\u0640+/g, '');
};
/*********
** Script methods
*********/
/**
* Clean up OCR errors in the text, and push <noinclude> content at the top
* & bottom of the page into the header & footer boxes respectively.
* @param {object} editor The script helpers for the page.
*/
var pageCleanup = function(editor) {
// Strip characters that have no effect on the rendered output at all, before anything
// else runs: control characters, BOM and soft hyphen have no legitimate place in body
// text; \r is a Windows line-ending artifact once \n is already there; a run of exotic
// Unicode space characters (NBSP, thin space, ideographic space, etc.) right before a
// line break just disappears, a tab/NBSP right after one disappears too (plain space is
// deliberately left alone there, since a leading space triggers MediaWiki's <pre>
// formatting), and any exotic space left in running text becomes a plain space. Ported
// from the fa.wikipedia/fa.wikisource gadget's applyOrthography(), and cross-checked
// against Virastar (github.com/aziz/virastar and its ports, a widely used standalone
// Persian text cleaner) - both do this same normalisation as a first step.
var text = editor.get()
.replace(/\r/g, '')
.replace(/[\x00-\x08\x0B\x0C\x0E-\x1F\x7F-\x9F\uFEFF\u00AD\u0085]+/g, '')
.replace(/[ \xA0\u1680\u180E\u2000-\u200A\u202F\u205F\u3000]+\n/g, '\n')
.replace(/\n[\t\u00A0]+/g, '\n')
.replace(/[\xA0\u1680\u180E\u2000-\u200A\u202F\u205F\u3000]/g, ' ');
// Normalise the raw text next: fold OCR/PDF presentation-form glyphs and non-Persian
// letter variants to standard Persian characters (this also mops up stray ZWNJ-family
// characters left over from OCR - see _cleanupZwnjArtifacts, including a ZWNJ that's now
// got a plain space next to it instead of a letter, which can no longer do anything),
// strip decorative tatweel, convert Latin/Arabic-Indic digits to Persian, and fix vowel-
// mark (harakat) placement - a mark separated from its letter by a stray space moves back
// next to the letter, and duplicate marks on the same letter collapse to one. The digit
// conversion runs unconditionally on the whole page - if a work legitimately needs
// Western digits somewhere (an ISBN, a citation in another language), those get
// converted too; there's no separate opt-in step for that anymore.
text = _toStandardPersianCharacters(text);
text = _stripTatweel(text);
text = _digitsToPersian(text);
text = text.replace(new RegExp('([' + _persianCharacters + _persianVowels + _hamza + '])(\\s)([' + _persianVowels + _hamza + '])', 'g'), '$1$3$2');
text = text.replace(new RegExp('([' + _persianVowels + _hamza + ']){2,}', 'g'), '$1');
editor.set(text);
// Strip leading lines that are pure page-number "running header" junk - the digits-only
// case of the en.wikisource WsCleanup gadget's running-header patterns. Dropped that
// gadget's other pattern (digits plus an ALL-CAPS title fragment) since Persian has no
// letter case to signal "this is header text, not body text" the way English capitals
// do. Caveat carried over from the source: a work with numbered stanzas/verses starting
// right at the top of a page could have its number line mistaken for header junk here -
// worth a glance after running this on poetry.
(function() {
var lines = editor.get().split(/\r?\n/);
var pageNumberOnly = /^[\s0-9۰-۹٠-٩.,\-–—]+$/;
var start = 0;
for (var li = 0; li < lines.length; li++) {
if (lines[li].trim().length === 0 || (pageNumberOnly.test(lines[li]) && lines[li].trim().length > 0)) {
start++;
continue;
}
break;
}
if (start > 0) {
editor.set(lines.slice(start).join('\n'));
}
})();
// push <noinclude> content at the top & bottom into the header & footer
if (editor.get().match(/^<noinclude\>/)) {
var text = editor.get();
var e = text.indexOf("</noinclude>");
$('#wpHeaderTextbox').val(function(i, val) {
return $.trim(val + "\n" + text.substr(11, e-11).replace(/^\s+|\s+$/g, ''));
});
editor.set(text.substr(e+12));
}
if (editor.get().match(/<\/noinclude\>$/)) {
var text = editor.get();
var s = text.lastIndexOf("<noinclude>");
$('#wpFooterTextbox').val(function(i, val) {
return $.trim(text.substr(s+11, text.length-s-11-12).replace(/^\s+|\s+$/g, '') + "\n" + val);
});
editor.set(text.substr(0, s));
}
// clean up text
editor
// remove trailing spaces at the end of each line
.replace(/ +\n/g, '\n')
// remove trailing whitespace preceding a hard line break
.replace(/ +<br *\/?>/g, '<br />')
// remove trailing whitespace and numerals at the end of page text - matches
// Latin, Persian and Arabic-Indic digits, since page-number artifacts at the
// bottom of a scanned fa.wikisource page can OCR as any of the three
.replace(/[\s0-9۰-۹٠-٩]+$/g, '')
// remove trailing spaces at the end of refs
.replace(/ +<\/ref>/g, '</ref>')
// remove trailing spaces at the end of template calls
.replace(/ +}}/g, '}}')
// convert double-hyphen to mdash (avoiding breaking HTML comment syntax)
.replace(/([^\!])--([^>])/g, '$1—$2')
// remove spacing around mdash, but only if it has spaces on both sides
.replace(/ +— +/g, '—')
// "Digitized by Google" scan-watermark text and stray OCR of the word "Google" -
// the watermark itself is always in English regardless of the scanned work's
// language, so this is just as relevant to a Google Books-sourced Persian scan
.replace(/\s?D[il]g[il]t[il][sz][eco]d\s+by[^\n]*\s+([6G][Oo0]{2}g[lI]e)?/g, '')
.replace(/\bG[oO0]{2}gle\b/g, '')
// highly suspicious OCR noise characters
.replace(/[■•]/g, '')
// a line containing nothing but a stray punctuation mark is likely scan noise
.replace(/^[.,^،؛]$/gm, '')
// underscores (often OCR of underlined text or a form field) aren't meaningful in wikitext
.replace(/_/g, ' ')
// number ranges get an en dash, not a hyphen, per fa.wikisource's وپ:خط تیره
.replace(/([۰-۹]+)[ \t]?-[ \t]?(?=[۰-۹])/g, '$1–')
// <section begin=x/> needs to be <section begin="x"/> to parse
.replace(/<section (begin|end)=(\w[^/]+)\/>/g, '<section $1="$2"/>')
// a <ref> shouldn't have a space glued before it
.replace(/ <ref/g, '<ref')
// {{hws}}/{{hwe}} shorthand for a word split across a page boundary, expanded to the
// full template names (same English names used across Wikisource language versions,
// fa.wikisource included, for this kind of utility template)
.replace(/{{hws\|/g, '{{hyphenated word start|')
.replace(/{{hwe\|/g, '{{hyphenated word end|')
// {{c|...}} shorthand for {{center|...}}, if used
.replace(/{{c\|/g, '{{center|');
// clean up pages if they don't have <poem>
if (!editor.contains('<poem>') && !editor.contains('{{' + 'ppoem')) {
editor
// remove single line breaks; preserve multiple.
// but not if there's a tag, template or table syntax either side of the line break
.replace(/([^>}\|\n])\n([^:#\*<{\|\n])/g, '$1 $2')
// collapse sequences of spaces into a single space
.replace(/ +/g, ' ');
}
// more page cleanup
editor
// dump spurious hard breaks at the end of paragraphs
.replace(/<br *\/?>\n\n/g, '\n\n')
// remove unwanted spaces before punctuation marks - both Latin (;:?!,.)) and
// Persian (،؛؟) forms, since either can turn up depending on the OCR engine
.replace(/ ([;:\?!,.)،؛؟])/g, '$1')
// and no space after an opening parenthesis
.replace(/\( +/g, '(')
// convert ASCII look-alikes to proper Persian punctuation when they follow Persian
// text. Properly printed Persian sources always used ؟/؛/، - an ASCII ?/;/, next to
// Persian text is an OCR/encoding artifact to restore, not an authorial style choice
.replace(new RegExp('([' + _persianCharacters + '])\\?', 'g'), '$1؟')
.replace(new RegExp('([' + _persianCharacters + ']);', 'g'), '$1؛')
.replace(new RegExp('([' + _persianCharacters + '])(\\]\\]|»|)\\,', 'g'), '$1$2،')
// ensure a space after punctuation when it's glued directly to the next word...
.replace(/([;:\?!,.)،؛؟])([^\s0-9۰-۹٠-٩\n}|"«»'’”])/g, '$1 $2')
// ...but consecutive punctuation marks (or end of line) don't get a space between them
.replace(/([;:\?!,.)،؛؟]) +([\n;:\?!,.)،؛؟\]]|$)/g, '$1$2')
// a double period after a Persian letter is almost always meant to be one
.replace(new RegExp('([' + _persianCharacters + '])\\.\\. (?=[' + _persianCharacters + '])', 'g'), '$1. ')
// three-or-more dots after Persian text is an ellipsis, not literal periods
.replace(new RegExp('([' + _persianCharacters + '])( *)\\.{3,}', 'g'), '$1$2…')
// canonicalise ۀ / هٴ / هٓ to ISIRI 6219's ه + combining hamza above (هٔ) - this is a
// same-character encoding choice, not a wording change, so it's safe to automate
.replace(/[ۂۀ]/g, 'هٔ')
.replace(/هٴ/g, 'هٔ')
.replace(/هٓ/g, 'هٔ')
// unicodify
.replace(/—/g, '—')
.replace(/–/g, '–')
.replace(/"/g, '"')
.replace(/<center>\s*([.\n]*?)\s*<\/center>/g, '{{center|$1}}');
// Run the invisible-ZWNJ-artifact cleanup once more: the punctuation and digit changes
// above can newly strand a ZWNJ next to something it can no longer affect (e.g. a ZWNJ
// that's now right before a converted Persian digit or a newly-inserted punctuation mark).
editor.set(_cleanupZwnjArtifacts(editor.get()));
};
// Verb stems used by smartZwnj() below to decide when "می"/"نمی" is a verb prefix that should
// join to what follows with a ZWNJ, rather than something else (most importantly, the classical
// word «می» meaning "wine" - extremely common in the poetry Wikisource hosts a lot of, e.g.
// Hafez or Khayyam). Matching only against a real verb stem from this list, with a real person/
// tense suffix, is what makes this safe enough to automate at all; a blind "می" + space match
// would wrongly rewrite «می ناب» (pure wine) in a ghazal. Ported verbatim from the fa.wikipedia/
// fa.wikisource "Persian text style improvement tools" gadget, since retyping ~500 verb stems by
// hand is exactly how new bugs get introduced.
var _persianPastVerbs = '(' +
'ارزید|افتاد|افراشت|افروخت|افزود|افسرد|افشاند|افکند|انباشت|انجامید|انداخت|اندوخت|اندود|اندیشید|انگاشت|انگیخت|انگیزاند|اوباشت|ایستاد' +
'|آراست|آراماند|آرامید|آرمید|آزرد|آزمود|آسود|آشامید|آشفت|آشوبید|آغازید|آغشت|آفرید|آکند|آگند|آلود|آمد|آمرزید|آموخت|آموزاند' +
'|آمیخت|آهیخت|آورد|آویخت|باخت|باراند|بارید|بافت|بالید|باوراند|بایست|بخشود|بخشید|برازید|برد|برید|بست|بسود|بسیجید|بلعید' +
'|بود|بوسید|بویید|بیخت|پاشاند|پاشید|پالود|پایید|پخت|پذیراند|پذیرفت|پراکند|پراند|پرداخت|پرستید|پرسید|پرهیزید|پروراند|پرورد|پرید' +
'|پژمرد|پژوهید|پسندید|پلاسید|پلکید|پناهید|پنداشت|پوسید|پوشاند|پوشید|پویید|پیچاند|پیچانید|پیچید|پیراست|پیمود|پیوست|تاباند|تابید|تاخت' +
'|تاراند|تازاند|تازید|تافت|تپاند|تپید|تراشاند|تراشید|تراوید|ترساند|ترسید|ترشید|ترکاند|ترکید|تکاند|تکانید|تنید|توانست|جَست|جُست' +
'|جست|جنباند|جنبید|جنگید|جهاند|جهید|جوشاند|جوشید|جوید|چاپید|چایید|چپاند|چپید|چراند|چربید|چرخاند|چرخید|چرید|چسباند|چسبید' +
'|چشاند|چشید|چکاند|چکید|چلاند|چلانید|چمید|چید|خاراند|خارید|خاست|خایید|خراشاند|خراشید|خرامید|خروشید|خرید|خزید|خشکاند' +
'|خشکید|خفت|خلید|خمید|خنداند|خندانید|خندید|خواباند|خوابانید|خوابید|خواست|خواند|خوراند|خورد|خوفید|خیساند|خیسید|داد|داشت|دانست' +
'|درخشانید|درخشید|دروید|درید|دزدید|دمید|دواند|دوخت|دوشید|دوید|دید|دیدم|راند|ربود|رخشید|رساند|رسانید|رست|رَست|رُست' +
'|رسید|رشت|رفت|رُفت|رقصاند|رقصید|رمید|رنجاند|رنجید|رندید|رهاند|رهانید|رهید|روبید|روفت|رویاند|رویید|ریخت|رید|ریسید' +
'|زاد|زارید|زایید|زد|زدود|زیست|سابید|ساخت|سپارد|سپرد|سپوخت|ستاند|ستد|سترد|ستود|ستیزید|سرایید|سرشت|سرود|سرید' +
'|سزید|سفت|سگالید|سنجید|سوخت|سود|سوزاند|شاشید|شایست|شتافت|شد|شست|شکافت|شکست|شکفت|شکیفت|شگفت|شمارد|شمرد|شناخت' +
'|شناساند|شنید|شوراند|شورید|طپید|طلبید|طوفید|غارتید|غرید|غلتاند|غلتانید|غلتید|غلطاند|غلطانید|غلطید|غنود|فرستاد|فرسود|فرمود|فروخت' +
'|فریفت|فشاند|فشرد|فهماند|فهمید|قاپید|قبولاند|کاست|کاشت|کاوید|کرد|کشاند|کشانید|کشت|کشید|کفت|کفید|کند|کوبید|کوچید' +
'|کوشید|کوفت|گَزید|گُزید|گایید|گداخت|گذارد|گذاشت|گذراند|گذشت|گرازید|گرایید|گرداند|گردانید|گردید|گرفت|گروید|گریاند|گریخت|گریست' +
'|گزارد|گزید|گسارد|گستراند|گسترد|گسست|گسیخت|گشت|گشود|گفت|گمارد|گماشت|گنجاند|گنجانید|گنجید|گندید|گوارید|گوزید|لرزاند|لرزید' +
'|لغزاند|لغزید|لمباند|لمدنی|لمید|لندید|لنگید|لهید|لولید|لیسید|ماسید|مالاند|مالید|ماند|مانست|مرد|مکشید|مکید|مولید|مویید' +
'|نازید|نالید|نامید|نشاند|نشست|نکوهید|نگاشت|نگریست|نمایاند|نمود|نهاد|نهفت|نواخت|نوردید|نوشاند|نوشت|نوشید|نیوشید|هراسید|هشت' +
'|ورزید|وزاند|وزید|یارست|یازید|یافت' +
')';
var _persianPresentVerbs = '(' +
'ارز|افت|افراز|افروز|افزا|افزای|افسر|افشان|افکن|انبار|انباز|انجام|انداز|اندای|اندوز|اندیش|انگار|انگیز|انگیزان' +
'|اوبار|ایست|آرا|آرام|آرامان|آرای|آزار|آزما|آزمای|آسا|آسای|آشام|آشوب|آغار|آغاز|آفرین|آکن|آگن|آلا|آلای' +
'|آمرز|آموز|آموزان|آمیز|آهنج|آور|آویز|آی|بار|باران|باز|باش|باف|بال|باوران|بای|باید|بخش|بخشا|بخشای' +
'|بر|بَر|بُر|براز|بساو|بسیج|بلع|بند|بو|بوس|بوی|بیز|بین|پا|پاش|پاشان|پالا|پالای|پذیر|پذیران' +
'|پر|پراکن|پران|پرداز|پرس|پرست|پرهیز|پرور|پروران|پز|پژمر|پژوه|پسند|پلاس|پلک|پناه|پندار|پوس|پوش|پوشان' +
'|پوی|پیچ|پیچان|پیرا|پیرای|پیما|پیمای|پیوند|تاب|تابان|تاران|تاز|تازان|تپ|تپان|تراش|تراشان|تراو|ترس|ترسان' +
'|ترش|ترک|ترکان|تکان|تن|توان|توپ|جنب|جنبان|جنگ|جه|جهان|جو|جوش|جوشان|جوی|چاپ|چای|چپ|چپان' +
'|چر|چران|چرب|چرخ|چرخان|چسب|چسبان|چش|چشان|چک|چکان|چل|چلان|چم|چین|خار|خاران|خای|خر|خراش' +
'|خراشان|خرام|خروش|خز|خشک|خشکان|خل|خم|خند|خندان|خواب|خوابان|خوان|خواه|خور|خوران|خوف|خیز|خیس' +
'|خیسان|دار|درخش|درخشان|درو|دزد|دم|ده|دو|دوان|دوز|دوش|ران|ربا|ربای|رخش|رس|رسان' +
'|رشت|رقص|رقصان|رم|رنج|رنجان|رند|ره|رهان|رو|روب|روی|رویان|ریز|ریس|رین|زا|زار|زای|زدا' +
'|زدای|زن|زی|ساب|ساز|سای|سپار|سپر|سپوز|ستا|ستان|ستر|ستیز|سر|سرا|سرای|سرشت|سز|سگال|سنب' +
'|سنج|سوز|سوزان|شاش|شای|شتاب|شکاف|شکف|شکن|شکوف|شکیب|شمار|شمر|شناس|شناسان|شنو|شو|شور|شوران|شوی' +
'|طپ|طلب|طوف|غارت|غر|غلت|غلتان|غلط|غلطان|غنو|فرسا|فرسای|فرست|فرما|فرمای|فروش|فریب|فشار|فشان|فشر' +
'|فهم|فهمان|قاپ|قبولان|کار|کاه|کاو|کش|کَش|کُش|کِش|کشان|کف|کن|کوب|کوچ|کوش|گا|گای|گداز' +
'|گذار|گذر|گذران|گرا|گراز|گرای|گرد|گردان|گرو|گری|گریان|گریز|گز|گزار|گزین|گسار|گستر|گستران|گسل|گشا' +
'|گشای|گمار|گنج|گنجان|گند|گو|گوار|گوز|گوی|گیر|لرز|لرزان|لغز|لغزان|لم|لمبان|لند|لنگ|له|لول' +
'|لیس|ماس|مال|مان|مک|مول|موی|میر|ناز|نال|نام|نشان|نشین|نکوه|نگار|نگر|نما|نمای|نمایان|نه' +
'|نهنب|نواز|نورد|نوش|نوشان|نویس|نیوش|هراس|هست|هل|ورز|وز|وزان|یاب|یار|یاز' +
')';
var _persianComplexPastVerbs = {
'باز': 'آفرید|آمد|آموخت|آورد|ایستاد|تابید|جست|خواند|داشت|رساند|ستاند|شمرد|ماند|نمایاند|نهاد|نگریست|پرسید|گذارد' +
'|گرداند|گردید|گرفت|گشت|گشود|گفت|یافت',
'در': 'بر ?داشت|بر ?گرفت|آمد|آمیخت|آورد|آویخت|افتاد|افکند|انداخت|رفت|ماند|نوردید|کشید|گرفت',
'بر': 'آشفت|آمد|آورد|افتاد|افراشت|افروخت|افشاند|افکند|انداخت|انگیخت|تاباند|تابید|تافت|تنید|جهید|خاست|خواست|خورد' +
'|داشت|دمید|شمرد|نهاد|چید|کرد|کشید|گرداند|گردانید|گردید|گزید|گشت|گشود|گمارد|گماشت',
'فرو': 'آمد|خورد|داد|رفت|نشاند|کرد|گذارد|گذاشت',
'وا': 'داشت|رهاند|ماند|نهاد|کرد',
'ور': 'آمد|افتاد|رفت',
'یاد': 'گرفت',
'پراکنده': 'ساخت',
'زمین': 'خورد',
'گول': 'زد',
'لخت': 'کرد'
};
var _persianComplexPresentVerbs = {
'باز': 'آفرین|آموز|آور|ایست|تاب|جو|خوان|دار|رس|ستان|شمار|مان|نمایان|نه|نگر|پرس|گذار|گردان|گرد|گشا|گو|گیر|یاب',
'در': 'بر ?دار|بر ?گیر|آمیز|آور|آویز|افت|افکن|انداز|مان|نورد|کش|گذر|گیر',
'بر': 'آشوب|آور|افت|افراز|افروز|افشان|افکن|انداز|انگیز|تابان|تاب|تن|جه|خواه|خور|خیز|دار|دم|شمار|نه|چین|کش|کن' +
'|گردان|گزین|گشا|گمار',
'فرو': 'خور|ده|رو|نشین|کن|گذار',
'وا': 'دار|رهان|مان|نه|کن',
'ور': 'افت|رو',
'یاد': 'گیر',
'پراکنده': 'ساز',
'زمین': 'خور',
'گول': 'زن',
'لخت': 'کن'
};
/**
* Join compound verbs (باز آفرید, در آمد, بر داشت, etc.) with a ZWNJ instead of a space, using
* the curated prefix/stem pairs above so only real compound verbs match.
* @param {string} text The text to process.
*/
var _applyComplexVerbZwnj = function(text) {
var prefix, stems;
for (prefix in _persianComplexPastVerbs) {
if (!_persianComplexPastVerbs.hasOwnProperty(prefix)) continue;
stems = _persianComplexPastVerbs[prefix];
text = text.replace(new RegExp(
'(^|[^' + _persianCharacters + ']) *(' + prefix + ') +(می|نمی|)( |\u200c|)(ن|)(' +
stems + ')(م|ی|یم|ید|ند|ه|ن|)($|[^' + _persianCharacters + '])', 'g'),
'$1$2\u200c$3\u200c$5$6$7$8');
}
for (prefix in _persianComplexPresentVerbs) {
if (!_persianComplexPresentVerbs.hasOwnProperty(prefix)) continue;
stems = _persianComplexPresentVerbs[prefix];
text = text.replace(new RegExp(
'(^|[^' + _persianCharacters + ']) *(' + prefix + ') +(می|نمی|)( |\u200c|)(ن|)(' +
stems + ')(م|ی|د|یم|ید|ند|ن)($|[^' + _persianCharacters + '])', 'g'),
'$1$2\u200c$3\u200c$5$6$7$8');
}
return text;
};
/**
* Join می/نمی to a following verb stem with a ZWNJ (میرود), join the plural/superlative
* suffixes ها/ترین with a ZWNJ, and a handful of specific, well-tested exceptions (میدانی,
* میتوان, میگوی دریایی, میدوی) that would otherwise be wrongly split by the general rules.
*
* Every pattern below requires an ACTUAL existing space to convert - می and the verb stem
* must already be written as two separate space-divided words. An already-solid word like
* «میشود» or «برداشت», with no space anywhere, is left exactly as-is: there is nothing there
* to "fix", and inserting a ZWNJ that wasn't in the source at all is exactly the kind of edit
* that changes what's actually on the page without a real reason to.
*
* Ported from the same gadget's applyZwnj(); see the smartZwnj() comment for the caveat this
* still doesn't (and can't) fully resolve.
* @param {string} text The text to process.
*/
var _applyZwnj = function(text) {
text = _applyComplexVerbZwnj(text);
return _cleanupZwnjArtifacts(text)
.replace(
new RegExp('(^|[^' + _persianCharacters + '])(می|نمی) +' + _persianPastVerbs +
'(م|ی|یم|ید|ند|ه|)($|[^' + _persianCharacters + '])', 'g'),
'$1$2\u200c$3$4$5'
)
.replace(
new RegExp('(^|[^' + _persianCharacters + '])(می|نمی) +' + _persianPresentVerbs +
'(م|ی|د|یم|ید|ند)($|[^' + _persianCharacters + '])', 'g'),
'$1$2\u200c$3$4$5'
)
// ماضی نقلی: خوردهام, رفتهاید, ...
.replace(
new RegExp('(^|[^' + _persianCharacters + '])(ن|)' + _persianPastVerbs +
'ه (ام|ای|ایم|اید|اند)($|[^' + _persianCharacters + '])', 'g'),
'$1$2$3ه\u200c$4$5'
)
.replace(new RegExp('([' + _persianCharacters + '])هاست($|[^' + _persianCharacters + '])', 'g'), '$1ه است$2')
// «دان» handled separately: its «ی» suffix collides with the verb «دانی»/میدانی
.replace(
new RegExp('(^|[^' + _persianCharacters + '])(می|نمی) +(دان)(م|د|یم|ید|ند)($|[^' + _persianCharacters + '])', 'g'),
'$1$2\u200c$3$4$5'
)
.replace(/(\s)(می|نمی) +توان/g, '$1$2\u200cتوان')
// «ها» and «ترین» always join with a ZWNJ, never a full space
.replace(/ ها([\]\.،\:»\)\s]|\'{2,3}|\={2,})/g, '\u200cها$1')
.replace(/ ها(ی|یی|یم|یت|یش|ی?مان|ی?تان|ی?شان)([\]\.،\:»\)\s])/g, '\u200cها$1$2')
.replace(/ ترین([\]\.،\:»\)\s]|\'{2,3}|\={2,})/g, '\u200cترین$1')
// a few specific words that would otherwise be caught by the general rules above
.replace(new RegExp('می\u200cگوی($|[^' + _persianCharacters + '\u200c])', 'g'), 'میگوی$1') // میگوی دریایی
.replace(new RegExp('می\u200cدوی($|[^' + _persianCharacters + '\u200c])', 'g'), 'میدوی$1'); // میدوی (ابهامزدایی)
};
/**
* Run a text transform on the user's selection if they have one, or on the whole page
* otherwise. General-purpose - not specific to any one action - so the same one-button-does-
* both pattern can be reused for other linguistically-risky transforms later, not just
* smartZwnj below.
* @param {object} editor The script helpers for the page.
* @param {function} transform A function that takes a string and returns the transformed string.
*/
var _runOnSelectionOrWholePage = function(editor, transform) {
var box = $('#wpTextbox1').get(0);
if (box && box.selectionStart !== box.selectionEnd) {
editor.replaceSelection(transform);
} else {
editor.set(transform(editor.get()));
}
};
/**
* Join می/نمی verb prefixes (and a few common suffixes) to what follows with a ZWNJ instead of
* a full space, using curated Persian verb-stem lists so it only fires on real verbs.
*
* Runs on your selection if you have one, or the whole page if you don't (see
* _runOnSelectionOrWholePage above) - deliberately your choice rather than always-automatic,
* for two separate reasons:
*
* 1. Even with the word lists, «می» on its own is also the classical word for "wine" (می ناب،
* میگلگون، ساقی می ده…), and Wikisource hosts a lot of exactly the poetry where that comes
* up. The word-list approach makes this far safer than a blind "می" + space replace, but it
* can't tell «می» the prefix from «می» the noun by context alone.
* 2. More fundamentally: on Wikisource, "می شود" written with a full space isn't necessarily an
* error to begin with. Older typesetting predates the ZWNJ/half-space convention entirely,
* so a full space between می and the verb may be exactly what the original book printed, not
* an OCR artifact - "fixing" it would mean silently changing the source's actual typesetting
* to match a modern convention it never used, which the whole point of a transcription
* project is to avoid.
*
* Given both of these, review before running this on a whole page - it's not something to
* reach for by default the way cleanup-ocr's encoding-level fixes are.
* @param {object} editor The script helpers for the page.
*/
var smartZwnj = function(editor) {
_runOnSelectionOrWholePage(editor, _applyZwnj);
};
});
// </nowiki>
t1saqvpte9qykkos8a1bkkw1czr7bbn
برگه:تاریخ سی ساله ایران - بیژن جزنی (جلد اول).pdf/۷۱
104
95540
299773
299534
2026-09-20T15:43:36Z
Hanooz
17889
299773
proofread-page
text/x-wiki
<noinclude><pagequality level="1" user="Hanooz" />
{{سصم|||تاریخ سی ساله ایران}}</noinclude>دوم - جبهه ملی و مصدق در جنبش ملی کردن نفت
محمد مصدق در یک خانواده اشرافی بزرگ شده و در فرانسه در رشته حقوق دکترا گرفته بود. مصدق پس از بازگشت به ایران تمایلات ملی از خود نشان داد و در پایان جنگ جهانی اول در جریان قرار داد ۱۹۱۹ چهره او آشکار شد ولی در این دوره جناح، ملیون چهره های سرشناس تری از مصدق دارد. در کودتای ۱۲۹۹ هنگامی که سید ضیاالدین نخست وزیر میشود و دستور بازداشت رجال قاجار را صادر میکند، مصدق والی فارس بود و هنگامی که به تهران احضار شد در اصفهان مسیر خود را عوض کرد و به چهار محال بختیاری رفت. پس از سقوط کابینه صد روزه و تشکیل کابینه قوام، مصدق پست وزارت دارائی را بعهده گرفت. با توجه به موضعی که ملیون در مقابل جنبشهای انقلابی آن دوره داشتند مصدق در مقابل جنبش<noinclude></noinclude>
tqjwk2zldwonvawnbfn08zp0lqwtggx
299775
299773
2026-09-20T18:25:42Z
Hanooz
17889
/* نمونهخوانیشده */
299775
proofread-page
text/x-wiki
<noinclude><pagequality level="3" user="Hanooz" />{{dhr|12em}}</noinclude>{{زیرخط|'''''دوم - جبهه ملی و مصدق در جنبش ملی کردن نفت'''''}}
محمد مصدق در یک خانواده اشرافی بزرگ شده و در فرانسه در رشته حقوق دکترا گرفته بود. مصدق پس از بازگشت به ایران تمایلات ملی از خود نشان داد و در پایان {{w|جنگ جهانی اول}} در جریان {{w|قرارداد ۱۹۱۹}} چهرهٔ او آشکار شد ولی در این دورهٔ جناح ملیون چهره های سرشناس تری از مصدق دارد. در {{w|کودتای ۳ اسفند ۱۲۹۹|کودتای ۱۲۹۹}} هنگامی که {{w|سید ضیاءالدین طباطبایی|سید ضیاالدین}} نخست وزیر میشود و دستور بازداشت رجال قاجار را صادر میکند، مصدق والی فارس بود و هنگامی که به تهران احضار شد در اصفهان مسیر خود را عوض کرد و به چهار محال بختیاری رفت. پس از سقوط {{w|کابینه سیاه|کابینه صد روزه}} و تشکیل کابینه قوام، مصدق پست وزارت دارائی را بعهده گرفت. با توجه به موضعی که ملیون در مقابل جنبشهای انقلابی آن دوره داشتند مصدق در مقابل جنبش<noinclude></noinclude>
rvb35wz5g7bvp01zn47odd6zic47vjh
برگه:تاریخ سی ساله ایران - بیژن جزنی (جلد اول).pdf/۷۲
104
95541
299776
299535
2026-09-20T18:35:06Z
Hanooz
17889
/* نمونهخوانیشده */
299776
proofread-page
text/x-wiki
<noinclude><pagequality level="3" user="Hanooz" />{{سرصفحهمستمر|۷۰||تاریخ سی ساله ایران}}</noinclude>جنگل و قیام {{w|محمدتقی پسیان|کلنل}} و مانند آن روشی اعتدالی ولی محافظه کارانه داشت (در مقابل رضاخان، ملیون مدتها روش محافظه کارانه داشتند.) {{w|سید حسن مدرس|مدرس}} که سرشناس ترین چهرهٔ این دوره بود در ابتدای امر سعی کرد از تضادهای رضا خان با دربار استفاده کند. ملیون نمی توانستند عواقب سیاست خود را پیش بینی کنند و موقعی واقعیت را درک کردند که دیگر دور کردن رضا خان از قدرت، در توانائی آنان نبود. مدرس و همکارانش میخواستند رضا خان را با "پولیتیک" از میدان بدر کنند ولی دستهایی که رضاخان را روی کار آورده بودند در استفاده از مانورهای دیپلماتیک و برپا کردن صحنه های فریبنده بر شاگردان مکتب دیپلماسی "{{w|میرزا علیاصغر اتابک|امین السطان}}" و "{{w|حسین پیرنیا (مؤتمنالملک)|موتمن الملک}}"<ref>مقصود یکی کردن امین السطان با موتمن الملک نیست، بلکه توجه دادن به روشهای سیاسی و دیپلماسی کهنه و فرسودهای بود که رجال آن دوره در هر سنی که بودند بکار میبستند. آثار این "دلقک بازی" تا سالهای پس از شهریور ۲۰ نیز در {{کذا|چهان|جهان}} سیاسی دیده میشود.</ref> استاد تر بودند. با همین معیارها بود که مصدق در مجلس سیاستمدارانه به مخالفت با سلطنت رضا خان میپردازد و اظهار میدارد که چون رضاخان (که سردار سپه و نخست وزیر بود) مرد لایق و کاردانی است و با نشستن او بر جایگاه سلطنت، ملت ایران نمیتواند از این لیاقت بهره مند شود. بهتر است رضاخان شاه شود و در همان سمتهای خود بماند سرانجام رضاخان حاکم شد و دیکتاتوری خود را برقرار کرد. مصدق خانه نشین شد و منتظر حوادث ماند. هنگامیکه متفقین وارد ایران شدند و رضا شاه سقوط کرد مصدق بیش از شصت سال داشت و این امتیاز را بر دیگران<noinclude>{{خطکش}}
{{پانویس}}</noinclude>
hknu5p9whjyyfii7mmbsis0q8qkc0xr
برگه:تاریخ سی ساله ایران - بیژن جزنی (جلد اول).pdf/۷۳
104
95542
299777
299536
2026-09-20T19:11:07Z
Hanooz
17889
/* نمونهخوانیشده */
299777
proofread-page
text/x-wiki
<noinclude><pagequality level="3" user="Hanooz" />
{{سصم|||تاریخ سی ساله ایران}}</noinclude>داشت که از دوره "{{w|مرتضیقلی صنیعالدوله|صنیع الدوله}}" و "موتمن الملک" تا مدرس در صف ملیون مبارزه کرده بود. برای مردم تهران بخصوص، برای قشرهای بازاری و کسبه که قدیمی تر بودند. (تهران در دوره رضاشاه رشد زیادی کرده بود و جمعیت زیادی از شهرستانها به تهران آمده بودند) باین ترتیب پس از شهریور ۲۰ مصدق دوباره پا به میدان گذاشت. در دوره چهاردهم مجلس شورای ملی مصدق از تهران نامزد شد و بعنوان اولین نماینده تهران انتخاب شد. در مجلس عده ای از نمایندگان منفرد و نمایندگان عضو حزب ایران گرد او جمع شده بودند و مصدق سخنگوی آنها بود. اولین مساله ای که در مجلس چهاردهم موضع مصدق را روشن کرد اعتراض به اعتبار نامهٔ سید ضیا الدین طباطبائی بود که مصدق طی سخنرانی بجای خود پرده از کودتای ۱۲۹۹ و نقش سید ضیا در آن برداشت. بر سر این مساله فراکسیون حزب توده، در کنار مصدق بود ولی مصدق مانورهائی در مجلس داد تا جدائی خود را از این فراکسیون آشکار سازد. در مجلس چهاردهم، امتیاز نامهٔ "استاندارد اویل کمپانی" مطرح شد که مصدق با آن مخالفت ورزید.<ref>سخنرانی معروف مصدق در مورد امتیازنامه استاندارد اویل کمپانی در ۷ آبان ۱۳۲۳ در مجلس {{کذا|چهاردم|چهاردهم}} انجام گرفته</ref> و نه فقط بر علیه آن بلکه بر علیه {{w|شرکت نفت ایران و انگلیس|شرکت نفت ایران و انگلیس}} و امتیاز نفت جنوب نیز صحبت کرد. هنوز مدتی از این ماجرا نگذشته بود که مساله امتیاز نفت شمال پیش آمد. در اینجا مصدق به روش خود ادامه داد. اما فراکسیون خود ادامه داد. اما فراکسیون حزب توده خطائی بزرگ مرتکب شد و از امتیاز نفت شمال دربست حمایت کرد. در اینجا مصدق و حزب توده در<noinclude>{{خطکش}}
{{پانویس}}</noinclude>
mv0ggsdj74ml7x9zubtccyd4kluvcoq
299778
299777
2026-09-20T19:11:25Z
Hanooz
17889
299778
proofread-page
text/x-wiki
<noinclude><pagequality level="3" user="Hanooz" />{{سرصفحهمستمر|تاریخ سی ساله ایران||۷۱}}</noinclude>داشت که از دوره "{{w|مرتضیقلی صنیعالدوله|صنیع الدوله}}" و "موتمن الملک" تا مدرس در صف ملیون مبارزه کرده بود. برای مردم تهران بخصوص، برای قشرهای بازاری و کسبه که قدیمی تر بودند. (تهران در دوره رضاشاه رشد زیادی کرده بود و جمعیت زیادی از شهرستانها به تهران آمده بودند) باین ترتیب پس از شهریور ۲۰ مصدق دوباره پا به میدان گذاشت. در دوره چهاردهم مجلس شورای ملی مصدق از تهران نامزد شد و بعنوان اولین نماینده تهران انتخاب شد. در مجلس عده ای از نمایندگان منفرد و نمایندگان عضو حزب ایران گرد او جمع شده بودند و مصدق سخنگوی آنها بود. اولین مساله ای که در مجلس چهاردهم موضع مصدق را روشن کرد اعتراض به اعتبار نامهٔ سید ضیا الدین طباطبائی بود که مصدق طی سخنرانی بجای خود پرده از کودتای ۱۲۹۹ و نقش سید ضیا در آن برداشت. بر سر این مساله فراکسیون حزب توده، در کنار مصدق بود ولی مصدق مانورهائی در مجلس داد تا جدائی خود را از این فراکسیون آشکار سازد. در مجلس چهاردهم، امتیاز نامهٔ "استاندارد اویل کمپانی" مطرح شد که مصدق با آن مخالفت ورزید.<ref>سخنرانی معروف مصدق در مورد امتیازنامه استاندارد اویل کمپانی در ۷ آبان ۱۳۲۳ در مجلس {{کذا|چهاردم|چهاردهم}} انجام گرفته</ref> و نه فقط بر علیه آن بلکه بر علیه {{w|شرکت نفت ایران و انگلیس|شرکت نفت ایران و انگلیس}} و امتیاز نفت جنوب نیز صحبت کرد. هنوز مدتی از این ماجرا نگذشته بود که مساله امتیاز نفت شمال پیش آمد. در اینجا مصدق به روش خود ادامه داد. اما فراکسیون خود ادامه داد. اما فراکسیون حزب توده خطائی بزرگ مرتکب شد و از امتیاز نفت شمال دربست حمایت کرد. در اینجا مصدق و حزب توده در<noinclude>{{خطکش}}
{{پانویس}}</noinclude>
psj5bqzx5wp5iz57k3qnaobiymhydld
برگه:تاریخ سی ساله ایران - بیژن جزنی (جلد اول).pdf/۷۴
104
95543
299779
299537
2026-09-20T19:17:54Z
Hanooz
17889
/* نمونهخوانیشده */
299779
proofread-page
text/x-wiki
<noinclude><pagequality level="3" user="Hanooz" />{{سرصفحهمستمر|۷۲||تاریخ سی ساله ایران}}</noinclude>مقابل هم قرار گرفتند. ولی مصدق برای اینکه مخالفتش با قرارداد حمل بر مخالفت با شوروی که در مقابل آلمان هیتلری میجنگید نشود، طی نامه ای به سفیر شوروی در ایران موضع خود را درباره نفت روشن کرد. مصدق در این نامه پس از تشریح رابطه ایران با شوروی و تائید فعالیت یعنی همزیستی و کمک برادرانه شوروی جوان به تهران، صرفنظر کردن از کلیه مطالبات دولت تزاری و الغای قراردادهای یک جانبه، به دولت شوروی توصیه میکرد که پیشنهاد امتیاز نفت شمال را پس بگیرد و بجای آن پیشنهادی برای کمک به دولت ایران برای استخراج نفت در شمال بدهد. فشردهٔ پیشنهاد مصدق این است که دولت شوروی وام برای (که همه از شوروی تهیه شود.) با این وام شوروی نفت شمال استخراج و انحصراً به نرخ مبادله (بنابر بازار روز) به شوروی فروخته میشود. وام شوروی و بهرهٔ عادلانه آن از این طریق مستهلک خواهد شد. این پیشنهادها که امروز برای دولت شوروی حداکثر چیزی است که ممکن است در یک کشور به آن برسد (با توجه به انحصار فروش محصول در شوروی) و در اغلب موارد بدون توجه به ماهیت رژیم طرف معامله خود قراردادهائی به مراتب سودمندتر از نظر کشور طرف قرارداد (مثلاً ایران با مصر و مانند آن) امضاء می کند، با بی اعتنائی شوروی ها روبرو شد و انعکاس {{کذا|ایادی|زیادی}} هم نیافت. موضع مصدق در ایران در این مورد نشان دهندهٔ موضع ملی اوست.
پس از تشکیل فرقهٔ دمکرات، مصدق خواسته های عمومی فرقه را در زمینه زبان مادری و مصرف مالیاتها و تشکیل انجمن ایالتی با روح قانون اساسی مطابق دانسته، خواستار رسیدگی به آنها شد. ولی هنگامیکه فرقه دمکرات حکومت را در دست گرفت و خودمختاری اعلام کرد و "توافق قوام"<noinclude></noinclude>
g14enpsmfyl2ll6hv13cko0cdawcr08
ویکینبشته:نامهای پدیدآورندگان
4
95582
299771
299747
2026-09-20T14:59:16Z
Hanooz
17889
299771
wikitext
text/x-wiki
{{process header
| title = نامهای پدیدآورندگان
| section =
| previous = [[ویکینبشته:جستارها]]
| next =
| shortcut =
| notes = خلاصهای کوتاه از دیدگاهها دربارهٔ نام پدیدآورندگان، قراردادهای نامگذاری، نامهای مستعار و استفاده از حروف اول نام.
}}{{جستار}}
تعیین نام درستی که باید برای صفحهٔ یک پدیدآورنده در فضای نام پدیدآورنده به کار برد، در برخی موارد که پدیدآورندهای با نامهای متفاوتی شناخته میشود — خواه با نام مستعار یا با گونههای مختلفی از نام واقعیاش — میتواند چندان ساده نباشد. هیچ سیاست ثابتی دربارهٔ اینکه کدام نام باید استفاده شود وجود ندارد. در حال حاضر، بهکارگیری قراردادها و شیوههای گوناگون به این معناست که در فضای پدیدآورندگانِ ویکینبشته یکدستی وجود ندارد. این صفحه برخی از دیدگاهها را دربارهٔ استفاده از نام پدیدآورندگان در ویکینبشته خلاصه میکند.
این بحث در نهایت به تقابل میان قراردادهای استاندارد کتابخانهای و مسائل مرتبسازی، در برابر شهرت عمومی و خواست پدیدآورنده باز میگردد.
==نامهای کامل==
استفاده از حروف اول نام در نام یک پدیدآورنده گاهی در ویکینبشته مورد مخالفت قرار میگیرد. استفاده از نام کامل پدیدآورندگان استاندارد شناختهشدهٔ بینالمللی برای کتابخانههاست. کتابخانهها برای فهرستنویسیِ آثار خود باید تعداد زیادی از پدیدآورندگان را ردهبندی کنند. اگر از حروف اول نام استفاده شود، ممکن است باعث ردهبندی نادرست پدیدآورندگان شود (در مورد ویکینبشته، ممکن است در صفحات رده و در [[ویکینبشته:پدیدآورندگان]] به ترتیب نادرست فهرست شوند). همچنین ممکن است در میان همهٔ پدیدآورندگان، نامها و حروف اول تکراری وجود داشته باشد که سردرگمی غیرضروری ایجاد میکند. در این مورد اخیر، ویکینبشته میتواند از صفحات ابهامزدایی استفاده کند، هرچند این کار ممکن است غیرضروری باشد یا تنها بهعنوان روشی پشتیبان، در کنار استفاده از نام درست، ترجیح داده شود.
ویکیپدیا از این قرارداد نامگذاری استفاده نمیکند، چون یک دانشنامه است نه یک کتابخانه، و بهطور استاندارد حروف اول نام را در عنوان مقالههای خود میگنجاند. این موضوع میتواند پیونددهی میان پروژهها را با مشکل مواجه کند.
برخی پدیدآورندگان ممکن است ترجیح داده باشند به شیوهای خاص شناخته شوند و بیشتر با همان نام معروفاند. برخی از کاربران ویکینبشته بر این باورند که پراستفادهترین نام باید در اولویت نخست قرار گیرد و سایر نامها بهصورت تغییرمسیر درآیند. کاربران دیگر معتقدند خواستهٔ پدیدآورنده دربارهٔ چگونگی شناختهشدنش باید محترم شمرده شود.
{| class=wikitable width="70%" {{ts|mc}}
! style="width:50%;" | نام کامل
! style="width:50%;" | نام رایج
|-
| [[پدیدآورنده:جیمز متیو بری]]
| [[پدیدآورنده:جی. ام. بری]]
|-
| [[پدیدآورنده:توماس استرنز الیوت]]
| [[پدیدآورنده:تی. اس. الیوت]]
|-
| [[پدیدآورنده:کلارنس مایکل جیمز استانیسلاوس دنیس]]
| [[پدیدآورنده:سی. جی. دنیس]]
|-
| [[پدیدآورنده:ادیت نزبیت]]
| [[پدیدآورنده:ای. نزبیت]]
|-
| [[پدیدآورنده:ویلیام متیو فلیندرز پیتری]]
| [[پدیدآورنده:فلیندرز پیتری]]
|-
| [[پدیدآورنده:ادلین ویرجینیا وولف]]
| [[پدیدآورنده:ویرجینیا وولف]]
|-
|
| [[پدیدآورنده:هومر]]
|}
===فنی===
الگوی {{tlx|حروف اول اسم}} را میتوان در صفحات پدیدآورندگان قرار داد تا نشان دهد در عنوان صفحه از حرف اول نام استفاده شده و چنانچه بازکردن آن حرف اول ممکن شود، صفحه باید جابهجا شود. اینگونه نمایش داده میشود:
{{حروف اول اسم|nocat=yes}}
اگر صفحهٔ یک پدیدآورنده با اطلاعات نام کامل بهروزرسانی شده، لطفاً در [[ویکینبشته:دفترخانه|دفترخانه]] اطلاع دهید تا جابهجا شود.
==نامهای دیگر==
بسیاری از پدیدآورندگان از نام مستعار، نام قلمی، لقب یا دیگر نامهای جعلی استفاده میکنند. برخی به دلایل گوناگون بیش از یک نام جایگزین به کار میبرند. برخی پدیدآورندگان به نام مستعار خود بیش از نام کاملشان شهرت دارند. این وضعیت شرایطی مشابه با استفاده از نامهای کامل ایجاد میکند.
{| class=wikitable width="70%" {{ts|mc}}
! style="width:50%;" | نام کامل
! style="width:50%;" | نام رایج
|-
| [[پدیدآورنده:ساموئل لنگهورن کلمنز]]
| rowspan="2" | [[پدیدآورنده:مارک تواین]]
|-
| [[پدیدآورنده:ساموئل کلمنز]]
|-
| [[پدیدآورنده:اریک آرتور بلر]]
| [[پدیدآورنده:جورج اورول]]
|}
در مواردی که یک پدیدآورنده چند نام مستعار دارد، میتوان نامهای دیگر را به نام کامل تغییرمسیر داد. در جایی که تنها یک نام مستعار وجود دارد، انتخاب همچنان میان نام کامل و نام مورداستناد محل بحث است. ویکیپدیا و دیگر پروژههای خواهر از نام رایج استفاده میکنند. بااینحال، نامهای مستعار مشکل مرتبسازی را در ویکینبشته حتی بزرگتر میکنند، چون ممکن است نام خانوادگی یکسان نباشد و این کار نام مستعار و نام واقعی را — بهجای صرفاً نامرتببودن در یک صفحه — بهطور کامل در صفحات جداگانه قرار میدهد.
عنوانها (مانند «پاپ»، «شاه»، «لرد»، «سر» و مانند آن) نباید در عنوان صفحه استفاده شوند، حتی اگر نام رایج پدیدآورنده باشند. در این موارد، عنوان همچنان میتواند در الگوی پدیدآورنده به کار رود یا در توضیحات ذکر شود.
{| class=wikitable width="70%" {{ts|mc}}
! style="width:50%;" | نام بدون عنوان
! style="width:50%;" | نام رایج
|-
| [[پدیدآورنده:جورج گوردون بایرون]]
| [[پدیدآورنده:لرد بایرون]]
|-
| [[پدیدآورنده:پیوس دوم]]
| [[پدیدآورنده:پاپ پیوس دوم]]
|-
| [[پدیدآورنده:ویکتوریای بریتانیا]]
| [[پدیدآورنده:ملکه ویکتوریا]]
|}
==یادداشتهای تکمیلی==
* بهصورت اختیاری، نامی که در سربرگ یک اثر استفاده میشود میتواند همان نامی باشد که در خودِ اثر آمده، حتی اگر با عنوان صفحهٔ پدیدآورنده یکسان نباشد. باید از یک تغییرمسیر برای پیونددادن نام پیوندشدهٔ ویکی به صفحهٔ درست استفاده کرد.
* هر نامی که برای صفحهٔ پدیدآورنده انتخاب شود، همهٔ نامهای دیگر باید بهصورت صفحات تغییرمسیر به صفحهٔ اصلی پدیدآورنده ساخته شوند.
* هنگام ابهامزدایی میان دو پدیدآورنده با نام کامل یکسان با استفاده از اطلاعات درونپرانتز، تاریخهای شناختهشدهٔ تولد و مرگ را بر دیگر اطلاعات ترجیح دهید؛ برای نمونه Author:William Ellery Channing ('''1818-1901''')، بهجای Author:William Ellery Channing (poet). برای چنین عنوانهایی، میان تاریخها از یک خطتیرهٔ ساده استفاده کنید.
==همچنین ببینید==
*[[ویکینبشته:شیوهنامه#صفحات پدیدآورندگان]]
*[[الگو:پدیدآورنده]]
*[[:رده:پدیدآورندگان با حروف اول ناشناخته]]
===نمایهها===
*[[ویکینبشته:پدیدآورندگان]]
*[[:رده:پدیدآورندگان]]
[[رده:پدیدآورندگان با حروف اول ناشناخته| ]]
dafnx4wgtwlqb2m8wix6ru1kpijxzze
برگه:فرهنگ جغرافیایی ایران جلد دوم.pdf/۳۲۸
104
95583
299774
299749
2026-09-20T18:04:30Z
Hanooz
17889
299774
proofread-page
text/x-wiki
<noinclude><pagequality level="1" user="204.18.193.198" /></noinclude>{{حس|خالی}}<noinclude><references/></noinclude>
qq4jnc9gglbflvflyaz44jtydryqjk7
الگو:جستار
10
95585
299767
2026-09-20T14:34:44Z
Hanooz
17889
صفحهای تازه حاوی «{{ambox | image = [[پرونده:Text-x-generic_with_pencil.svg|40px]] | text = این یک [[ویکینبشته:جستارها|جستار]] است؛ شامل توصیهها و/یا دیدگاههای یک یا چند مشارکتکنندهٔ ویکینبشته است. این صفحه یک [[ویکینبشته:سیاستها و رهنمودها|سیاست یا رهنمود]] '''نیست''' و ویرایشگرا...» ایجاد کرد
299767
wikitext
text/x-wiki
{{ambox
| image = [[پرونده:Text-x-generic_with_pencil.svg|40px]]
| text = این یک [[ویکینبشته:جستارها|جستار]] است؛ شامل توصیهها و/یا دیدگاههای یک یا چند مشارکتکنندهٔ ویکینبشته است. این صفحه یک [[ویکینبشته:سیاستها و رهنمودها|سیاست یا رهنمود]] '''نیست''' و ویرایشگران ملزم به پیروی از آن نیستند.
{{#if:{{{1|}}}{{{2|}}}{{{3|}}}{{{4|}}}{{{5|}}}
| <br />{{shortcut
|{{{1|}}}|{{{2|}}}|{{{3|}}}|{{{4|}}}|{{{5|}}}
}}
}}
}}<includeonly>{{#ifeq:{{NAMESPACE}}|{{ns:4}}|{{{category|[[رده:جستارهای ویکینبشته|{{PAGENAME}}]]}}}}}{{#ifeq:{{NAMESPACE}}|{{ns:2}}|{{{category|[[رده:جستارهای کاربران|{{PAGENAME}}]]}}}}}</includeonly><noinclude>{{documentation}}</noinclude>
rp6gogwmrx1we78i60jvapc5az5o8o4
299768
299767
2026-09-20T14:36:15Z
Hanooz
17889
ترجمه هوش مصنوعی
299768
wikitext
text/x-wiki
{{ambox
| image = [[پرونده:Text-x-generic_with_pencil.svg|40px]]
| text = این یک [[ویکینبشته:جستارها|جستار]] است؛ شامل توصیهها و/یا دیدگاههای یک یا چند مشارکتکنندهٔ ویکینبشته است. این صفحه یک [[ویکینبشته:سیاستها و رهنمودها|سیاست یا رهنمود]] '''نیست''' و ویرایشگران ملزم به پیروی از آن نیستند.
{{#if:{{{1|}}}{{{2|}}}{{{3|}}}{{{4|}}}{{{5|}}}
| <br />{{shortcut
|{{{1|}}}|{{{2|}}}|{{{3|}}}|{{{4|}}}|{{{5|}}}
}}
}}
}}<includeonly>{{#ifeq:{{NAMESPACE}}|{{ns:4}}|{{{category|[[رده:جستارهای ویکینبشته|{{PAGENAME}}]]}}}}}{{#ifeq:{{NAMESPACE}}|{{ns:2}}|{{{category|[[رده:جستارهای کاربران|{{PAGENAME}}]]}}}}}</includeonly><noinclude>{{documentation}}</noinclude>
i530a4rykch1ux5x1qypcw8lagrv0n8
الگو:حروف اول اسم
10
95586
299770
2026-09-20T14:52:27Z
Hanooz
17889
ترجمه هوش مصنوعی
299770
wikitext
text/x-wiki
{{#if: {{{justification|}}}
| {{ombox
|type=notice
|text='''این صفحهٔ پدیدآورنده در نام موضوع خود از حروف اختصاری استفاده میکند. این کار قابلقبول دانسته شده است، زیرا:''' <br />
{{smaller block|{{{justification}}}}}
}}<includeonly>{{category handler
| all = [[رده:پدیدآورندگانی که حروف اختصاری نامشان نیازی به شناسایی ندارد]]
| nocat = {{{nocat|}}}
}}</includeonly>
| {{ombox
|type=content
|text='''این صفحهٔ پدیدآورنده در نام موضوع خود، بهجای نوشتن کامل هر یک از نامها، از حروف اختصاری استفاده میکند. ''' <br />
{{smaller block|برای انتقال این صفحهٔ پدیدآورنده به عنوان کامل و درست آن، پژوهش بیشتری لازم است؛ پس از آن میتوان این اعلان را حذف کرد.}}
}}<includeonly>{{category handler
| all = [[رده:پدیدآورندگان دارای حروف اختصاری شناسایینشده]]
| nocat = {{{nocat|}}}
}}</includeonly>
}}<noinclude>{{documentation}}</noinclude>
5rcpk8j5mph2k6u7653tfepst0tugdb