Files
pdf/audit_results_final_pretty.tmp
T

1934 lines
232 KiB
Plaintext
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
{
"value": [
{
"document": "govinfo.pdf",
"target": "docx",
"input_bytes": 185839,
"source_pages": 1,
"source_chars": 3487,
"source_words": 545,
"source_images": 0,
"elapsed_s": 3.9358,
"cpu_s": 10.5,
"cpu_to_wall": 2.668,
"rss_start_mb": 83.45,
"rss_peak_mb": 722.56,
"output_bytes": 39815,
"fidelity": "lossy",
"quality": 0.8483641387543525,
"media": "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
"output": {
"text": "SUBCHAPTER A—GENERAL \n2.2 Administrative Committee of the Fed\nPART 1—DEFINITIONS \neral Register. \n2.3 Office of the Federal Register; location; AUTHORITY: 44 U.S.C. 1506; sec. 6, E.O. 10530, office hours. \n19 FR 2709; 3 CFR, 19541958 Comp., p.189. 2.4 General authority of Director. \n2.5 Publication of statutes, regulations, and \n§ 1.1 Definitions. \nrelated documents. \nAs used in this chapter, unless the 2.6 Unrestricted use. \ncontext requires otherwise— \nAdministrative Committee means the 19 FR 2709; 3 CFR, 19541958 Comp., p. 189; 1 Administrative Committee of the Fed U.S.C. 112; 1 U.S.C. 113. \nAUTHORITY: 44 U.S.C. 1506; sec. 6, E.O. 10530, \neral Register established under section \nSOURCE: 37 FR 23603, Nov. 4, 1972, unless \n1506 of title 44, United States Code; otherwise noted. \nAgency means each authority, wheth\ner or not within or subject to review by § 2.1 Scope and purpose. \nanother agency, of the United States, \nother than the Congress, the courts, (a) This chapter sets forth the poli\nthe District of Columbia, the Common cies, procedures, and delegations under \nwealth of Puerto Rico, and the terri which the Administrative Committee tories and possessions of the United of the Federal Register carries out its States; \ngeneral responsibilities under chapter \nDocument includes any Presidential 15 of title 44, United States Code. proclamation or Executive order, and (b) A primary purpose of this chapter \nany rule, regulation, order, certificate, is to inform the public of the nature code of fair competition, license, no and uses of Federal Register publica\ntice, or similar instrument issued, pre tions. \nscribed, or promulgated by an agency; \nDocument having general applicability § 2.2 A dministrative Committee of the \nand legal effect means any document Federal Register. \nissued under proper authority pre (a) The Administrative Committee of \nscribing a penalty or course of conduct, the Federal Register is established by conferring a right, privilege, authority, section 1506 of title 44, United States or immunity, or imposing an obliga Code. \ntion, and relevant or applicable to the \n(b) The Committee consists of— \n(1) The Archivist, or Acting Archi\ngeneral public, members of a class, or \npersons in a locality, as distinguished \nfrom named individuals or organiza v st, o t e U ted States, w o s t ei f h ni h i h tions; and Chairman;\nFiling means making a document (2) An officer of the Department of available for public inspection at the Justice designated by the Attorney \nOffice of the Federal Register during General; and \nofficial business hours. A document is (3) The Public Printer or Acting Pub filed only after it has been received, lic Printer. \nprocessed and assigned a publication (c) The Director of the Federal Reg\ndate according to the schedule in part ister is the Secretary of the Com 17 of this chapter. \nmittee. \nRegulation and rule have the same (d) Any material required by law to \nmeaning. \nbe filed with the Committee, and any \n[37 FR 23603, Nov. 4, 1972, as amended at 50 correspondence, inquiries, or other ma\nBthe Federal Register. \nOJ\n$$_Sec. \nhit \n2w \nNI5 \n07\nWRF\nOM\nVn \noe \nny\nterial intended for the Committee or \nFR 12466, Mar. 28, 1985] \nwhich relate to Federal Register publi cations shall be sent to the Director of \nPART 2—GENERAL INFORMA\nTION \n2.1 Scope and purpose. \nkapVerDate Sep\u003c11\u003e2014 15:54 Mar 19, 2019 Jkt 247004 PO 00000 Frm 00015 Fmt 8010 Sfmt 8010 Q:\\01\\1V1.TXT PC31",
"paragraphs": 66,
"mean_paragraph_words": 8.53,
"tables": [
],
"table_shapes": [
],
"images": 0,
"page_breaks": 0
},
"content": {
"expected_tokens": 563,
"actual_tokens": 599,
"matched_tokens": 529,
"recall": 0.939609,
"precision": 0.883139,
"order": 0.495697,
"f1": 0.910499,
"duplicate_tokens": 6,
"missing_count": 34,
"extra_count": 70,
"missing": [
"5",
"the",
"the",
"the",
"administrative",
"committee",
"of",
"federal",
"federal",
"register",
"united",
"whether",
"commonwealth",
"territories",
"notice",
"prescribed",
"prescribing",
"obligation",
"public",
"organizations",
"is",
"information",
"policies",
"publications",
"publications",
"archivist",
"who",
"material",
"with",
"verdate",
"pc31kpayne",
"on"
],
"extra": [
"a",
"fed",
"fed",
"eral",
"eral",
"u",
"s",
"e",
"o",
"o",
"wheth",
"er",
"poli",
"common",
"cies",
"wealth",
"terri",
"tories",
"no",
"publica",
"tice",
"pre",
"pre",
"tions",
"tions",
"scribed",
"dministrative",
"scribing",
"obliga",
"tion",
"tion",
"archi"
]
},
"source_derived_table_match": null,
"visual": {
"output_pages": 3,
"compared_pages": 1,
"pixel_similarity": 0.9505
},
"warnings": [
"Layout fidelity is lossy; headers/footers/fonts are not fully preserved.",
"auto: text_based; ocr=never",
"layout_ml=onnx",
"Page 1: layout_ml=onnx regions=31",
"ocr_policy=never — skipped OCR (digital text layer kept).",
"Low text precision vs source PDF - output may contain duplicated or invented content (precision=0.88).",
"Low text order similarity vs source PDF - columns/paragraphs may be reordered (order=0.50)."
],
"warning_count": 7,
"error": null,
"models": {
"onnx_providers": [
"AzureExecutionProvider",
"CPUExecutionProvider"
],
"ocr_available": true,
"ocr_ar_status": "on",
"layout_model": {
"path": "gateway\\models\\layout\\v1\\layout.onnx",
"bytes": 130502330,
"sha256": "250dbad1dfb9e4983fab75e1bf5085cd56ec3f41d5c7d0f8623ec74856e7aa67"
}
}
},
{
"document": "irs_f1040_sample.pdf",
"target": "docx",
"input_bytes": 220237,
"source_pages": 2,
"source_chars": 10415,
"source_words": 2224,
"source_images": 0,
"elapsed_s": 4.1979,
"cpu_s": 12.6875,
"cpu_to_wall": 3.022,
"rss_start_mb": 718.03,
"rss_peak_mb": 1119.82,
"output_bytes": 43485,
"fidelity": "lossy",
"quality": 0.8712558711299784,
"media": "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
"output": {
"text": "rm Department of the Treasury—Internal Revenue Service \n2\nFo1040 U.S. Individual Income Tax Return \n025 OMB No. 1545-0074 IRS Use Only—Do not write or staple in this space. \n, 20 See separate instructions.\nFo rthe yea rJan . 1Dec .31 ,2025 ,o rothe rtax yea rbeginning \n, 2025, ending \nFiled pursuant to section 301.9100-2 Combat zone \nDeceased MM / DD / YYYY Spouse MM / DD / YYYY\nOther\nYour first name and middle initial \nLast name \nYour social security number \nLast name \nIf joint return, spouses first name and middle initia l\nHome address (number and street). If you have a P.O. box, see instructions. \nSpouses socia lsecurity number Check here if your main home, and your Apt. no. \nspouses if filing a joint return, was in \nthe U.S. for more than half of 2025. \nPresidential Election Campaign\nCheck here if you ,o ryou rspouse if filing jointly ,want $3 to go to this fund .Checking a box below \nZIP code\nState \nCity, town, or post office. If you have a foreign address, also complete spaces below. \nForeign province/state/co\nForeign country name \nForeign postal code \nunty \nwil lnot change you rtax o rrefund.\nYou Spouse \nHead of household (HOH)\nFiling Status Single \nMarried filing jointly (even if only one had income) \nQualifying surviving spouse (QSS)\nCheck onl \ny\nMarried filing separately (MFS). Enter spouses SSN above \nIf you checked the HOH or QSS box, enter the childs name \nif the qualifying person is a child but not your dependent:\none box. \nand full name here:\nIf treating a nonresident alien or dual-status alien spouse as a U.S. resident for the entire tax year, check the box and enter their \nname (see instructions and attach statement if required):\nDi ital Assets At any time during 2025, did you: (a) receive (as a reward, award, or payment for property or services); or (b) sell, g\nexchange, or otherwise dispose of a digital asset (or a financial interest in a digital asset)? (See instructions.) . . Yes No\nDependentsDependent 1 Dependent 2 Dependent 3 Dependent 4 (see instructions) (1) First name\n(2) Last name\nIf more \nthan four (3) SSN\ndependents, (4) Relationship\nsee instructions and check with you more (5) Check if lived(a) Yes(a) Yes(a) Yes(a) Yes\nhere . . than half of 2025 ( )b And in th . .eUS( )b And in th . .eUS( )b And in th . .eUS(b) And in th . .eUS\n(6) Check if Full-time Permanently Full-time Permanently Full-time Permanently Full-time Permanently \nstudent and totally student and totally\ndisableddisableddisableddisabled\n(7) Credits Child tax Credit for Child tax Credit for Child tax Credit for Child tax Credit for \ncreditother creditother creditother credit other \ndependentsdependentsdependentsdependents\nCheck if your filing status is MFS or HOH and you lived apart from your spouse for the last 6 months of 2025, or you are legally \nseparated according to your state law under a written separation agreement or a decree of separate maintenance and you did not \nlive in the same household as your spouse at the end of 2025. \nIncome 1 a Total amount from Form(s) W-2, box 1 (see instructions) . . . . . . . . . . . . . 1a\nAttach Forms b Household employee wages not reported on Form(s) W-2 . . . . . . . . . . . . . 1b( ) W-2 here. Also c Tip income not reported on line 1a (see instructions) . . . . . . . . . . . . . . 1c W-2G and d Medicaid waiver payments not reported on Form(s) W 2 (see instructions) . . . . . . . . 1dattach Forms -1099-R if tax e Taxable dependent care benefits from Form 2441, line 26 . . . . . . . . . . . . 1e\nwas withheld. \nIf ou did not f Employer-provided adoption benefits from Form 8839, line 31 . . . . . . . . . . . 1fy get a Form g Wages from Form 8919, line 6 . . . . . . . . . . . . . . . . . . . . . 1g W-2, see instructions. h Other earned income (see instructions). Enter type and amount :1h\ni Nontaxable combat pay election (see instructions) . . . . . . . 1i\nz Add lines 1a through 1h . . . . . . . . . . . . . . . . . . . . . . 1z\nAttach Sch. B 2a Tax-exempt interest . . . 2a\nif required. 3a Qualified dividends . . . 3a\nc Chec kif you rchilds dividends are included in 1Line 3a 2 Line 3b\n4a IRA distributions . . . . 4a\nc Check if (see instructions) . . . . . 1Rollove r2 QCD 3\n5a Pensions and annuities . . 5a\nc Check if (see instructions) . . . . . 1Rollove r2 PSO 3\n6a Social security benefits . . 6a\nc If you elect to use the lump-sum election method, check here (see instructions) . . . . .\nd If you are married filing separately and lived apart from you rspouse the entire yea r(see inst.) ,chec khere 7a Capital gain or (loss). Attach Schedule D if required . . . . . . . . . . . . . . 7a b Check if : Schedule D not required Includes childs capita l gain o r(loss)\n8 Additional income from Schedule 1, line 10 . . . . . . . . . . . . . . . . . 8 9 Add lines 1z, 2b, 3b, 4b, 5b, 6b, 7a, and 8. This is your total income . . . . . . . . . 9 10 Adjustments to income from Schedule 1, line 26 . . . . . . . . . . . . . . . 10 11a Subtract line 10 from line 9. This is your adjusted gross income . . . . . . . . . . 11a For Disclosure, Privacy Act, and Paperwork Reduction Act Notice, see separate instructions. Cat. No. 11320B Form 1040 (2025) Created 9/5/25\nForm 1040 (2025) Page 2\nCredits 12a Someone can claim \nStandard \nMarried filing \nMarried filing \nsurviving \nHead of \nIf you checked \no r12d ,see inst.\nPayments 25 Federal income tax withheld from:\nand \nCredits\nIf you have a \nqualifying child, \nenter their SSN (see instructions):\nSee instructions.d Account number\nYou Owe\nThird Part Do you want to allow another person to discuss this return with the IRS? See instructions. yYes. Complete below. No\nDesignee Designees Phone Personal identification \nSi n Under penalties of perjury, I declare that I have examined this return and accompanying schedules and statements, and to the best of my knowledge and gbelief, they are true, correct, and complete. Declaration of preparer (other than taxpayer) is based on all information of which preparer has any knowledge.\nname no. number (PIN) \nHere \nYour signature Date Your occupation If the IRS sent you an Identity \nProtection PIN, enter it here \nJoint return? Spouse\nSee instructions. \nouses \nu t i n. \nDat bo\ns sg\no ti n \nccupa o\ne\nturn, th m\n. If a joint re\ns signature\nSp\nyour records. Keep a copy for (see inst.) \nPhone no. Email address \nPaid Preparers name Preparers signature Date PTIN Check if: Preparer Self-employed\n(see inst.) \nIf the IRS sent your spouse an \nIdentity Protection PIN, enter it here \nUse Only Firms name \nPhone no. \nFirms EIN \nFirms address \nGo to www.irs.gov/Form1040 for instructions and the latest information .\nForm 1040 (2025) \nb Taxable interest\n. . . . .\n2b\nb Ordinary dividends\n. . . . .\n3b\nb Taxable amount .\n. . . . .\n4b\nb Taxable amount .\n. . . . .\n5b\nb Taxable amount .\n. . . . .\n6b\nTax and 11b Amount from line 11a (adjusted gross income) . .\n. . . .\n.\n. .\n11b\nYou as a dependent c b Spouse itemizes on a separate return d You: Were born before January 2, 1961 Spouse: Was born before January 2, 1961\nYour spouse as a dependent You were a dual-status alien Are blind Is blind\n\n\n\ndeduction for—e Standard deduction oritemized deductions (from Sc\nhedule A). . . . .\n.\n. .\n12e\n• Single or 13a Qualified business income deduction from Form 8995\nor Form 8995-A . . .\n.\n. .\n13a\nseparately, b Additional deductions from Schedule 1-A, line 38 .\n. . . .\n.\n. .\n13b\n$15,750 14 Add lines 12e, 13a, and 13b . . . . . . .\n. . . .\n.\n. .\n14\njointly or 15 Subtract line 14 from line 11b. If zero or less, enter -0-.\nThis is your taxable income\n.\n. .\n15\nQualifying 16 Tax (see instructions). Check if any from Form(s): 1\n8814 2 4972\n3\n\n16\nspouse, 17 Amount from Schedule 2, line 3 . . . . . .\n. . . .\n.\n. .\n17\n$31,500 18 Add lines 16 and 17 . . . . . . . . . .\n. . . .\n.\n. .\n18\nhousehold, 19 Child tax credit or credit for other dependents from\nSchedule 8812 . . . .\n.\n. .\n19\n$23,625 20 Amount from Schedule 3, line 8 . . . . . .\n. . . .\n.\n. .\n20\na box on line 21 Add lines 19 and 20 . . . . . . . . . .\n. . . .\n.\n. .\n21\n12a ,12b ,12c, 22 Subtract line 21 from line 18. If zero or less, enter -0-\n. . . .\n.\n. .\n22\n23 Other taxes, including self-employment tax, from Schedule\n2, line 21 . . .\n.\n. .\n23\n24 Add lines 22 and 23. This is your total tax . . . a Form(s) W-2 . . . . . . . . . . . . Refundable b Form(s) 1099 . . . . . . . . . . . . c Other forms (see instructions) . . . . . . . d Add lines 25a through 25c . . . . . . . .\n. . . . . . . . . . 25a . . . . . . 25b . . . . . . 25c . . . .\n. .\n. . . .\n24 25d\n26 2025 estimated tax payments and amount applied from If you made estimated tax payments with your former you may need to 27a Earned income credit (EIC) . . . . . . . . attach Sch .EIC. b Clergy filing Schedule SE (see instructions) . . . c If you do not want to claim the EIC, check here . . 28 Additional child tax credit (ACTC) from Schedule 8812 to claim the ACTC, check here . . . . . . .\n2024 return . . . . spouse in 2025, . . . . . . 27a . . . . . . . . . . . . . . . . . If you do not want . . . . . 28\n. . .\n. . . . .\n26\n29 American opportunity credit from Form 8863, line 8 .\n. . . . . . 29\n\n\n\n30 Refundable adoption credit from Form 8839, line 13\n. . . . . . 30\n\n\n\n31 Amount from Schedule 3, line 15 . . . . . .\n. . . . . . 31\n\n\n\n32 Add lines 27a, 28, 29, 30, and 31. These are your total\nother payments and\nrefundable\ncredits .\n32\n33 Add lines 25d, 26, and 32. These are your total payments Refund 34 If line 33 is more than line 24, subtract line 24 from line 35a Amount of line 34 you want refunded to you. If Form Direct deposit? b Routing number\n. . . 33. This is the amount you 8888 is attached, check here c Type: Checking\n. .\n. . overpaid . . . . . Savings\n33 34 35a\n36 Amount of line 34 you want applied to your 2026 Amount 37 Subtract line 33 from line 24. This is the amount you For details on how to pay, go to www.irs.gov/Payments\nestimated tax . . . 36 owe. or see instructions . .\n.\n. .\n37\n38 Estimated tax penalty (see instructions) . . . .\n. . . . . . 38\n\n\n",
"paragraphs": 127,
"mean_paragraph_words": 10.11,
"tables": [
[
[
"b Taxable interest",
". . . . .",
"2b"
],
[
"b Ordinary dividends",
". . . . .",
"3b"
],
[
"b Taxable amount .",
". . . . .",
"4b"
],
[
"b Taxable amount .",
". . . . .",
"5b"
],
[
"b Taxable amount .",
". . . . .",
"6b"
]
],
[
[
"Tax and 11b Amount from line 11a (adjusted gross income) . .",
". . . .",
".",
". .",
"11b"
],
[
"You as a dependent c b Spouse itemizes on a separate return d You: Were born before January 2, 1961 Spouse: Was born before January 2, 1961",
"Your spouse as a dependent You were a dual-status alien Are blind Is blind",
"",
"",
""
],
[
"deduction for—e Standard deduction oritemized deductions (from Sc",
"hedule A). . . . .",
".",
". .",
"12e"
],
[
"• Single or 13a Qualified business income deduction from Form 8995",
"or Form 8995-A . . .",
".",
". .",
"13a"
],
[
"separately, b Additional deductions from Schedule 1-A, line 38 .",
". . . .",
".",
". .",
"13b"
],
[
"$15,750 14 Add lines 12e, 13a, and 13b . . . . . . .",
". . . .",
".",
". .",
"14"
],
[
"jointly or 15 Subtract line 14 from line 11b. If zero or less, enter -0-.",
"This is your taxable income",
".",
". .",
"15"
],
[
"Qualifying 16 Tax (see instructions). Check if any from Form(s): 1",
"8814 2 4972",
"3",
"",
"16"
],
[
"spouse, 17 Amount from Schedule 2, line 3 . . . . . .",
". . . .",
".",
". .",
"17"
],
[
"$31,500 18 Add lines 16 and 17 . . . . . . . . . .",
". . . .",
".",
". .",
"18"
],
[
"household, 19 Child tax credit or credit for other dependents from",
"Schedule 8812 . . . .",
".",
". .",
"19"
],
[
"$23,625 20 Amount from Schedule 3, line 8 . . . . . .",
". . . .",
".",
". .",
"20"
],
[
"a box on line 21 Add lines 19 and 20 . . . . . . . . . .",
". . . .",
".",
". .",
"21"
],
[
"12a ,12b ,12c, 22 Subtract line 21 from line 18. If zero or less, enter -0-",
". . . .",
".",
". .",
"22"
],
[
"23 Other taxes, including self-employment tax, from Schedule",
"2, line 21 . . .",
".",
". .",
"23"
],
[
"24 Add lines 22 and 23. This is your total tax . . . a Form(s) W-2 . . . . . . . . . . . . Refundable b Form(s) 1099 . . . . . . . . . . . . c Other forms (see instructions) . . . . . . . d Add lines 25a through 25c . . . . . . . .",
". . . . . . . . . . 25a . . . . . . 25b . . . . . . 25c . . . .",
". .",
". . . .",
"24 25d"
],
[
"26 2025 estimated tax payments and amount applied from If you made estimated tax payments with your former you may need to 27a Earned income credit (EIC) . . . . . . . . attach Sch .EIC. b Clergy filing Schedule SE (see instructions) . . . c If you do not want to claim the EIC, check here . . 28 Additional child tax credit (ACTC) from Schedule 8812 to claim the ACTC, check here . . . . . . .",
"2024 return . . . . spouse in 2025, . . . . . . 27a . . . . . . . . . . . . . . . . . If you do not want . . . . . 28",
". . .",
". . . . .",
"26"
],
[
"29 American opportunity credit from Form 8863, line 8 .",
". . . . . . 29",
"",
"",
""
],
[
"30 Refundable adoption credit from Form 8839, line 13",
". . . . . . 30",
"",
"",
""
],
[
"31 Amount from Schedule 3, line 15 . . . . . .",
". . . . . . 31",
"",
"",
""
],
[
"32 Add lines 27a, 28, 29, 30, and 31. These are your total",
"other payments and",
"refundable",
"credits .",
"32"
],
[
"33 Add lines 25d, 26, and 32. These are your total payments Refund 34 If line 33 is more than line 24, subtract line 24 from line 35a Amount of line 34 you want refunded to you. If Form Direct deposit? b Routing number",
". . . 33. This is the amount you 8888 is attached, check here c Type: Checking",
". .",
". . overpaid . . . . . Savings",
"33 34 35a"
],
[
"36 Amount of line 34 you want applied to your 2026 Amount 37 Subtract line 33 from line 24. This is the amount you For details on how to pay, go to www.irs.gov/Payments",
"estimated tax . . . 36 owe. or see instructions . .",
".",
". .",
"37"
],
[
"38 Estimated tax penalty (see instructions) . . . .",
". . . . . . 38",
"",
"",
""
]
]
],
"table_shapes": [
[
5,
3
],
[
24,
5
]
],
"images": 0,
"page_breaks": 1
},
"content": {
"expected_tokens": 1667,
"actual_tokens": 1663,
"matched_tokens": 1563,
"recall": 0.937612,
"precision": 0.939868,
"order": 0.667267,
"f1": 0.938739,
"duplicate_tokens": 17,
"missing_count": 104,
"extra_count": 100,
"missing": [
"form1040",
"2025u",
"s",
"s",
"s",
"s",
"s",
"s",
"tax",
"tax",
"return",
"the",
"the",
"the",
"the",
"the",
"only",
"not",
"or",
"or",
"or",
"or",
"or",
"or",
"for",
"year",
"year",
"year",
"jan",
"1",
"1",
"1"
],
"extra": [
"rm",
"fo1040",
"025",
"fo",
"rthe",
"yea",
"yea",
"yea",
"rjan",
"o",
"o",
"o",
"o",
"o",
"o",
"o",
"rothe",
"rtax",
"rtax",
"rbeginning",
"initia",
"l",
"l",
"you",
"you",
"socia",
"lsecurity",
"ryou",
"rspouse",
"rspouse",
"co",
"unty"
]
},
"source_derived_table_match": null,
"visual": {
"output_pages": 7,
"compared_pages": 2,
"pixel_similarity": 0.8913
},
"warnings": [
"Layout fidelity is lossy; headers/footers/fonts are not fully preserved.",
"auto: text_based; ocr=never",
"layout_ml=onnx",
"Page 1: layout_ml=onnx regions=2",
"Page 1: ML table-\u003egrid rows=5 cols=3 conf=0.97.",
"Page 2: layout_ml=onnx regions=4",
"Page 2: ML table-\u003egrid rows=36 cols=5 conf=0.76.",
"ocr_policy=never — skipped OCR (digital text layer kept).",
"No source-derived grid to compare against; table score is structural (cells checked against the source text)."
],
"warning_count": 9,
"error": null,
"models": {
"onnx_providers": [
"AzureExecutionProvider",
"CPUExecutionProvider"
],
"ocr_available": true,
"ocr_ar_status": "on",
"layout_model": {
"path": "gateway\\models\\layout\\v1\\layout.onnx",
"bytes": 130502330,
"sha256": "250dbad1dfb9e4983fab75e1bf5085cd56ec3f41d5c7d0f8623ec74856e7aa67"
}
}
},
{
"document": "mozilla_pdf_spec_excerpt.pdf",
"target": "docx",
"input_bytes": 1016315,
"source_pages": 14,
"source_chars": 82701,
"source_words": 13452,
"source_images": 90,
"elapsed_s": 30.3707,
"cpu_s": 94.5625,
"cpu_to_wall": 3.114,
"rss_start_mb": 1081.18,
"rss_peak_mb": 1140.94,
"output_bytes": 77695,
"fidelity": "lossy",
"quality": 0.8956293227806041,
"media": "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
"output": {
"text": "Trace-based Just-in-Time Type Specialization for Dynamic\nLanguages\nAndreas Gal+, Brendan Eich, Mike Shaver, David Anderson, David Mandelin, Mohammad R. Haghighat$, Blake Kaplan, Graydon Hoare, Boris Zbarsky, Jason Orendorff, Jesse Ruderman, Edwin Smith#, Rick Reitmaier#, Michael Bebenita+, Mason Chang+#, Michael Franz+\nMozilla Corporation\n{gal,brendan,shaver,danderson,dmandelin,mrbkap,graydon,bz,jorendorff,jruderman}@mozilla.com\nAdobe Corporation# {edwsmith,rreitmai}@adobe.com\nIntel Corporation$ {mohammad.r.haghighat}@intel.com University of California, Irvine+ {mbebenit,changm,franz}@uci.edu\nAbstract\nDynamic languages such as JavaScript are more difficult to com pile than statically typed ones. Since no concrete type information is available, traditional compilers need to emit generic code that can handle all possible type combinations at runtime. We present an al ternative compilation technique for dynamically-typed languages that identifies frequently executed loop traces at run-time and then generates machine code on the fly that is specialized for the ac tual dynamic types occurring on each path through the loop. Our method provides cheap inter-procedural type specialization, and an elegant and efficient way of incrementally compiling lazily discov ered alternative paths through nested loops. We have implemented a dynamic compiler for JavaScript based on our technique and we have measured speedups of 10x and more for certain benchmark programs.\nand is used for the application logic of browser-based productivity applications such as Google Mail, Google Docs and Zimbra Col laboration Suite. In this domain, in order to provide a fluid user experience and enable a new generation of applications, virtual ma chines must provide a low startup time and high performance.\nCompilers for statically typed languages rely on type informa tion to generate efficient machine code. In a dynamically typed pro gramming language such as JavaScript, the types of expressions may vary at runtime. This means that the compiler can no longer easily transform operations into machine instructions that operate on one specific type. Without exact type information, the compiler must emit slower generalized machine code that can deal with all potential type combinations. While compile-time static type infer ence might be able to gather type information to generate opti mized machine code, traditional static analysis is very expensive and hence not well suited for the highly interactive environment of\nCategoriesa nd SubjectD escriptors D.3.4 [ProgrammingL ana web browser. guages]: Processors— Incremental compilers, code generation.\nWe present a trace-based compilation technique for dynamic languages that reconciles speed of compilation with excellent per formance of the generated machine code. Our system uses a mixed mode execution approach: the system starts running JavaScript in a fast-starting bytecode interpreter. As the program runs, the system identifies hot (frequently executed) bytecode sequences, records them, and compiles them to fast native code. We call such a se quence of instructions a trace.\nGeneral Terms Design, Experimentation, Measurement, Perfor mance.\nKeywords JavaScript,j ust-in-time compilation, trace trees.\n1. Introduction\nUnlike method-based dynamic compilers, our dynamic com piler operates at the granularity of individual loops. This design choice is based on the expectation that programs spend most of their time in hot loops. Even in dynamically typed languages, we expect hot loops to be mostly type-stable, meaning that the types of values are invariant. (12) For example, we would expect loop coun ters that start as integers to remain integers for all iterations. When both of these expectations hold, a trace-based compiler can cover the program execution with a small number of type-specialized, ef ficiently compiled traces.\nDynamic languages such as JavaScript, Python, and Ruby, are pop ular since they are expressive, accessible to non-experts, and make deployment as easy as distributing a source file. They are used for small scripts as well as for complex applications. JavaScript, for example, is the de facto standard for client-side web programming\nEach compiled trace covers one path through the program with one mapping of values to types. When the VM executes a compiled trace, it cannot guarantee that the same path will be followed or that the same types will occur in subsequent loop iterations.\nPLDI09, June 1520, 2009, Dublin, Ireland.\nCopyright c 2009 ACM 978-1-60558-392-1/09/06. . . $5.00\nflow is different, or a value of a different type is generated), the6 } trace exits. If an exit becomes hot, the VM can record a branch trace starting at the exit to cover the new path. In this way, the VM records a trace tree covering all the hot paths through the loop.\nN¨ested loops can be difficult to optimize for tracing VMs. Insnippet.\nFigure 1. Sample program: sieve of Eratosthenes. primes is initialized to an array of 100 false values on entry to this code a naıve implementation, inner loops would become hot first, and\nthe VM would start tracing there. When the inner loop exits, the\nVM would detect that a different branch was taken. The VM would\ntry to record a branch trace, and find that the trace reaches not the\ninner loop header, but the outer loop header. At this point, the VMBIntterpretyecodesOverhead\nthus tracing the outer loop inside a trace tree for the inner loop.could continue tracing until it reaches the inner loop header again,elodop Interpreting\nBut this requires tracing a copy of the outer loop for every side exit\nand type combination in the inner loop. In essence, this is a formrecaobrodritn gMonitorcompreileadd ytrace of unintended tail duplication, which can easily overflow the code\ncache. Alternatively, the VM could simply stop tracing, and give upLRIRe cTorarcdeloohpo/etxitCoped racemEilnte rT on ever tracing outer loops.\nWe solve the nested loop problem by recording n¨ested trace\ntrees. Our system traces the inner loop exactly as the naıve version.\nThe system stops extending the inner tree when it reaches an outerLCIRo mTrpaicleeComExielecu Tteracp d e loop, but then it starts a new trace at the outer loop header. When\nthe outer loop reaches the inner loop header, the system tries to call\nthe trace tree for the inner loop. If the call succeeds, the VM records\nthe call to the inner tree as part of the outer trace and finishesCopiled TracemLeave the outer trace as normal. In this way, our system can trace any\nnumber of loops nested to any depth without causing excessive tail\nSymbol Key \nge\ncoldlo/bolpa/cekxliistted Native\nloofipn isheh aadt er\nlosoapm eed gtyep wesith \nno esxidiset inegx ittr,ace\nexsiidstein egx titr atoc e\nduplication.\nThese techniques allow a VM to dynamically translate a pro gram to nested, type-specialized trace trees. Because traces can cross function call boundaries, our techniques also achieve the ef fects of inlining. Because traces have no internal control-flowj oins, they can be optimized in linear time by a simple compiler (10). Thus, our tracing VM efficiently performs the same kind of op timizations that would require interprocedural analysis in a static optimization setting. This makes tracing an attractive and effective tool to type specialize even complex function call-rich code.\nFigure 2. State machine describing the major activities of Trace Monkey and the conditions that cause transitions to a new activ ity. In the dark box, TM executes JS as compiled traces. In the light gray boxes, TM executes JS in the standard interpreter. White boxes are overhead. Thus, to maximize performance, we need to maximize time spent in the darkest box and minimize time spent in the white boxes. The best case is a loop where the types at the loop\nedge are the same as the types on entrythen TM can stay in native code until the loop is done.\nWe implemented these techniques for an existing JavaScript in terpreter, SpiderMonkey. We call the resulting tracing VM Trace Monkey. TraceMonkey supports all the JavaScript features of Spi derMonkey, with a 2x-20x speedup for traceable programs.\na set of industry benchmarks. The paper ends with conclusions in Section 9 and an outlook on future work is presented in Section 10.\nThis paper makes the following contributions:\n2. Overview: Example Tracing Run\n• We explain an algorithm for dynamically forming trace trees to\ncover a program, representing nested loops as nested trace trees.\nThis section provides an overview of our system by describing how TraceMonkey executes an example program. The example program, shown in Figure 1, computes the first 100 prime numbers with nested loops. The narrative should be read along with Figure 2, which describes the activities TraceMonkey performs and when it transitions between the loops.\n• We explain how to speculatively generate efficient type-specialized\ncode for traces from dynamic language programs.\n• We validate our tracing techniques in an implementation based\non the SpiderMonkey JavaScript interpreter, achieving 2x-20x speedups on many programs.\nraceMonkey always begins executing a program in the byte code interpreter. Every loop back edge is a potential trace point. When the interpreter crosses a loop edge, TraceMonkey invokes the trace monitor, which may decide to record or execute a native trace. At the start of execution, there are no compiled traces yet, so the trace monitor counts the number of times each loop back edge is executed until a loop becomes hot, currently after 2 crossings. Note that the way our loops are compiled, the loop edge is crossed before entering the loop, so the second crossing occurs immediately after\nT\nThe remainder of this paper is organized as follows. Section 3 is a general overview of trace tree based compilation we use to cap ture and compile frequently executed code regions. In Section 4 we describe our approach of covering nested loops using a num ber of individual trace trees. In Section 5 we describe our trace compilation based speculative type specialization approach we use to generate efficient machine code from recorded bytecode traces.\nOur implementation of a dynamic type-specializing compiler forthe first iteration. JavaScript is described in Section 6. Related work is discussed in Section 8. In Section 7 we evaluate our dynamic compiler based oniteration:\nHere is the sequence of events broken down by outer loop\nv0 := ld state[748] // load primes from the trace activation record\nst sp[0], v0 // store primes to interpreter stack v1 := ld state[764] // load k from the trace activation record\nv2 := i2f(v1)\n // convert k from int to double\nst sp[8], v1 // store k to interpreter stack st sp[16], 0 // store false to interpreter stack\n // load class word for primes\nv3 := ld v0[4]\nv4 := and v3, -4\n // m ask out object class tag for primes v5 := eq v4, Array // test whether primes is an array\n // side exit if v5 is false\nxf v5\nv6 : js_Array_set(v0, v2, false) // call function to set array element =\nv7 := eq v6, 0\n // test return value from call\n // side exit if js_Array_set returns false. xt v7\nFigure 3. LIR snippet for sample program. This is the LIR recorded for line 5 of the sample program in Figure 1. The LIR encodes\nthe semantics in SSA form using temporary variables. The LIR also encodes all the stores that the interpreter would do to its data stack.\nSometimes these stores can be optimized away as the stack locations are live only on exits to the interpreter. Finally, the LIR records guards\nand side exits to verify the assumptions made in this recording: that primes is an array and that the call to set its element succeeds.\nmov edx, ebx(748) // load primes from the trace activation record\n // (*) store primes to interpreter stack mov edi(0), edx\nmov esi, ebx(764) // load k from the trace activation record\n// (*) store k to interpreter stack // (*) store false to interpreter stack // (*) load object class word for primes // (*) m ask out object class tag for primes // (*) test whether primes is an array // (*) side exit if primes is not an array // bump stack for call alignment convention // push last argument for call\nmov edi(8), esi mov edi(16), 0 mov eax, edx(4) and eax, -4\ncmp eax, Array jne side_exit_1 sub esp, 8\npush false\n // push first argument for call\npush esi\ncall js_Array_set // call function to set array element\n // clean up extra stack space\nadd esp, 8\n// (*) created by register allocator // (*) test return value of js_Array_set // (*) side exit if call failed\nmov ecx, ebx test eax, eax je side_exit_2 ...\nside_exit_1:\nmov ecx, ebp(-4) // restore ecx mov esp, ebp jmp epilog\n // restore esp\n // jump to ret statement\nFigure 4. x86 snippet for sample program. This is the x86 code compiled from the LIR snippet in Figure 3. Most LIR instructions compile\nto a single x86 instruction. Instructions marked with (*) would be omitted by an idealized compiler that knew that none of the side exits\nwould ever be taken. The 17 instructions generated by the compiler compare favorably with the 100+ instructions that the interpreter would\nexecute for the same code snippet, including 4 indirectj umps.\ni=2. This is the first iteration of the outer loop. The loop on lines 4-5 becomes hot on its second iteration, so TraceMonkey en ters recording mode on line 4. In recording mode, TraceMonkey records the code along the trace in a low-level compiler intermedi ate representation we call LIR. The LIR trace encodes all the oper ations performed and the types of all operands. The LIR trace also encodes guards, which are checks that verify that the control flow and types are identical to those observed during trace recording. Thus, on later executions, if and only if all guards are passed, the trace has the required program semantics.\ninterpreter PC and the types of values match those observed when trace recording was started. The first trace in our example, T45, covers lines 4 and 5. This trace can be entered if the PC is at line 4, i and k are integers, and primes is an object. After compiling T45, TraceMonkey returns to the interpreter and loops back to line 1.\ni=3. Now the loop header at line 1 has become hot, so Trace Monkey starts recording. When recording reaches line 4, Trace Monkey observes that it has reached an inner loop header that al ready has a compiled trace, so TraceMonkey attempts to nest the inner loop inside the current trace. The first step is to call the inner trace as a subroutine. This executes the loop on line 4 to completion and then returns to the recorder. TraceMonkey verifies that the call was successful and then records the call to the inner trace as part of the current trace. Recording continues until execution reaches line 1, and at which point TraceMonkey finishes and compiles a trace for the outer loop, T16.\nTraceMonkey stops recording when execution returns to the loop header or exits the loop. In this case, execution returns to the loop header on line 4.\nAfter recording is finished, TraceMonkey compiles the trace to native code using the recorded type information for optimization. The result is a native code fragment that can be entered if the\nA trace records all its intermediate values in a small activation record area. To make variable accesses fast on trace, the trace also imports local and global variables by unboxing them and copying them to its activation record. Thus, the trace can read and write these variables with simple loads and stores from a native activation recording, independently of the boxing mechanism used by the interpreter. When the trace exits, the VM boxes the values from this native storage location and copies them back to the interpreter\ni=4. On this iteration, TraceMonkey calls T16. Because i=4, the if statement on line 2 is taken. This branch was not taken in the original trace, so this causes T16 to fail a guard and take a side exit. The exit is not yet hot, so TraceMonkey returns to the interpreter, which executes the continue statement.\ni=5. TraceMonkey calls T16, which in turn calls the nested trace T45. T16 loops back to its own header, starting the next iteration without ever returning to the monitor.\ni=6. On this iteration, the side exit on line 2 is taken again. Thisstructures. time, the side exit becomes hot, so a trace T23,1 is recorded that covers line 3 and returns to the loop header. Thus, the end of T23,1 jumps directly to the start of T16. The side exit is patched so that on future iterations, itj umps directly to T23,1.\nF\nor every controlflow branch in the source program, the recorder generates conditional exit LIR instructions. These instruc tions exit from the trace if required control flow is different from what it was at trace recording, ensuring that the trace instructions are run only if they are supposed to. We call these instructions\nAt this point, TraceMonkey has compiled enough traces to cover\nthe entire nested loop structure, so the rest of the program runsguard instructions. entirely as native code.\nMost of our traces represent loops and end with the special loop LIR instruction. This isj ust an unconditional branch to the top of the trace. Such traces return only via guards.\nNow, we describe the key optimizations that are performed as part of recording LIR. All of these optimizations reduce complex dynamic language constructs to simple typed constructs by spe cializing for the current trace. Each optimization requires guard in structions to verify their assumptions about the state and exit the trace if necessary.\n3. Trace Trees\nIn this section, we describe traces, trace trees, and how they are formed at run time. Although our techniques apply to any dynamic language interpreter, we will describe them assuming a bytecode interpreter to keep the exposition simple.\nType specialization.\nAll LIR primitives apply to operands of specific types. Thus,\n3.1 Traces\nIR traces are necessarily typespecialized, and a compiler can easily produce a translation that requires no type dispatches. A typical bytecode interpreter carries tag bits along with each value, and to perform any operation, must check the tag bits, dynamically dispatch, mask out the tag bits to recover the untagged value, perform the operation, and then reapply tags. LIR omits everything except the operation itself.\nA trace is simply a program path, which may cross function call boundaries. TraceMonkey focuses on loop traces, that originate at a loop edge and represent a single iteration through the associated loop.\nL\nSimilar to an extended basic block, a trace is only entered at the top, but may have many exits. In contrast to an extended basic block, a trace can containj oin nodes. Since a trace always only follows one single path through the original program, however,j oin nodes are not recognizable as such in a trace and have a single predecessor node like regular nodes.\nA potential problem is that some operations can produce values of unpredictable types. For example, reading a property from an object could yield a value of any type, not necessarily the type observed during recording. The recorder emits guard instructions that conditionally exit if the operation yields a value of a different type from that seen during recording. These guard instructions guarantee that as long as execution is on trace, the types of values match those of the typed trace. When the VM observes a side exit along such a type guard, a new typed trace is recorded originating at the side exit location, capturing the new type of the operation in\nA typed trace is a trace annotated with a type for every variable (including temporaries) on the trace. A typed trace also has an entry type map giving the required types for variables used on the trace before they are defined. For example, a trace could have a type map (x: int, b: boolean), meaning that the trace may be entered only if the value of the variable x is of type int and the value of b is of type boolean. The entry type map is much like the signature of a function.question.\nIn this paper, we only“ discu”ss typed loop traces, and we will refer to them simply as traces . The key property of typed loop traces is that they can be compiled to efficient machine code using the same techniques used for typed languages.\nRepresentation specialization: objects. In JavaScript, name lookup semantics are complex and potentially expensive because they include features like object inheritance and eval. To evaluate an object property read expression like o.x, the interpreter must search the property map of o and all of its prototypes and parents. Property maps can be implemented with different data structures (e.g., per-object hash tables or shared hash tables), so the search process also must dispatch on the representation of each object found during search. TraceMonkey can simply observe the result of the search process and record the simplest possible LIR to access the property value. For example, the search might finds the value of o.x in the prototype of o, which uses a shared hash-table represen tation that places x in slot 2 of a property vector. Then the recorded can generate LIR that reads o.x withj ust two or three loads: one to get the prototype, possibly one to get the property value vector, and one more to get slot 2 from the vector. This is a vast simplification and speedup compared to the original interpreter code. Inheritance relationships and object representations can change during execu tion, so the simplified code requires guard instructions that ensure the object representation is the same. In TraceMonkey, objects rep-\nIn TraceMonkey, traces are recorded in trace-flavored SSA LIR (low-level intermediate representation). In trace-flavored SSA (or TSSA), phi nodes appear only at the entry point, which is reached both on entry and via loop edges. The important LIR primitives are constant values, memory loads and stores (by address and offset), integer operators, floating-point operators, function calls, and conditional exits. Type conversions, such as integer to double, are represented by function calls. This makes the LIR used by TraceMonkey independent of the concrete type system and type conversion rules of the source language. The LIR operations are generic enough that the backend compiler is language independent. Figure 3 shows an example LIR trace.\nBytecode interpreters typically represent values in a various complex data structures (e.g., hash tables) in a boxed format (i.e., with attached type tag bits). Since a trace is intended to represent efficient code that eliminates all that complexity, our traces oper ate on unboxed values in simple variables and arrays as much as possible.\nStarting a tree. Tree trees always start at loop headers, because they are a natural place to look for hot paths. In TraceMonkey, loop headers are easy to detectthe bytecode compiler ensures that a bytecode is a loop header iff it is the target of a backward branch. TraceMonkey starts a tree when a given loop header has been exe cuted a certain number of times (2 in the current implementation). Starting a treej ust means starting recording a trace for the current point and type map and marking the trace as the root of a tree. Each tree is associated with a loop header and type map, so there may be several trees for a given loop header.\nresentations are assigned an integer key called the object shape. Thus, the guard is a simple equality check on the object shape.\nRepresentation specialization: numbers. JavaScript has no integer type, only a Number typ“e that is” the set of 64-bit IEEE 754 floating-pointer numbers ( doubles ). But many JavaScript operators, in particular array accesses and bitwise operators, really operate on integers, so they first convert the number to an integer, and then convert any integer result back to a double.1 Clearly, a JavaScript VM that wants to be fast must find a way to operate on integers directly and avoid these conversions.\nClosing the loop. Trace recording can end in several ways.\nIn TraceMonkey, we support two representations for numbers: integers and doubles. The interpreter uses integer representations as much as it can, switching for results that can only be represented as doubles. When a trace is started, some values may be imported and represented as integers. Some operations on integers require guards. For example, adding two integers can produce a value too large for the integer representation.\nIdeally, the trace reaches the loop header where it started with the same type map as on entry. This is called a type-stable loop iteration. In this case, the end of the trace canj ump right to the beginning, as all the value representations are exactly as needed to enter the trace. Thej ump can even skip the usual code that would copy out the state at the end of the trace and copy it back in to the trace activation record to enter a trace.\nFunction inlining. LIR traces can cross function boundaries in either direction, achieving function inlining. Move instructions need to be recorded for function entry and exit to copy arguments in and return values out. These move statements are then optimized away by the compiler using copy propagation. In order to be able to return to the interpreter, the trace must also generate LIR to record that a call frame has been entered and exited. The frame entry and exit LIR savesj ust enough information to allow the intepreter call stack to be restored later and is much simpler than the interpreters standard call code. If the function being entered is not constant (which in JavaScript includes any call by function name), the recorder must also emit LIR to guard that the function is the same.\nIn certain cases the trace might reach the loop header with a different type map. This scenario is sometime observed for the first iteration of a loop. Some variables inside the loop might initially be undefined, before they are set to a concrete type during the first loop iteration. When recording such an iteration, the recorder cannot link the trace back to its own loop header since it is type-unstable. Instead, the iteration is terminated with a side exit that will always fail and return to the interpreter. At the same time a new trace is recorded with the new type map. Every time an additional type unstable trace is added to a region, its exit type map is compared to the entry map of all existing traces in case they complement each other. With this approach we are able to cover type-unstable loop iterations as long they eventually form a stable equilibrium.\nGuards and side exits. Each optimization described above\nFinally, the trace might exit the loop before reaching the loop header, for example because execution reaches a break or return statement. In this case, the VM simply ends the trace with an exit\nrequires one or more guards to verify the assumptions made in doing the optimization. A guard isj ust a group of LIR instructions that performs a test and conditional exit. The exit branches to a\nside exit, a small off-trace piece of LIR that returns a pointer toto the trace monitor. a structure that describes the reason for the exit along with the\nAs mentioned previously, we may speculatively chose to rep resent certain Number-typed values as integers on trace. We do so when we observe that Number-typed variables contain an integer value at trace entry. If during trace recording the variable is unex pectedly assigned a non-integer value, we have to widen the type of the variable to a double. As a result, the recorded trace becomes inherently type-unstable since it starts with an integer value but ends with a double value. This represents a mis-speculation, since at trace entry we specialized the Number-typed value to an integer, assuming that at the loop edge we would again find an integer value in the variable, allowing us to close the loop. To avoid future spec ulative failures involving this variable, and to obtain a type-stable trace we note the fact that the variable in question as been observed to sometimes hold“ non-in”teger values in an advisory data structure which we call the oracle .\ninterpreter PC at the exit point and any other data needed to restore the interpreters state structures.\nAborts. Some constructs are difficult to record in LIR traces. For example, eval or calls to external functions can change the program state in unpredictable ways, making it difficult for the tracer to know the current type map in order to continue tracing. A tracing implementation can also have any number of other limi tations, e.g.,a small-memory device may limit the length of traces. When any situation occurs that prevents the implementation from continuing trace recording, the implementation aborts trace record ing and returns to the trace monitor.\n3.2 Trace Trees\nWhen compiling loops, we consult the oracle before specializ ing values to integers. Speculation towards integers is performed only if no adverse information is known to the oracle about that particular variable. Whenever we accidentally compile a loop that is type-unstable due to mis-speculation of a Number-typed vari able, we immediately trigger the recording of a new trace, which based on the now updated oracle information will start with a dou ble value and thus become type stable.\nEspecially simple loops, namely those where control flow, value types, value representations, and inlined functions are all invariant, can be represented by a single trace. But most loops have at least some variation, and so the program will take side exits from the main trace. When a side exit becomes hot, TraceMonkey starts a new branch trace from that point and patches the side exit toj ump directly to that trace. In this way, a single trace expands on demand to a single-entry, multiple-exit trace tree.\nThis section explains how trace trees are formed during execu tion. The goal is to form trace trees during execution that cover all the hot paths of the program.\nExtending a tree. Side exits lead to different paths through the loop, or paths with different types or representations. Thus, to completely cover the loop, the VM must record traces starting at all side exits. These traces are recorded much like root traces: there is a counter for each side exit, and when the counter reaches a hotness threshold, recording starts. Recording stops exactly as for the root trace, using the loop header of the root trace as the target to reach.\nOur implementation does not extend at all side exits. It extends only if the side exit is for a control-flow branch, and only if the side exit does not leave the loop. In particular we do not want to extend a trace tree along a path that leads to an outer loop, because we want to cover such paths in an outer tree through tree nesting.\n3.3 BlacklistingGuard\nSometimes, a program follows a path that cannot be compiled into a trace, usually because of limitations in the implementation. TraceMonkey does not currently support recording throwing and catching of arbitrary exceptions. This design trade off was chosen, because exceptions are usually rare in JavaScript. However, if a program opts to use exceptions intensively, we would suddenly incur a punishing runtime overhead if we repeatedly try to record a trace for this path and repeatedly fail to do so, since we abort tracing every time we observe an exception being thrown.\nT\nTree Anchor Trunk Trace Trace Anchor Branch Trace\nSide Exit\nFigure 5. A tree with two traces, a trunk trace and one branch trace. The trunk trace contains a guard to which a branch trace was\nAs a result, if a hot loop contains traces that always fail, the VM could potentially run much more slowly than the base interpreter: the VM repeatedly spends time trying to record traces, but is never able to run any. To avoid this problem, whenever the VM is about to start tracing, it must try to predict whether it will finish the trace.\nattached. The branch trace contain a guard that may fail and trigger a side exit. Both the trunk and the branch trace loop back to the tree\nanchor, which is the beginning of the trace tree.\nOur prediction algorithm is based on blacklisting traces that\nhave been tried and failed. When the VM fails to finish a trace startTrace 1 Trace 2 Trace 1 Trace 2\ning at a given point, the VM records that a failure has occurred. TheNumberBoolean NumberBoolean VM also sets a counter so that it will not try to record a trace starting\nat that point until it is passed a few more times (32 in our imple\nmentation). This backoff counter gives temporary conditions that\nprevent tracing a chance to end. For example, a loop may behave Number Boolean Number\nter a given number of failures (2 in our implementation), the VMdifferently during startup than during its steady-state execution. Af Closed Linked Linked\n(a) (b)\nmarks the fragment as blacklisted, which means the VM will never\nagain start recording at that point.\nsmall loops that get blacklisted, the system can spend a noticeableAfter implementing this basic strategy, we observed that forNumberBoolean String\namount of timej ust finding the loop fragment and determining that\nit has been blacklisted. We now avoid that problem by patching the\nbytecode. We define an extra no-op bytecode that indicates a loopNumber String String header. The VM calls into the trace monitor every time the inter\npreter executes a loop header no-op. To blacklist a fragment, weLinkedLinked Linked Closed simply replace the loop header no-op with a regular no-op. Thus,\nthe interpreter will never again even call into the trace monitor.\nThere is a related problem we have not yet solved, which occurs when a loop meets all of these conditions:\nTrace 1\nTrace 2\nTrace 3\nString\nFigure 6. We handle type-unstable loops by allowing traces to compile that cannot loop back to themselves due to a type mis\n• The VM can form at least one root trace for the loop.\n• There is at least one hot side exit for which the VM cannot\ncomplete a trace.\n• The loop body is short.\nmatch. As such traces accumulate, we attempt to connect their loop edges to form groups of trace trees that can execute without having to side-exit to the interpreter to cover odd type cases. This is par ticularly important for nested trace trees where an outer tree tries to call an inner tree (or in this case a forest of inner trees), since inner\nloops frequently have initially undefined values which change type\nIn this case, the VM will repeatedly pass the loop header, search for a trace, find it, execute it, and fall back to the interpreter. With a short loop body, the overhead of finding and calling the trace is high, and causes performance to be even slower than the basic interpreter. So far, in this situation we have improved the implementation so that the VM can complete the branch trace. But it is hard to guarantee that this situation will never happen. As future work, this situation could be avoided by detecting and blacklisting loops for which the average trace call executes few bytecodes before returning to the interpreter.\nto a concrete value after the first iteration.\nthrough the inner loop, {i2, i3, i5, α}. The α symbol is used to indicate that the trace loops back the tree anchor.\nWhen execution leaves the inner loop, the basic design has two choices. First, the system can stop tracing and give up on compiling the outer loop, clearly an undesirable solution. The other choice is to continue tracing, compiling traces for the outer loop inside the inner loops trace tree.\nFor example, the program might exit at i5 and record a branch trace that incorporates the outer loop: {i5, i7, i1, i6, i7, i1, α}. Later, the program might take the other branch at i2 and then exit, recording another branch trace incorporating the outer loop: {i2, i4, i5, i7, i1, i6, i7, i1, α}. Thus, the outer loop is recorded and compiled twice, and both copies must be retained in the trace cache.\n4. Nested Trace Tree Formation\nFigure 7 shows basic trace tree compilation (11) applied to a nested loop where the inner loop contains two paths. Usually, the inner loop (with header at i2) becomes hot first, and a trace tree is rooted at that point. For example, the first recorded trace may be a cycle\ni1t1 Outer Treei1 t1 Nested Tree i2i2t2\nTree Call\nNested Tree\ni6 i3 i4t2i3\ni4t4Exit Guard i5\ni5\ni7Exit Guard\ni6\nFigure 8. Control flow graph of a loop with two nested loops (left) and its nested trace tree configuration (right). The outer tree calls the two inner nested trace trees and places guards at their side exit with an if statem\ninside the inner most loop (a). An inner tre“e cap”tures the innerlocations.\nent\nFigure 7. Control flow graph of a nested loop\nloop, and is nested inside an outer tree which calls the inner tree.\nThe inner tree returns to the outer tree once it exits along its loop\ncondition guard (b).\nloop is entered with m different type maps (on geometric average), then we compile O(mk) copies of the innermost loop. As long as m is close to 1, the resulting trace trees will be tractable.\nAn important detail is that the call to the inner trace tree must act like a function call site: it must return to the same point every time. The goal of nesting is to make inner and outer loops independent; thus when the inner tree is called, it must exit to the same point in the outer tree every time with the same type map. Because we cannot actually guarantee this property, we must guard on it after the call, and side exit if the property does not hold. A common reason for the inner tree not to return to the same point would be if the inner tree took a new side exit for which it had never compiled a trace. At this point, the interpreter PC is in the inner tree, so we cannot continue recording or executing the outer tree. If this happens during recording, we abort the outer trace, to give the inner tree a chance to finish growing. A future execution of the outer tree would then be able to properly finish and record a call to the inner tree. If an inner tree side exit happens during execution of a compiled trace for the outer tree, we simply exit the outer trace and start recording a new branch in the inner tree.\nIn general, if loops are nested to d¨epth k, and each loop has n paths (on geometric average), this naıve strategy yields O(nk) traces, which can easily fill the trace cache.\nIn order to execute programs with nested loops efficiently, a tracing system needs a technique for covering the nested loops with native code without exponential trace duplication.\n4.1 Nesting Algorithm\nThe key insight is that if each loop is represented by its own trace tree, the code for each loop can be contained only in its own tree, and outer loop paths will not be duplicated. Another key fact is that we are not tracing arbitrary bytecodes that might have irreduceable control flow graphs, but rather bytecodes produced by a compiler for a language with structured control flow. Thus, given two loop edges, the system can easily determine whether they are nested and which is the inner loop. Using this knowledge, the system can compile inner and outer loops separately, and make the outer loops traces call the inner loops trace tree.\nThe algorithm for building nested trace trees is as follows. We start tracing at loop headers exactly as in the basic tracing system.\n4.2 Blacklisting with Nesting\nWhen we exit a loop (detected by comparing the interpreter PC with the range given by the loop edge), we stop the trace. The key step of the algorithm occurs when we are recording a trace for loop LR (R for loop being recorded) and we reach the header of a different loop LO (O for other loop). Note that LO must be analgorithm. inner loop of LR because we stop the trace when we exit a loop.\nThe blacklisting algorithm needs modification to work well with nesting. The problem is that outer loop traces often abort during startup (because the inner tree is not available or takes a side exit), which would lead to their being quickly blacklisted by the basic\nThe key observation is that when an outer trace aborts because the inner tree is not ready, this is probably a temporary condition. Thus, we should not count such aborts toward blacklisting as long as we are able to build up more traces for the inner tree.\n• If LO has a type-matching compiled trace tree, we call LO as\na nested trace tree. If the call succeeds, then we record the call in the trace for LR. On future executions, the trace for LR will call the inner trace directly.\nIn our implementation, when an outer tree aborts on the inner tree, we increment the outer trees blacklist counter as usual and back off on compiling it. When the inner tree finish“es a trace”, we decrement the blacklist counter on the outer loop, forgiving the outer loop for aborting previously. We also undo the backoff so that the outer tree can start immediately trying to compile the next time we reach it.\n• If LO does not have a type-matching compiled trace tree yet,\nwe have to obtain it before we are able to proceed. In order to do this, we simply abort recording the first trace. The trace monitor will see the inner loop header, and will immediately start recording the inner loop. 2\nIf all the loops in a nest are type-stable, then loop nesting creates no duplication. Otherwise, if loops are nested to a depth k, and each\n5. Trace Tree Optimization\nThis section explains how a recorded trace is translated to an optimized machine code trace. The trace compilation subsystem,\nNANOJIT, is separate from the VM and can be used for other\napplications.\n5.1 OptimizationsTag JS Type Description\nBecause traces are in SSA form and have noj oin points or φ nodes, certain optimizations are easy to implement. In order to get good startup performance, the optimizations must run quickly, so we chose a small set of optimizations. We implemented the optimizations as pipelined filters so that they can be turned on and off independently, and yet all run inj ust two loop passes over the trace: one forward and one backward.\nnull, or\nundefined\nEvery time the trace recorder emits a LIR instruction, the in struction is immediately passed to the first filter in the forward pipeline. Thus, forward filter optimizations are performed as the trace is recorded. Each filter may pass each instruction to the next filter unchanged, write a different instruction to the next filter, or write no instruction at all. For example, the constant folding filter can replace a multiply instruction like v13 := mul3, 1000 with a constant instruction v13 = 3000.\nFigure 9. Tagged values in the SpiderMonkey JS interpreter.\nTesting tags, unboxing (extracting the untagged value) and boxing (creating tagged values) are significant costs. Avoiding these costs is a key benefit of tracing.\nWe currently apply four forward filters:\nheuristic selects v with minimum vm. The motivation is that this frees up a register for as long as possible given a single spill.\nIf we need to spill a value vs at this point, we generate the restore codej ust after the code for the current instruction. The corresponding spill code is generatedj ust after the last point where vs was used. The register that was assigned to vs is marked free for the preceding code, because that register can now be used freely without affecting the following code\n• On ISAs without floating-point instructions, a soft-float filter\nconverts floating-point LIR instructions to sequences of integer instructions.\n• CSE (constant subexpression elimination),\n• expression simplification, including constant folding and a few\nalgebraic identities (e.g., a a = 0), and\n• source language semantic-specific expression simplification,6. Implementation\nprimarily algebraic identities that allow DOUBLE to be replaced\nTo demonstrate the effectiveness of our approach, we have im plemented a trace-based dynamic compiler for the SpiderMonkey JavaScript Virtual Machine (4). SpiderMonkey is the JavaScript VM embedded in Mozillas Firefox open-source web browser (2), which is used by more than 200 million users world-wide. The core of SpiderMonkey is a bytecode interpreter implemented in C++.\nwith INT. For example, LIR that converts an INT to a DOUBLE\nand then back again would be removed by this filter.\nWhen trace recording is completed, nanojit runs the backward optimization filters. These are used for optimizations that require backward program analysis. When running the backward filters, nanojit reads one LIR instruction at a time, and the reads are passed through the pipeline.\nIn SpiderMonkey, all JavaScript values are represented by the type jsval. A jsval is machine word in which up to the 3 of the least significant bits are a type tag, and the remaining bits are data. See Figure 6 for details. All pointers contained in jsvals point to GC-controlled blocks aligned on 8-byte boundaries.\nWe currently apply three backward filters:\nJavaScript object values are mappings of string-valued property names to arbitrary values. They are represented in one of two ways in SpiderMonkey. Most objects are represented by a shared struc tural description, called the object shape, that maps property names to array indexes using a hash table. The object stores a pointer to the shape and the array of its own property values. Objects with large, unique sets of property names store their properties directly\n• Dead data-stack store elimination. The LIR trace encodes many\nstores to locations in the interpreter stack. But these values are never read back before exiting the trace (by the interpreter or another trace). Thus, stores to the stack that are overwritten before the next exit are dead. Stores to locations that are off the top of the interpreter stack at future exits are also dead.\n• Dead call-stack store elimination. This is the same optimization\nas above, except applied to the interpreters call stack used forin a hash table. function call inlining.\nThe garbage collector is an exact, non-generational, stop-the world mark-and-sweep collector.\nIn the rest of this section we discuss key areas of the TraceMon key implementation.\n• Dead code elimination. This eliminates any operation that\nstores to a value that is never used.\nAfter a LIR instruction is successfully read (“pulled”) from the backward filter pipeline, nanojits code generator emits native machine instruction(s) for it.\n6.1 Calling Compiled Traces\nCompiled traces are stored in a trace cache, indexed by intepreter PC and type map. Traces are compiled so that they may be called as functions using standard native calling conventions (e.g.,\n5.2 Register Allocation\nWe use a simple greedy register allocator that makes a singleFASTCALL on x86). backward pass over the trace (it is integrated with the code gen\nThe interpreter must hit a loop edge and enter the monitor in order to call a native trace for the first time. The monitor computes the current type map, checks the trace cache for a trace for the current PC and type map, and if it finds one, executes the trace.\nerator). By the time the allocator has reached an instruction like\nv3 = add v1, v2, it has already assigned a register to v3. If v1 and\nv2 have not yet been assigned registers, the allocator assigns a free\nTo execute a trace, the monitor must build a trace activation record containing imported local and global variables, temporary stack space, and space for arguments to native calls. The local and global values are then copied from the interpreter state to the trace activation record. Then, the trace is called like a normal C function\nregister to each. If there are no free registers, a val“ue is se”lected for\nspilling. We use a class heuristic that selects the oldest register\ncarried value (6).\nThe heuristic considers the set R of values v in registers imme diately after the current instruction for spilling. Let vm be the last instruction before the current where each v is referred to. Then thepointer.\nWhen a trace call returns, the monitor restores the interpreter state. First, the monitor checks the reason for the trace exit and applies blacklisting if needed. Then, it pops or synthesizes inter preter JavaScript call stack frames as needed. Finally, it copies the imported variables back from the trace activation record to the in terpreter state.\nRecording is activated by a pointer“ swap tha”t sets the inter preters dispatch table to call a single interrupt routine for ev ery bytecode. The interrupt routine first calls a bytecode-specific recording routine. Then, it turns off recording if necessary (e.g., the trace ended). Finally, itj umps to the standard interpreter byte code implementation. Some bytecodes have effects on the type map that cannot be predicted before executing the bytecode (e.g., call ing String.charCodeAt, which returns an integer or NaN if the index argument is out of range). For these, we arrange for the inter preter to call into the recorder again after executing the bytecode. Since such hooks are relatively rare, we embed them directly into the interpreter, with an additional runtime check to see whether a recorder is currently active.\nAt least in the current implementation, these steps have a non negligible runtime cost, so minimizing the number of interpreter to-trace and trace-to-interpreter transitions is essential for perfor mance. (see also Section 3.3). Our experiments (see Figure 12) show that for programs we can trace well such transitions hap pen infrequently and hence do not contribute significantly to total runtime. In a few programs, where the system is prevented from recording branch traces for hot side exits by aborts, this cost can rise to up to 10% of total execution time.\nWhile separating the interpreter from the recorder reduces indi vidual code complexity, it also requires careful implementation and extensive testing to achieve semantic equivalence.\nIn some cases achieving this equivalence is difficult since Spi derMonkey follows a fat-bytecode design, which was found to be beneficial to pure interpreter performance.\n6.2 Trace Stitching\nTransitions from a trace to a branch trace at a side exit avoid the costs of calling traces from the monitor, in a feature called trace stitching. At a side exit, the exiting trace only needs to write live register-carried values back to its trace activation record. In our im plementation, identical type maps yield identical activation record layouts, so the trace activation record can be reused immediately by the branch trace.\nIn fat-bytecode designs, individual bytecodes can implement complex processing (e.g., the getprop bytecode, which imple ments full JavaScript property value access, including special cases for cached and dense array access).\nFat bytecodes have two advantages: fewer bytecodes means lower dispatch cost, and bigger bytecode implementations give the compiler more opportunities to optimize the interpreter.\nIn programs with branchy trace trees with small traces, trace stitching has a noticeable cost. Although writing to memory and then soon reading back would be expected to have a high L1 cache hit rate, for small traces the increased instruction count has a noticeable cost. Also, if the writes and reads are very close in the dynamic instruction stream, we have found that current x86 processors often incur penalties of 6 cycles or more (e.g., if the instructions use different base registers with equal values, the processor may not be able to detect that the addresses are the samebase interpreter. right away).\nFat bytecodes are a problem for TraceMonkey because they require the recorder to reimplement the same special case logic in the same way. Also, the advantages are reduced because (a) dispatch costs are eliminated entirely in compiled traces, (b) the traces contain only one special case, not the interpreters large chunk of code, and (c) TraceMonkey spends less time running the\nOne way we have mitigated these problems is by implementing certain complex bytecodes in the recorder as sequences of simple bytecodes. Expressing the original semantics this way is not too dif ficult, and recording simple bytecodes is much easier. This enables us to retain the advantages of fat bytecodes while avoiding some of their problems for trace recording. This is particularly effective for fat bytecodes that recurse back into the interpreter, for example to convert an object into a primitive value by invoking a well-known method on the object, since it lets us inline this function call.\nThe alternate solution is to recompile an entire trace tree, thus achieving inter-trace register allocation (10). The disadvantage is that tree recompilation takes time quadratic in the number of traces. We believe that the cost of recompiling a trace tree every time a branch is added would be prohibitive. That problem might be mitigated by recompiling only at certain points, or only for very hot, stable trees.\nIn the future, multicore hardware is expected to be common, making background tree recompilation attractive. In a closely re lated project (13) background recompilation yielded speedups of up to 1.25x on benchmarks with many branch traces. We plan to apply this technique to TraceMonkey as future work.\nIt is important to note that we split fat opcodes into thinner op codes only during recording. When running purely interpretatively (i.e. code that has been blacklisted), the interpreter directly and ef ficiently executes the fat opcodes.\n6.3 Trace Recording\nThej ob of the trace recorder is to emit LIR with identical semantics6.4 Preemption to the currently running interpreter bytecode trace. A good imple\nmentation should have low impact on non-tracing interpreter per\nSpiderMonkey, like many VMs, needs to preempt the user program periodically. The main reasons are to prevent infinitely looping scripts from locking up the host system and to schedule GC. “\nformance and a convenient way for implementers to maintain se\nIn the i”nterpreter, this had been implemented by setting a pre empt now flag that was checked on every backwardj ump. This strategy carried over into TraceMonkey: the VM inserts a guard on the preemption flag at every loop edge. We measured less than a 1% increase in runtime on most benchmarks for this extra guard. In practice, the cost is detectable only for programs with very short\nmantic equivalence.\nIn our implementation, the only direct modification to the inter preter is a call to the trace monitor at loop edges. In our benchmark results (see Figure 12) the total time spent in the monitor (for all activities) is usually less than 5%, so we consider the interpreter impact requirement met. Incrementing the loop hit counter is ex pensive because it requires us to look up the loop in the trace cache,loops. but we have tuned our loops to become hot and trace very quickly (on the second iteration). The hit counter implementation could be improved, which might give us a small increase in overall perfor mance, as well as more flexibility with tuning hotness thresholds. Once a loop is blacklisted we never call into the trace monitor for that loop (see Section 3.3).\nWe tested and rejected a solution that avoided the guards by compiling the loop edge as an unconditionalj ump, and patching thej ump target to an exit routine when preemption is required. This solution can make the normal case slightly faster, but then preemption becomes very slow. The implementation was also very complex, especially trying to restart execution after the preemption.\n6.5 Calling External Functions\nLike most interpreters, SpiderMonkey has a foreign function inter face (FFI) that allows it to call C builtins and host system functions (e.g., web browser control and DOM access). The FFI has a stan dard signature for JS-callable functions, the key argument of which is an array of boxed values. External functions called through the FFI interact with the program state through an interpreter API (e.g., to read a property from an argument). There are also certain inter preter builtins that do not use the FFI, but interact with the program state in the same way, such as the CallIteratorNext function /8A\u003e98FG8E.92/09?@D2#3$4!56# used with iterator objects. TraceMonkey must support this FFI in order to speed up code that interacts with the host system inside hot loops.\n?\u003e9@AJ.D\u003cF@-\u003c\u003e2.@A:0\u003e#3$4,56#\n?\u003e9@A?J\u003e.90@AAJ:.\u003c\u003e\u003c/JC.//F880--2##33$$44%$5566##\n?\u003e9?@\u003eA9J@.A1J\u003c.B?\u003c2?)\u003e\u0027\u003c##33%$44((5566##\n92J25:.-A\u003c#3\u00274%56#\n77\u003c\u003e\u003c;\u003e.;?.::2\u003c/9\u003eI9\u003c\u003cFF..A?08797?##33(*44,$5566##\n7\u003c\u003e;./89-@/#3\u00274,56#\n--\u003c\u003c\u003e\u003e22.B.8B89977\u003c\u003c\u003e.\u003e5.\u003e:8\u003cH921##33$$44!$5566##\n/9=:\u003e8.?;\u003c$#3(4,56#\n/9/=9=::\u003e8\u003e8.7.\u003c-2(?##33%$44\u0026)5566##\n1@\u003e18@\u003e:8?.:1?@.\u003eAE?@@?22D.2\u003c.A1-@\u003e#?3#%3(%44*%5566##\n1@\u003e8:?.1@\u003e?.@A.1=\u003e2#3+4*56#\nCalling external functions from TraceMonkey is potentially dif ficult because traces do not update the interpreter state until exit ing. In particular, external functions may need the call stack or the global variables, but they may be out of date.\n1@\u003e8:?.\u00261@\u003e\u003c.1/@/\u003e2?.?@?A..A1?=@2\u003e2D#23#%3\u0026(44!(5566##\n\u003c//2??.A18-=#3\u00274%56#\n\u003c .BAAC ;#%4%5#\n\u003c//2/?/?2.1??@A\u003c\u003c9=.\u003e902/2?#33!4,566#\n\u0026-\u0026.-9.\u003c7=\u003e89\u003c9:/;2##33%$44,%5566##\nFor the out-of-date call stack problem, we refactored some of the interpreter API implementation functions to re-materialize the interpreter call stack on demand.\n\u0026-./012#3%4%56#\n!\"# $!\"# %!\"# \u0026!\"# \u0027!\"# (!\"# )!\"# *!\"# +!\"# ,!\"# $!!\"#\nWe developed a C++ static analysis and annotated some inter preter functions in order to verify that the call stack is refreshed at any point it needs to be used. In order to access the call stack, a function must be annotated as either FOR CESSTACK or RE\nKA\u003e29:92\u003e# L\u003cID2#\nFigure 11. Fraction of dynamic bytecodes executed by inter preter and on native traces. The speedup vs. interpreter is shown\nQUIRESSTACK. These annotations are also required in order to call\nin parentheses next to each test. The fraction of bytecodes exe\nREQUIRESSTACK functions, which are presumed to access the call stack transitively. FOR CESSTACK is a trusted annotation, applied to only 5 functions, that means the function refreshes the call stack. REQUIRESSTACK is an untrusted annotation that means the func tion may only be called if the call stack has already been refreshed.\ncuted while recording is too small to see in this figure, except for crypto-md5, where fully 3% of bytecodes are executed while recording. In most of the tests, almost all the bytecodes are exe cuted by compiled traces. Three of the benchmarks are not traced\nSimilarly, we detect when host functions attempt to directly read or write global variables, and force the currently running trace to side exit. This is necessary since we cache and unbox global variables into the activation record during trace execution.\nat all and run in the interpreter.\nloops and heavily branching code, and a specialized fuzz tester in deed revealed several regressions which we subsequently corrected.\nSince both call-stack access and global variable access are rarely performed by host functions, performance is not significantly affected by these safety mechanisms.\n7. Evaluation\nAnother problem is that external functions can reenter the inter preter by calling scripts, which in turn again might want to access the call stack or global variables. To address this problem, we made the VM set a flag whenever the interpreter is reentered while a com piled trace is running.\nWe evaluated our JavaScript tracing implementation using Sun Spider, the industry standard JavaScript benchmark suite. SunSpi der consists of 26 short-running (less than 250ms, average 26ms) JavaScript programs. This is in stark contrast to benchmark suites such as SpecJVM98 (3) used to evaluate desktop and server Java VMs. Many programs in those benchmarks use large data sets and execute for minutes. The SunSpider programs carry out a variety of tasks, primarily 3d rendering, bit-bashing, cryptographic encoding, math kernels, and string processing.\nEvery call to an external function then checks this flag and exits the trace immediately after returning from the external function call if it is set. There are many external functions that seldom or never reenter, and they can be called without problem, and will cause trace exit only if necessary.\nAll experiments were performed on a MacBook Pro with 2.2 GHz Core 2 processor and 2 GB RAM running MacOS 10.5.\nThe FFIs boxed value array requirement has a performance cost, so we defined a new FFI that allows C functions to be an notated with their argument types so that the tracer can call them directly, without unnecessary argument conversions.\nBenchmark results. The main question is whether programs run faster with tracing. For this, we ran the standard SunSpider test driver, which starts a JavaScript interpreter, loads and runs each program once for warmup, then loads and runs each program 10 times and reports the average time taken by each. We ran 4 differ ent configurations for comparison: (a) SpiderMonkey, the baseline interpreter, (b) TraceMonkey, (d) SquirrelFish Extreme (SFX), the call-threaded JavaScript interpreter used in Apples WebKit, and\nCurrently, we do not support calling native property get and set override functions or DOM functions directly from trace. Support is planned future work.\n6.6 Correctness\n(e) V8, the method-compiling JavaScript VM from Google.\nDuring development, we had access to existing JavaScript test suites, but most of them were not designed with tracing VMs in mind and contained few loops.\nFigure 10 shows the relative speedups achieved by tracing, SFX, and V8 against the baseline (SpiderMonkey). Tracing achieves the best speedups in integer-heavy benchmarks, up to the 25x speedup\nOne tool that helped us greatly was Mozillas JavaScript fuzz\ntester, JSFUNFUZZ, which generates random JavaScript programson bitops-bitwise-and. by nesting random language elements. We modified JSFUNFUZZ\nTraceMonkey is the fastest VM on 9 of the 26 benchmarks (3d-morph, bitops-3bit-bits-in-byte, bitops-bitwise and, crypto-sha1, math-cordic, math-partial-sums, math spectral-norm, string-base64, string-validate-input).\nto generate loops, and also to test more heavily certain constructs\nwe suspected would reveal flaws in our implementation. For exam\nple, we suspected bugs in TraceMonkeys handling of type-unstable\nFigure 10. Speedup vs. a baseline JavaScript interpreter (SpiderMonkey) for our trace-based JIT compiler, Apples SquirrelFish Extreme\ninline threading interpreter and Googles V8 JS compiler. Our system generates particularly efficient code for programs that benefit most from\ntype specialization, which includes SunSpider Benchmark programs that perform bit manipulation. We type-specialize the code in question\nto use integer arithmetic, which substantially improves performance. For one of the benchmark programs we execute 25 times faster than\nthe SpiderMonkey interpreter, and almost 5 times faster than V8 and SFX. For a large number of benchmarks all three VMs produce similar\nresults. We perform worst on benchmark programs that we do not trace and instead fall back onto the interpreter. This includes the recursive\nbenchmarks access-binary-trees and control-flow-recursive, for which we currently dont generate any native code.\nIn particular, the bitops benchmarks are short programs that per form many bitwise operations, so TraceMonkey can cover the en tire program with 1 or 2 traces that operate on integers. TraceMon key runs all the other programs in this set almost entirely as native code.jit.\nregexp-dna is dominated by regular expression matching, which is implemented in all 3 VMs by a special regular expression compiler. Thus, performance on this benchmark has little relation to the trace compilation approach discussed in this paper.\n• Two programs trace well, but have a long compilation time.\naccess-nbody forms a large number of traces (81). crypto-md5 forms one very long trace. We expect to improve performance on this programs by improving the compilation speed of nano\n• Some programs trace very well, and speed up compared to\nthe interpreter, but are not as fast as SFX and/or V8, namely bitops-bits-in-byte, bitops-nsieve-bits, access fannkuch, access-nsieve, and crypto-aes. The reason is not clear, but all of these programs have nested loops with small bodies, so we suspect that the implementation has a rela tively high cost for calling nested traces. string-fasta traces well, but its run time is dominated by string processing builtins, which are unaffected by tracing and seem to be less efficient in SpiderMonkey than in the two other VMs.\nTraceMonkeys smaller speedups on the other benchmarks can be attributed to a few specific causes:\n• The implementation does not currently trace recursion, so\nTraceMonkey achieves a small speedup or no speedup on benchmarks that use recursion extensively: 3d-cube, 3d raytrace, access-binary-trees, string-tagcloud, and controlflow-recursive.\nDetailed performance metrics. In Figure 11 we show the frac tion of instructions interpreted and the fraction of instructions exe cuted as native code. This figure shows that for many programs, we are able to execute almost all the code natively.\n• The implementation does not currently trace eval and some\nother functions implemented in C. Because date-format tofte and date-format-xparb use such functions in their main loops, we do not trace them.\nFigure 12 breaks down the total execution time into four activ ities: interpreting bytecodes while not recording, recording traces (including time taken to interpret the recorded trace), compiling traces to native code, and executing native code traces.\n• The implementation does not currently trace through regular\nexpression replace operations. The replace function can be passed a function object used to compute the replacement text. Our implementation currently does not trace functions called as replace functions. The run time of string-unpack-code is dominated by such a replace call.\nThese detailed metrics allow us to estimate parameters for a simple model of tracing performance. These estimates should be considered very rough, as the values observed on the individual benchmarks have large standard deviations (on the order of the\nFigure 13. Detailed trace recording statistics for the SunSpider benchmark set.\nmean). We exclude regexp-dna from the following calculations,\nbecause most of its time is spent in the regular expression matcher,of 26 benchmarks. which has much different performance characteristics from the\nfastest available JavaScript inline threaded interpreter (SFX) on 9\nother programs. (Note that this only makes a difference of about\n10% in the results.) Dividing the total execution time in processor\nclock cycles by the number of bytecodes executed in the base8. Related Work interpreter shows that on average, a bytecode executes in about\n35 cycles. Native traces take about 9 cycles per bytecode, a 3.9x\nTrace optimization for dynamic languages. The closest area of related work is on applying trace optimization to type-specialize dynamic languages. Existing work shares the idea of generating type-specialized code speculatively with guards along interpreter\nspeedup over the interpreter.\nUsing similar computations, we find that trace recording takes about 3800 cycles per bytecode, and compilation 3150 cycles pertraces. bytecode. Hence, during recording and compiling the VM runs at 1/200 the speed of the interpreter. Because it costs 6950 cycles to compile a bytecode, and we save 26 cycles each time that code is run natively, we break even after running a trace 270 times.\nTo our knowledge, Rigos Psyco (16) is the only published type-specializing trace compiler for a dynamic language (Python). Psyco does not attempt to identify hot loops or inline function calls. Instead, Psyco transforms loops to mutual recursion before running and traces all operations.\nThe other VMs we compared with achieve an overall speedup of 3.0x relative to our baseline interpreter. Our estimated native code speedup of 3.9x is significantly better. This suggests that our compilation techniques can generate more efficient native code than any other current JavaScript VM.\nPalls LuaJIT is a Lua VM in development that uses trace com pilation ideas. (1). There are no publications on LuaJIT but the cre ator has told us that LuaJIT has a similar design to our system, but will use a less aggressive type speculation (e.g., using a floating point representation for all number values) and does not generate nested traces for nested loops.\nThese estimates also indicate that our startup performance could be substantially better if we improved the speed of trace recording and compilation. The estimated 200x slowdown for recording and compilation is very rough, and may be influenced by startup factors in the interpreter (e.g., caches that have not warmed up yet during recording). One observation supporting this conjecture is that in the tracer, interpreted bytecodes take about 180 cycles to run. Still, recording and compilation are clearly both expensive, and a better implementation, possibly including redesign of the LIR abstract syntax or encoding, would improve startup performance.\nGeneral trace optimization. General trace optimization has a longer history that has treated mostly native code and typed languages like Java. Thus, these systems have focused less on type specialization and more on other optimizations.\nDynamo (7) by Bala et al, introduced native code tracing as a replacement for profile-guided optimization (PGO). A major goal was to perform PGO online so that the profile was specific to the current execution. Dynamo used loop headers as candidate hot traces, but did not try to create loop traces specifically.\nOur performance results confirm that type specialization using trace trees substantially improves performance. We are able to outperform the fastest available JavaScript compiler (V8) and the\nTrace trees were originally proposed by Gal et al. (11) in the context of Java, a statically typed language. Their trace trees ac tually inlined parts of outer loops within the inner loops (because\n=\u003c6\u003e?J+B:F\u003e*:\u003c/+\u003e?7-\u003c#0(1923#\n=\u003c6\u003e?J+-?7+:,A+,5*/#0(1$23#\n=\u003c6\u003e?=J\u003c6\u003c\u003e?:JJ+,@F:5=-\u003c*:##00((11C(2233##\n=\u003c6\u003e?J+.:=+/\u0026%#0$1C23#\n4:\u003c8+=7/6,/\u003cJ6/:2F+7?5*6?4:##00%D11$(2233##\n4:\u003c8+7:6I:+F+=-4=#0C1923#\n*:\u003c/+@5464:\u003c:8\u003c+,2576:*6\u003e.,##00%(119!2233##\n*:\u003c/+@,566;47\u003c:5\u003c++\u003c=58H:/(##00C(119(2233##\n,6;7\u003c5+4*C#0$1)23#\n,5?\u003c65FG5E,+66;/7,\u003c-56+=:\u003eB//=##00((11!\u00262233##\n.\u003e\u003c57=+?=\u003e/B/+.\u003e\u003c=#0$1D23#\n.\u003e.\u003c5\u003e\u003c75=7+=.+\u003e.\u003cE\u003e\u003c=\u003e=+\u003e/?++:.?;*\u003c#/0#$0\u0027C11D$2233##\nerate native code with nearly the same structure but better perfor mance.\nCall threading, also known as context threading (8), compiles methods by generating a native call instruction to an interpreter method for each interpreter bytecode. A call-return pair has been shown to be a potentially much more efficient dispatch mechanism than the indirectj umps used in standard bytecode interpreters.\nInline threading (15) copies chunks of interpreter native code which implement the required bytecodes into a native code cache, thus acting as a simple per-method JIT compiler that eliminates the dispatch overhead.\nNeither call threading nor inline threading perform type special ization. \n.\u003e\u003c57=+).\u003e\u003c+.\u003e\u003c=+\u003e?++.;\u003c/#0$C1C23#\n::,,,,//====+??=.\u003e5/*B/;##00%)11!$2233##\n:,,:/,=,=/+.==\u003e?+@::6?;?+\u003cA6-/,/8=##00!$119$2233##\nApples SquirrelFish Extreme (5) is a JavaScript implementa tion based on call threading with selective inline threading. Com bined with efficient interpreter engineering, these threading tech niques have given SFX excellent performance on the standard Sun Spider benchmarks.\n)*)+*6+:4;\u003c56:67,8/##00$(119$2233##\n)*+,-./#0$1$23#\n!\"# $!\"# %!\"# \u0026!\"# \u0027!\"# (!!\"#Googles V8 is a JavaScript implementation primarily based\non inline threading, with call threading only for very complex operations.\nK?\u003c/676/\u003c# L5?\u003e\u003c56# M/,56*# N547\u003eF/# N:FF#O6:,/# M-?#O6:,/#\n9. Conclusions\nFigure 12. Fraction of time spent on major VM activities. The speedup vs. interpreter is shown in parentheses next to each test.\nThis paper described how to run dynamic languages efficiently by recording hot traces and generating type-specialized native code. Our technique focuses on aggressively inlined loops, and for each loop, it generates a tree of native code traces representing the paths and value types through the loop observed at run time. We explained how to identify loop nesting relationships and generate nested traces in order to avoid excessive code duplication due to the many paths through a loop nest. We described our type specialization algorithm. We also described our trace compiler, which translates a trace from an intermediate representation to optimized native code in two linear passes.\nMost programs where the VM spends the majority of its time run ning native code have a good speedup. Recording and compilation costs can be substantial; speeding up those parts of the implemen tation would improve SunSpider performance.\ninner loops become hot first), leading to much greater tail duplica tion.\nOur experimental results show that in practice loops typically are entered with only a few different combinations of value types of variables. Thus, a small number of traces per loop is sufficient to run a program efficiently. Our experiments also show that on programs amenable to tracing, we achieve speedups of 2x to 20x.\nYETI, from Zaleski et al. (19) applied Dynamo-style tracing to Java in order to achieve inlining, indirectj ump elimination, and other optimizations. Their primary focus was on designing an interpreter that could easily be gradually re-engineered as a tracing VM.\nSuganuma et al. (18) described region-based compilation (RBC),\na relative of tracing. A region is an subprogram worth optimizing10. Future Work that can include subsets of any number of methods. Thus, the com\npiler has more flexibility and can potentially generate better code,\nWork is underway in a number of areas to further improve the performance of our trace-based JavaScript compiler. We currently do not trace across recursive function calls, but plan to add the support for this capability in the near term. We are also exploring adoption of the existing work on tree recompilation in the context of the presented dynamic compiler in order to minimize JIT pause times and obtain the best of both worlds, fast tree stitching as well as the improved code quality due to tree recompilation.\nbut the profiling and compilation systems are correspondingly more complex.\nType specialization for dynamic languages. Dynamic lan guage implementors have long recognized the importance of type specialization for performance. Most previous work has focused on methods instead of traces.\nChambers et. al (9) pioneered the idea of compiling multiple\nWe also plan on adding support for tracing across regular ex pression substitutions using lambda functions, function applica tions and expression evaluation using eval. All these language constructs are currently executed via interpretation, which limits our performance for applications that use those features.\nversions of a procedure specialized for the input types in the lan guage Self. In one implementation, they generated a specialized method online each time a method was called with new input types.\nIn another, they used an offline whole-program static analysis to infer input types and constant receiver types at call sites. Interest ingly, the two techniques produced nearly the same performance.Acknowledgments\nSalib (17) designed a type inference algorithm for Python based on the Cartesian Product Algorithm and used the results to special ize on types and translate the program to C++.\nParts of this effort have been sponsored by the National Science Foundation under grants CNS-0615443 and CNS-0627747, as well as by the California MICRO Program and industrial sponsor Sun Microsystems under Project No. 07-127.\nMcCloskey (14) has work in progress based on a language independent type inference that is used to generate efficient C implementations of JavaScript and Python programs.\nThe U.S. Government is authorized to reproduce and distribute reprints for Governmental purposes notwithstanding any copyright annotation thereon. Any opinions, findings, and conclusions or rec ommendations expressed here are those of the author and should\nNative code generation by interpreters. The traditional inter preter design is a virtual machine that directly executes ASTs or machine-code-like bytecodes. Researchers have shown how to gen\nnot be interpreted as necessarily representing the official views, policies or endorsements, either expressed or implied, of the Na tional Science foundation (NSF), any other agency of the U.S. Gov ernment, or any of the companies mentioned above.\n[10] A. Gal. EfficientB ytecode Verification and Compilation in a Virtual\nMachineD issertation. PhD thesis, University Of California, Irvine, 2006.\nReferences\n[11] A. Gal, C. W. Probst, and M. Franz. HotpathVM: An effective JIT\ncompiler for resource-constrained devices. In Proceedings of the International Conference on Virtual Execution Environments, pages 144153. ACM Press, 2006.\n[1] LuaJIT roadmap 2008 - http://lua-users.org/lists/lua-l/2008-\n[12] C. Garrett, J. Dean, D. Grove, and C. Chambers. Measurement and\nApplication of Dynamic Receiver Class Distributions. 1994.\n02/msg00051.html.\nent - erbird email cli\n[2] Mozilla — Firefox web browser and Thund\nhttp://www.mozilla.com.\n[13] J. Ha, M. R. Haghighat, S. Cong, and K. S. McKinley. A concurrent\ntrace-basedj ust-in-time compiler forj avascript. Dept.of Computer Sciences, The University of Texas at Austin, TR-09-06, 2009. [14] B. McCloskey. Personal communication.\n[3] SPECJVM98 - http://www.spec.org/jvm98/. [4] SpiderMonkey (JavaScript-C)\n Engine\nhttp://www.mozilla.org/js/spidermonkey/.\n[5] Surfin Safari - Blog Archive - Announcing SquirrelFish Extre\nhttp://webkit.org/blog/214/introducing-squirrelfish-extreme/.\n[15] I. Piumarta and F. Riccardi. Optimizing direct threaded code by selec\ntive inlining. In Proceedings of theA CM SIGPLAN 1998 conference on Programming language design and implementation, pages 291\n[6] A. Aho, R. Sethi, J. Ullman, and M. Lam. Compilers: Principles,\ntechniques, and tools, 2006.\n300. ACM New York, NY, USA, 1998.\n[16] A. Rigo. Representation-Based Just-In-time Specialization and the\nPsyco Prototype for Python. In PEPM, 2004.\n[7] V. Bala, E. Duesterwald, and S. Banerjia. Dynamo: A transparent\ndynamic optimization system. In Proceedings of theA CM SIGPLAN Conference on ProgrammingL anguageD esign andI mplementation,\npages 112. ACM Press, 2000.\n[17] M. Salib. Starkiller: A Static Type Inferencer and Compiler for\nPython. In Masters Thesis, 2004.\n[8] M. Berndl, B. Vitale, M. Zaleski, and A. Brown. Context Threading:\na Flexible and Efficient Dispatch Technique for Virtual Machine In terpreters. In Code Generation and Optimization, 2005. CGO 2005.\n[18] T. Suganuma, T. Yasue, and T. Nakatani. A Region-Based Compila\ntion Technique for Dynamic Compilers. ACM Transactions on Pro grammingL anguages and Systems (TOPLAS), 28(1):134174, 2006. [19] M. Zaleski, A. D. Brown, and K. Stoodley. YETI: A graduallY\nInternational Symposium on, pages 1526, 2005.\n[9] C. Chambers and D. Ungar. Customization: Optimizing Compiler\nTechnology for SELF, a Dynamically-Typed O bject-Oriented Pro gramming Language. In Proceedings of theA CM SIGPLAN 1989 Conference on ProgrammingL anguageD esign andI mplementation,\npages 146160. ACM New York, NY, USA, 1989.\nExtensible Trace Interpreter. In Proceedings of theI nternational Conference on Virtual Execution Environments, pages 8393. ACM Press, 2007.\nHence, recording and compiling a trace speculates that the path and\n\n1 for (var i = 2; i \u003c 100; ++i) {\ntyping will be exactly as they were during recording for subsequent\n2\nif (!primes[i])\niterations of the loop.\n3\ncontinue;\nEvery compiled trace contains all the guards (checks) required\n4\nfor (var k = i + i; i \u003c 100; k += i)\nto validate the speculation. If one of the guards fails (if control\n5\nprimes[k] = false;\nxx1\nnumber\n31-bit integer representation\n000\nobject\npointer to JSObject handle\n010\nnumber\npointer to double handle\n100\nstring\npointer to JSString handle\n110\nboolean\nenumeration for null, undefined, true, false\n\nLoops\nTrees\nTraces\nAborts\nFlushes\nTrees/Loop\nTraces/Tree\nTraces/Loop\nSpeedup\n3d-cube\n25\n27\n29\n3\n0\n1.1\n1.1\n1.2\n2.20x\n3d-morph\n5\n8\n8\n2\n0\n1.6\n1.0\n1.6\n2.86x\n3d-raytrace\n10\n25\n100\n10\n1\n2.5\n4.0\n10.0\n1.18x\naccess-binary-trees\n0\n0\n0\n5\n0\n-\n-\n-\n0.93x\naccess-fannkuch\n10\n34\n57\n24\n0\n3.4\n1.7\n5.7\n2.20x\naccess-nbody\n8\n16\n18\n5\n0\n2.0\n1.1\n2.3\n4.19x\naccess-nsieve\n3\n6\n8\n3\n0\n2.0\n1.3\n2.7\n3.05x\nbitops-3bit-bits-in-byte\n2\n2\n2\n0\n0\n1.0\n1.0\n1.0\n25.47x\nbitops-bits-in-byte\n3\n3\n4\n1\n0\n1.0\n1.3\n1.3\n8.67x\nbitops-bitwise-and\n1\n1\n1\n0\n0\n1.0\n1.0\n1.0\n25.20x\nbitops-nsieve-bits\n3\n3\n5\n0\n0\n1.0\n1.7\n1.7\n2.75x\ncontrolflow-recursive\n0\n0\n0\n1\n0\n-\n-\n-\n0.98x\ncrypto-aes\n50\n72\n78\n19\n0\n1.4\n1.1\n1.6\n1.64x\ncrypto-md5\n4\n4\n5\n0\n0\n1.0\n1.3\n1.3\n2.30x\ncrypto-sha1\n5\n5\n10\n0\n0\n1.0\n2.0\n2.0\n5.95x\ndate-format-tofte\n3\n3\n4\n7\n0\n1.0\n1.3\n1.3\n1.07x\ndate-format-xparb\n3\n3\n11\n3\n0\n1.0\n3.7\n3.7\n0.98x\nmath-cordic\n2\n4\n5\n1\n0\n2.0\n1.3\n2.5\n4.92x\nmath-partial-sums\n2\n4\n4\n1\n0\n2.0\n1.0\n2.0\n5.90x\nmath-spectral-norm\n15\n20\n20\n0\n0\n1.3\n1.0\n1.3\n7.12x\nregexp-dna\n2\n2\n2\n0\n0\n1.0\n1.0\n1.0\n4.21x\nstring-base64\n3\n5\n7\n0\n0\n1.7\n1.4\n2.3\n2.53x\nstring-fasta\n5\n11\n15\n6\n0\n2.2\n1.4\n3.0\n1.49x\nstring-tagcloud\n3\n6\n6\n5\n0\n2.0\n1.0\n2.0\n1.09x\nstring-unpack-code\n4\n4\n37\n0\n0\n1.0\n9.3\n9.3\n1.20x\nstring-validate-input\n6\n10\n13\n1\n0\n1.7\n1.3\n2.2\n1.86x",
"paragraphs": 489,
"mean_paragraph_words": 26.18,
"tables": [
[
[
"Hence, recording and compiling a trace speculates that the path and",
"",
"1 for (var i = 2; i \u003c 100; ++i) {"
],
[
"typing will be exactly as they were during recording for subsequent",
"2",
"if (!primes[i])"
],
[
"iterations of the loop.",
"3",
"continue;"
],
[
"Every compiled trace contains all the guards (checks) required",
"4",
"for (var k = i + i; i \u003c 100; k += i)"
],
[
"to validate the speculation. If one of the guards fails (if control",
"5",
"primes[k] = false;"
]
],
[
[
"xx1",
"number",
"31-bit integer representation"
],
[
"000",
"object",
"pointer to JSObject handle"
],
[
"010",
"number",
"pointer to double handle"
],
[
"100",
"string",
"pointer to JSString handle"
],
[
"110",
"boolean",
"enumeration for null, undefined, true, false"
]
],
[
[
"",
"Loops",
"Trees",
"Traces",
"Aborts",
"Flushes",
"Trees/Loop",
"Traces/Tree",
"Traces/Loop",
"Speedup"
],
[
"3d-cube",
"25",
"27",
"29",
"3",
"0",
"1.1",
"1.1",
"1.2",
"2.20x"
],
[
"3d-morph",
"5",
"8",
"8",
"2",
"0",
"1.6",
"1.0",
"1.6",
"2.86x"
],
[
"3d-raytrace",
"10",
"25",
"100",
"10",
"1",
"2.5",
"4.0",
"10.0",
"1.18x"
],
[
"access-binary-trees",
"0",
"0",
"0",
"5",
"0",
"-",
"-",
"-",
"0.93x"
],
[
"access-fannkuch",
"10",
"34",
"57",
"24",
"0",
"3.4",
"1.7",
"5.7",
"2.20x"
],
[
"access-nbody",
"8",
"16",
"18",
"5",
"0",
"2.0",
"1.1",
"2.3",
"4.19x"
],
[
"access-nsieve",
"3",
"6",
"8",
"3",
"0",
"2.0",
"1.3",
"2.7",
"3.05x"
],
[
"bitops-3bit-bits-in-byte",
"2",
"2",
"2",
"0",
"0",
"1.0",
"1.0",
"1.0",
"25.47x"
],
[
"bitops-bits-in-byte",
"3",
"3",
"4",
"1",
"0",
"1.0",
"1.3",
"1.3",
"8.67x"
],
[
"bitops-bitwise-and",
"1",
"1",
"1",
"0",
"0",
"1.0",
"1.0",
"1.0",
"25.20x"
],
[
"bitops-nsieve-bits",
"3",
"3",
"5",
"0",
"0",
"1.0",
"1.7",
"1.7",
"2.75x"
],
[
"controlflow-recursive",
"0",
"0",
"0",
"1",
"0",
"-",
"-",
"-",
"0.98x"
],
[
"crypto-aes",
"50",
"72",
"78",
"19",
"0",
"1.4",
"1.1",
"1.6",
"1.64x"
],
[
"crypto-md5",
"4",
"4",
"5",
"0",
"0",
"1.0",
"1.3",
"1.3",
"2.30x"
],
[
"crypto-sha1",
"5",
"5",
"10",
"0",
"0",
"1.0",
"2.0",
"2.0",
"5.95x"
],
[
"date-format-tofte",
"3",
"3",
"4",
"7",
"0",
"1.0",
"1.3",
"1.3",
"1.07x"
],
[
"date-format-xparb",
"3",
"3",
"11",
"3",
"0",
"1.0",
"3.7",
"3.7",
"0.98x"
],
[
"math-cordic",
"2",
"4",
"5",
"1",
"0",
"2.0",
"1.3",
"2.5",
"4.92x"
],
[
"math-partial-sums",
"2",
"4",
"4",
"1",
"0",
"2.0",
"1.0",
"2.0",
"5.90x"
],
[
"math-spectral-norm",
"15",
"20",
"20",
"0",
"0",
"1.3",
"1.0",
"1.3",
"7.12x"
],
[
"regexp-dna",
"2",
"2",
"2",
"0",
"0",
"1.0",
"1.0",
"1.0",
"4.21x"
],
[
"string-base64",
"3",
"5",
"7",
"0",
"0",
"1.7",
"1.4",
"2.3",
"2.53x"
],
[
"string-fasta",
"5",
"11",
"15",
"6",
"0",
"2.2",
"1.4",
"3.0",
"1.49x"
],
[
"string-tagcloud",
"3",
"6",
"6",
"5",
"0",
"2.0",
"1.0",
"2.0",
"1.09x"
],
[
"string-unpack-code",
"4",
"4",
"37",
"0",
"0",
"1.0",
"9.3",
"9.3",
"1.20x"
],
[
"string-validate-input",
"6",
"10",
"13",
"1",
"0",
"1.7",
"1.3",
"2.2",
"1.86x"
]
]
],
"table_shapes": [
[
5,
3
],
[
5,
3
],
[
27,
10
]
],
"images": 0,
"page_breaks": 13
},
"content": {
"expected_tokens": 14700,
"actual_tokens": 14546,
"matched_tokens": 13825,
"recall": 0.940476,
"precision": 0.950433,
"order": 0.16575,
"f1": 0.945428,
"duplicate_tokens": 124,
"missing_count": 875,
"extra_count": 721,
"missing": [
"trace",
"-",
"-",
"-",
"-",
"-",
"-",
"-",
"based",
"just",
"just",
"just",
"just",
"just",
"just",
"just",
"just",
"just",
"just",
"just",
"time",
"type",
"specialization",
"for",
"for",
"for",
"for",
"for",
"for",
"languages",
"languages",
"r"
],
"extra": [
"in",
"in",
"com",
"com",
"com",
"com",
"com",
"com",
"pile",
"code",
"al",
"al",
"ternative",
"run",
"ac",
"ac",
"tual",
"inter",
"inter",
"inter",
"inter",
"inter",
"inter",
"inter",
"inter",
"inter",
"inter",
"inter",
"discov",
"ered",
"loops",
"loops"
]
},
"source_derived_table_match": null,
"visual": {
"output_pages": 27,
"compared_pages": 8,
"pixel_similarity": 0.8937
},
"warnings": [
"Layout fidelity is lossy; headers/footers/fonts are not fully preserved.",
"auto: text_based; ocr=never",
"layout_ml=onnx",
"Page 1: layout_ml=onnx regions=19",
"Page 2: layout_ml=onnx regions=20",
"Page 2: heuristic table under ML regions rows=5 conf=0.87.",
"Page 3: layout_ml=onnx regions=8",
"Page 4: layout_ml=onnx regions=21",
"Page 5: layout_ml=onnx regions=18",
"Page 6: layout_ml=onnx regions=22",
"Page 6: heuristic table under ML regions rows=4 conf=0.78.",
"Page 7: layout_ml=onnx regions=32",
"Page 8: layout_ml=onnx regions=38",
"Page 8: ML table region grid_failed conf=0.40; deferring to heuristic.",
"Page 9: layout_ml=onnx regions=22",
"Page 10: layout_ml=onnx regions=22",
"Page 11: layout_ml=onnx regions=13",
"Page 12: layout_ml=onnx regions=15",
"Page 12: ML table-\u003egrid rows=27 cols=10 conf=0.97.",
"Page 13: layout_ml=onnx regions=25",
"Page 14: layout_ml=onnx regions=23",
"ocr_policy=never — skipped OCR (digital text layer kept).",
"Low text order similarity vs source PDF - columns/paragraphs may be reordered (order=0.17).",
"No source-derived grid to compare against; table score is structural (cells checked against the source text)."
],
"warning_count": 24,
"error": null,
"models": {
"onnx_providers": [
"AzureExecutionProvider",
"CPUExecutionProvider"
],
"ocr_available": true,
"ocr_ar_status": "on",
"layout_model": {
"path": "gateway\\models\\layout\\v1\\layout.onnx",
"bytes": 130502330,
"sha256": "250dbad1dfb9e4983fab75e1bf5085cd56ec3f41d5c7d0f8623ec74856e7aa67"
}
}
},
{
"document": "usgs_factsheet.pdf",
"target": "docx",
"input_bytes": 1650386,
"source_pages": 4,
"source_chars": 14019,
"source_words": 1889,
"source_images": 18,
"elapsed_s": 14.7904,
"cpu_s": 31.3906,
"cpu_to_wall": 2.122,
"rss_start_mb": 1085.95,
"rss_peak_mb": 1363.86,
"output_bytes": 162128,
"fidelity": "lossy",
"quality": 0.8069623202386644,
"media": "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
"output": {
"text": "Prepared in cooperation with the Federal Interagency Sedimentation Project\nEstimating Sus ended Sediment pin Rivers Using \nAcoustic Do ler Meterspp\nKey Points\nThe U.S . Environmenta lProtection Agency (2009) estimates\nthat excessive sediment i sthe leading cause o fwater-quality\nimpairment in water bodie s in the United States . The cost of damage s attributable to sediment i shigh , estimated at more than\n$20 billion annually (Osterkamp and others , 2004).\nSediment monitoring i s essentia lto informed solution sto\nsediment-related issues . However , sediment monitoring by the\nconsiderably over\nS) ha s decreased \nl Survey (USG\nthe past quarter century.\nNew technique sthat make use o facoustic backscatter have shown”\ngreat potentia l for accurately and cost-effectively estim\nating\nsuspended-sediment concentrations.\nMonitoring sediment is important for the \nmanagement of water resources. Sediment “ monitoring data can be used to determine effectiveness of sediment reduction actions in \nthe watershed and guide adaptive sediment \nmanagement, states Richard Turner of the \nU.S. Army Corps of Engineers, Walla Walla District. “Monitoring data helps foster fact based working relationships with regulators \nand stakeholders, and contributes to the U.S. \nArmy Corps of Engineers public safety efforts and flood risk reduction.\nWhy Is Sediment Important to Measure?\n C\nD\ntransported a s suspended load\n(move swith the flow o fthe river) or a sbedload (rolls\nalong the riverbed) or can be deposited on the riverbed or bank . The concept s described\nin thi s Fact Sheet focu s on method s for\nestimating suspended sediment because it is B\ntypically the largest part o ftota l sediment transported in a river (Meade and others,\n1990) . Sediment i snaturally occurring\nand essentia lto supporting the ecological\nfunction o fa water body . High sediment\nconcentration s in river s and streams,\nhowever , can be detrimenta l (fig . 1).\nHow Is Suspended Sediment \nA\nMeasured?\nFor many years , USG S scientist shave\ncollected sediment sample s from multiple\nvertica l section s in river susing point\nor depth-integrating samplers . Sediment sample srepresent the sediment concentration\nin a particular river at a given point in time . To\ncontinuously estimate sediment concentrations\nduring period swhen sample s are not collected,\nscientist s develop relation sbetween sediment concentration s and other parameters , most comm streamflow measured at a nearby streamgage.\nonly\nE\nFigure 1 . High sediment concentrations can reduce biological productivity of \naquatc syste s (A), impair water quality (B), (C), (E), decrease flood protection \ni m\ncapacity of levees and dams (D), decrease reservoir storage capacity (D) and affect waterway navigation (E). \n[Image]\n[Image]\n[Image]\n[Image]\n[Image]\nino1,600 dnreal-time data can provide an early warning for operator s of\ntaroc\ntn r 2,5001,400 esmunicipa lwater supply and hydropower facilitie s concerned\nec iltere with avoiding damage to infrastructure from sediment.\nno re 1,200 tp \ntc p 2,000ee\nne sm1,000 fic \nim arbHow Does a Sediment Acoustic Surrogate \nde illg 1,500800 cu d m s- iin, Streamgage Work?\nden in 1,000600 lowfScientist s deploy an acoustic Doppler meter typically at\ne\nps400 maa fixed location in a river . The meter transmit spulse s o fsound\nSu 500terat a known frequency , along two or more beam s angled to\n200 S\n0\n0\n17:00 23:00 5:00 11:0017:0023:00using the Doppler principle but also output a return pulse\n10/29/09 10/29/09 10/30/09 10/30/09 10/30/09 10/30/09 \nEXPLANATION\nSamples\nSediment concentration estimated \nusing sediment acoustic relation\nStreamflow\nFigure 2 . Suspended-sediment concentrations and streamflow \nduring a storm event on Kickapoo Creek near Bloomington, Illinois \n(USGS streamgage 05579630). Sampled sediment concentrations peak prior to the peak in streamflow and are not equal at identical streamflows during the event. Sediment concentrations estimated using data from an acoustic Doppler meter at the streamgage closely match sampled concentrations.\nHowever , thi s approach often doe snot accurately estimate sediment concentrations . Sediment concentration smay differ\nfor the same streamflows ,particularly on the rising and falling limb o fthe streamflow hydrograph during a storm event (fig . 2).\nWhy Are Surrogate Technologies Useful for \nEstimating Suspended Sediment?\nSediment surrogate technologie s often can reliably\nestimate sediment concentration and typically are easier , safer, and (or) les s expensive to utilize than traditiona l sediment data collection methods . The use o facoustic Doppler meter s (fig . 3),\nin particular , show s great potentia l for estimating sediment because they:\nAre already extensively used in streamflow monitoring and\nprovide concurrent river velocity data;\nProvide a more direct measure o fsediment concentrations\nthan the use o fstreamflow (fig . 2);\nAre not a s susceptible to biofouling a s other surrogate\ntechnologies , such a sturbidity sensors;\nMeasure a larger sampling volume than other surrogate\ntechnologies ; and\nCan potentially provide information on sediment size if\nmultiple acoustic frequencie s are used.\nAAXSXurXrXo_gfaite 0t1echnologie s allow continuou s estimate s ofg\nsediment concentration and load , which can be made available real-time through the USG SNationa lWater Information System\nl Survey , 2014) . Real-time , continuou s sediment\ndata can be usefu l for monitoring river response downstream of\narea s affected by recent wildfires , construction or remediation activities , levee failures , or changing land uses .Additionally,\nflow , which reflect of fsediment in the water (fig . 4) .Acoustic Doppler meter s are primarily used to measure water velocity\nstrength indicator , called backscatter (Levesque and Oberg,“”\n2012). Although backscatter i smost often used to assure the quality o fvelocity data , it also can serve a s an indicator o fthe\nconcentration o fsediment in the meter smeasurement volume.\nScientist s collect sediment sample s from the river while the acoustic Doppler meter i s deployed (fig . 4) and relate the sediment concentration sto backscatter measurements . The measured backscatter data are corrected for losse sresulting\nfrom spreading o fthe acoustic beam s and absorption o fthe pulse by water and sediment .After sample s are collected over a\nrange o fhydrologic and sediment conditions , scientist s develop a relation between the sediment concentration s and corrected\nbackscatter data (fig . 5) that i sused to continuousl estimatey sediment concentrations . Research i s ongoing to evaluate the performance and operationa l limit s o facoustic Doppler meters a s a surrogate for sediment ,particularly during period s of changing sediment grain-size distribution.\nFigure 3 . Acoustic Doppler meters used for estimating \nsuspended-sediment concentrations in the Clearwater River at Spalding, Idaho (USGS streamgage 13342500). \nUSGS Sediment sampling vessel\nData\ncollection\nplatform\nAcousticAcoustic\nSuspended\nmeasurementDoppler \nsediment \nvolume meter \nsampler \nFigure 4 . Example of a sediment acoustic surrogate streamgage (adapted from image provided by SonTekTM - A Xylem Brand.)\nCorrected acoustic backscatter from 3,000 kiloHertz\nSediment Acoustic Surrogate Mon\nitoring in the \nUnited States50 55 60 65 70 75 80 85\n acoustic Doppler meter, in decibels\nThe USG S collect s suspended-sediment sample s at\nabout 673 streamgage s in the United State s (a s o f2012).EXPLANATION\nSamples, streamflow relation\nn\nSuspended-sediment sample s and acoustic Doppler meter data\nSamples, acoustic relation\nito\nare concurrently collected at 115 streamgage s in 22 States , and\nar\nt 100\nRelation between acoustic backscatter and \nsuspended-sediment concentration, 20082010:\nne re\nrelation shave been completed or are being developed at 5 1 of\n R² (Coefficient of Determination) = 0.93 \ncn ilt \nthese streamgage sto estimate sediment concentration s (fig . 6).\nA s o f2012 , the USG S ha s deployed acoustic Doppler meter s intc p n s\no re\ne m\nfixed location s at 470 streamgage s in the United State s for the\nim ar\npurpose o fmonitoring streamflow . Collecting sediment samples de illg 10s- i\nand developing surrogate relation s at these streamgage swouldde mn d iRelation between greatly enhance a national , continuou s sediment monitoringnepstreamflow and \nnetwork.sususpended-sediment concentration, 20082010: R² (Coefficient of Determination) 0.55=\nS1\n1,000 10,000 100,000\nStreamflow, in cubic feet per second\nWant to Learn More?\nThe Federa lInteragency Sedimentation Project (FISP)\nFigure 5 . Relations developed between suspended-sediment \nconduct s and sponsor sresearch on emerging technologie s for sediment monitoring , including the use o facoustic Doppler meters .Additionally , the USG S ha s created a Sediment\nconcentration and streamflow, and backscatter measurements from an acoustic Doppler meter in the Clearwater River at Spalding, Idaho (USGS streamgage 13342500). The relation developed using \nAcoustic Leadership Team (SALT) to help guide the direction\nan acoustic Doppler meter is better than the relation developed \no fsediment acoustic research in the United States . Learn more about FISP at http://water.usgs.gov/fisp/ and SALT at\nusing streamflow, partially because streamflow at the streamgage comes from a combination of regulated (dammed) and unregulated \nhttp://water.usgs.gov/osw/SALT/\n(free-flowing) sources, which have varying sediment contributions.\ntac13-0878_fig 04\ntac13-0878_fig 05\nEXPLANATION\nSediment acoustic surrogate streamgages with \nrelations completed or in development to estimate suspended-sediment concentration Number of streamgages per State where suspended\nsediment and acoustic data are collected\n0\n7 or more\n0 200 400 600 800 1,000 1 ,200 1,400 Miles\nFigure 6 . Number and locations of streamgages in the United States where suspended-sediment and acoustic Doppler meter data are collected by the U.S. Geological Survey (as of 2012).\nSediment Acoustic Surrogate Studies of Interest\nGray , J.R. , and Landers , M.N. , 2014 , Measuring suspended\nsediment , in Ahuja , S. , ed. , Comprehensive Water Quality and Purification : Elsevier ,Waltham , v . 1 ,p . 157204.\nLanders , M.N. , 2012 , Fluvia l suspended sediment\ncharacteristic sby high-resolution , surrogate metric s of turbidity , laser-diffraction , acoustic backscatter , and acoustic\nattenuation : Georgia Institute o fTechnology , Schoo l o fCivil and Environmenta lEngineering ,Atlanta , Georgia , Ph.D.\ndissertation , 236 p. , accessed February 6 , 2014 , at http://hdl. handle.net/1853/43747.\nTopping , D. ,Wright , S.A. , Melis , T.S. , and Rubin , D.M.,\n2006 , High-resolution monitoring o fsuspended-sediment\nconcentration and grain size in the Colorado River using\nlaser-diffraction instrument s and a three-frequency acoustic\nsystem—Proceeding s o fthe 8th Federa lInteragency\nSedimentation Conference ,Apri l26 , 2006 : Reno ,Nevada, CD-ROM , ISBN 0-9779007-1-1.\nWood , M.S. , and Teasdale , G.N. , 2013 , Use o fsurrogate\ntechnologie sto estimate suspended sediment in the\nClearwater River , Idaho , and Snake River ,Washington,\n200810 : U.S . Geologica l Survey Scientific Investigations\nReport 2013-5052 , 30 p., http://pubs.usgs.gov/Photographs taken by: \nsir/2013/5052/.\nAcknowledgments\nThe author wishe sto thank Mark Landers , Casey Lee, Tim Straub , Kevin Oberg , and Cory William s (U.S . Geological Survey) and Tim Calapp i and Richard Turner (U.S .Army Corps\no fEngineers) for contributing information to thi spublication.\nReferences Cited\nLevesque ,V.A. , and Oberg , K.A. , 2012 , Computing discharge\nusing the index velocity method : U.S .\nTechnique s and Methods ,book 3 , chap .A23 , 14 8 p. , http:// pubs.usgs.gov/tm/3a23/).\nMeade , R.H. ,Yuzyk , T.R. , and Day , T.J. , 1990 , Movement and\nstorage o fsediment in river s o fthe United State s and Canada,\nin Wolman , M.G. , and Riggs , H.C. , eds. , Surface water\nydrology The geology o fNorth America : Boulder , Colo., Geologica l Society o fAmerica ,p . 255280.\nh\nOsterkamp ,W.R. , Heilman , Phil , and Gray , J.R. , 2004 ,An\ninvitation to participate in a North American sediment\nmonitoring network : Eos , Transaction sAmerican Geophysical\nUnion , v . 85 , no . 40 ,p . 386388.\nU.S . Environmenta lProtection Agency , 2009 ,Nationa lwater\nquality inventory—Report to Congress , 2004 reporting cycle,\nJa ua y2009 : Office o fWater ,Washington , D.C. , EPA 841 R 08-001 , 43 p.\nn r \nl Survey , 2014 ,Nationa lWater Information\nS stem NWISWeb : U.S . Geolo ica l Surve database,\ny (\ng\ny\naccessed February 6 , 2014 , at http://waterdata.usgs.gov/nwis/.\nFigure 1E, Tom Roorda (www.RoordaAerial.com);\nFigure 1C, U.S. Army Corps of EngineersDetroit; \nFigure 1D, Bureau of Reclamation; \nAuthor: Molly S. Wood, Hydraulic Engineer\nIdaho Water Science Center mswood@usgs.gov\n,F, or additional information, contact :\n415 National Center, \nISSN 2327-6932 (online)\n12201 Sunrise Valley Drive\nhttp://dx.doi.org/10.3133/fs20140328\nReston, Va. 20192\nhttp://water.usgs.gov/osw/contact.html",
"paragraphs": 241,
"mean_paragraph_words": 8.494,
"tables": [
],
"table_shapes": [
],
"images": 5,
"page_breaks": 3
},
"content": {
"expected_tokens": 2156,
"actual_tokens": 2125,
"matched_tokens": 1685,
"recall": 0.78154,
"precision": 0.792941,
"order": 0.25181,
"f1": 0.787199,
"duplicate_tokens": 160,
"missing_count": 471,
"extra_count": 440,
"missing": [
"u",
"u",
"u",
"u",
"u",
"u",
"u",
"u",
"u",
"u",
"department",
"of",
"of",
"of",
"of",
"of",
"of",
"of",
"of",
"of",
"of",
"of",
"of",
"of",
"of",
"of",
"of",
"of",
"of",
"of",
"of",
"of"
],
"extra": [
"sus",
"ended",
"pin",
"do",
"ler",
"meterspp",
"s",
"s",
"s",
"s",
"s",
"s",
"s",
"s",
"s",
"s",
"s",
"s",
"s",
"s",
"s",
"s",
"s",
"s",
"s",
"s",
"s",
"s",
"s",
"s",
"s",
"s"
]
},
"source_derived_table_match": null,
"visual": {
"output_pages": 8,
"compared_pages": 4,
"pixel_similarity": 0.784
},
"warnings": [
"Layout fidelity is lossy; headers/footers/fonts are not fully preserved.",
"auto: text_based; ocr=never",
"layout_ml=onnx",
"Page 1: layout_ml=onnx regions=18",
"Skipped 1 large/full-page embedded image(s); prefer OCR text.",
"Page 2: layout_ml=onnx regions=17",
"Skipped 1 large/full-page embedded image(s); prefer OCR text.",
"Page 3: layout_ml=onnx regions=9",
"Skipped 1 large/full-page embedded image(s); prefer OCR text.",
"Page 4: layout_ml=onnx regions=22",
"ocr_policy=never — skipped OCR (digital text layer kept).",
"Low text precision vs source PDF - output may contain duplicated or invented content (precision=0.79).",
"Low text order similarity vs source PDF - columns/paragraphs may be reordered (order=0.25)."
],
"warning_count": 13,
"error": null,
"models": {
"onnx_providers": [
"AzureExecutionProvider",
"CPUExecutionProvider"
],
"ocr_available": true,
"ocr_ar_status": "on",
"layout_model": {
"path": "gateway\\models\\layout\\v1\\layout.onnx",
"bytes": 130502330,
"sha256": "250dbad1dfb9e4983fab75e1bf5085cd56ec3f41d5c7d0f8623ec74856e7aa67"
}
}
},
{
"document": "w3c_pdf_table.pdf",
"target": "xlsx",
"input_bytes": 66887,
"source_pages": 1,
"source_chars": 375,
"source_words": 65,
"source_images": 0,
"elapsed_s": 1.5245,
"cpu_s": 5.5469,
"cpu_to_wall": 3.639,
"rss_start_mb": 1163.64,
"rss_peak_mb": 1193.56,
"output_bytes": 5327,
"fidelity": "lossy",
"quality": 0.9217268041237113,
"media": "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
"output": {
"text": "Example table\nThis is an example of a data table.\nDisability Category\nParticipants\nBallots Completed\nBallots Incomplete/ Terminated\nAccuracy\nResults Time to complete\nBlind\n5\n1\n4\n34.5%, n=1\n1199 sec, n=1\nLow Vision\n5\n2\n3\n98.3% n=2 (97.7%, n=3)\n1716 sec, n=3 (1934 sec, n=2)\nDexterity\n5\n4\n1\n98.3%, n=4\n1672.1 sec, n=4\nMobility\n3\n3\n0\n95.4%, n=3\n1416 sec, n=3",
"paragraphs": 0,
"mean_paragraph_words": 0.0,
"tables": [
[
[
"Example table",
"",
"",
"",
"",
""
],
[
"This is an example of a data table.",
"",
"",
"",
"",
""
],
[
"Disability Category",
"Participants",
"Ballots Completed",
"Ballots Incomplete/ Terminated",
"Accuracy",
"Results Time to complete"
],
[
"Blind",
"5",
"1",
"4",
"34.5%, n=1",
"1199 sec, n=1"
],
[
"Low Vision",
"5",
"2",
"3",
"98.3% n=2 (97.7%, n=3)",
"1716 sec, n=3 (1934 sec, n=2)"
],
[
"Dexterity",
"5",
"4",
"1",
"98.3%, n=4",
"1672.1 sec, n=4"
],
[
"Mobility",
"3",
"3",
"0",
"95.4%, n=3",
"1416 sec, n=3"
]
]
],
"table_shapes": [
[
7,
6
]
],
"images": 0,
"page_breaks": 0
},
"content": {
"expected_tokens": 97,
"actual_tokens": 97,
"matched_tokens": 97,
"recall": 1.0,
"precision": 1.0,
"order": 0.989691,
"f1": 1.0,
"duplicate_tokens": 0,
"missing_count": 0,
"extra_count": 0,
"missing": [
],
"extra": [
]
},
"source_derived_table_match": null,
"visual": {
"output_pages": 1,
"compared_pages": 1,
"pixel_similarity": 0.9586
},
"warnings": [
"Table detection is heuristic/lossy.",
"auto: text_based; ocr=never",
"layout_ml=onnx",
"Page 1: layout_ml=onnx regions=2",
"Page 1: ML table-\u003egrid rows=5 cols=6 conf=0.97.",
"ocr_policy=never — skipped OCR (digital text layer kept).",
"No source-derived grid to compare against; table score is structural (cells checked against the source text)."
],
"warning_count": 7,
"error": null,
"models": {
"onnx_providers": [
"AzureExecutionProvider",
"CPUExecutionProvider"
],
"ocr_available": true,
"ocr_ar_status": "on",
"layout_model": {
"path": "gateway\\models\\layout\\v1\\layout.onnx",
"bytes": 130502330,
"sha256": "250dbad1dfb9e4983fab75e1bf5085cd56ec3f41d5c7d0f8623ec74856e7aa67"
}
}
},
{
"document": "multipage_table_001.pdf",
"target": "xlsx",
"input_bytes": 1904,
"source_pages": 2,
"source_chars": 72,
"source_words": 25,
"source_images": 0,
"elapsed_s": 3.0145,
"cpu_s": 11.375,
"cpu_to_wall": 3.773,
"rss_start_mb": 1150.61,
"rss_peak_mb": 1187.46,
"output_bytes": 5592,
"fidelity": "lossy",
"quality": 0.65125,
"media": "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
"output": {
"text": "ColA\nColB\nColC\nR1\nV1\nW1\nR2\nV2\nW2\nR3\nV3\nW3\nR4\nV4\nW4",
"paragraphs": 0,
"mean_paragraph_words": 0.0,
"tables": [
[
[
"ColA",
"ColB",
"ColC"
],
[
"R1",
"V1",
"W1"
],
[
"R2",
"V2",
"W2"
]
],
[
[
"R3",
"V3",
"W3"
],
[
"R4",
"V4",
"W4"
]
]
],
"table_shapes": [
[
3,
3
],
[
2,
3
]
],
"images": 0,
"page_breaks": 0
},
"content": {
"expected_tokens": 25,
"actual_tokens": 15,
"matched_tokens": 15,
"recall": 0.6,
"precision": 1.0,
"order": 0.75,
"f1": 0.75,
"duplicate_tokens": 0,
"missing_count": 10,
"extra_count": 0,
"missing": [
"|",
"|",
"|",
"|",
"|",
"|",
"|",
"|",
"|",
"|"
],
"extra": [
]
},
"source_derived_table_match": 0.6,
"visual": {
"output_pages": 2,
"compared_pages": 2,
"pixel_similarity": 0.9976
},
"warnings": [
"Table detection is heuristic/lossy.",
"auto: text_based; ocr=never",
"layout_ml=onnx",
"Page 1: layout_ml=onnx regions=1",
"Page 1: heuristic table under ML regions rows=3 conf=0.95.",
"Page 2: layout_ml=onnx regions=0",
"ocr_policy=never — skipped OCR (digital text layer kept)."
],
"warning_count": 7,
"error": null,
"models": {
"onnx_providers": [
"AzureExecutionProvider",
"CPUExecutionProvider"
],
"ocr_available": true,
"ocr_ar_status": "on",
"layout_model": {
"path": "gateway\\models\\layout\\v1\\layout.onnx",
"bytes": 130502330,
"sha256": "250dbad1dfb9e4983fab75e1bf5085cd56ec3f41d5c7d0f8623ec74856e7aa67"
}
}
},
{
"document": "scan_pack1.pdf",
"target": "docx",
"input_bytes": 549926,
"source_pages": 20,
"source_chars": 19,
"source_words": 0,
"source_images": 20,
"elapsed_s": 68.8266,
"cpu_s": 307.3594,
"cpu_to_wall": 4.466,
"rss_start_mb": 1169.62,
"rss_peak_mb": 1829.52,
"output_bytes": 58244,
"fidelity": "lossy",
"quality": 0.7091303879310346,
"media": "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
"output": {
"text": "ScanTokenpack1P000\nLowContrast pack1 line body text content.\nQty\nScanTokenpack1P001\nLowContrast pack1 line body text content.\nQty\nScanTokenpack1P002\nLowContrast pack1 line body text content.\nQty\nScanTokenpack1P003\nLowContrast pack1 line body text content.\nQty\nScanTokenpack1P004\nLowContrast pack1 line body text content.\nQty\nScanTokenpack1P005\nLowContrast pack1 line body text content.\nQty\nScanTokenpack1P006\nLowContrast pack1 line body text content.\nQty\nScanTokenpack1P007\nLowContrast pack1 line body text content.\nQty\nScanTokenpack1P008\nLowContrast pack1 line body text content.\nQty\nScanTokenpack1P009\nLowContrast pack1 line body text content.\nQty\nScanTokenpack1P010\nLowContrast pack1 line body text content.\nQty\nScanTokenpack1P011\nLowContrast pack1 line body text content.\nQty\nScanTokenpack1P012\nLowContrast pack1 line body text content.\nQty\nScanTokenpack1P013\nLowContrast pack1 line body text content.\nQty\nScanTokenpack1P014\nLowContrast pack1 line body text content.\nQty\nScanTokenpack1P015\nLowContrast pack1 line body text content.\nQty\nScanTokenpack1P016\nLowContrast pack1 line body text content.\nQty\nScanTokenpack1P017\nLowContrast pack1 line body text content.\nQty\nScanTokenpack1P018\nLowContrast pack1 line body text content.\nQty\nScanTokenpack1P019\nLowContrast pack1 line body text content.\nQty\nItem\nPrice\nWidget 2\n10\nItem\nPrice\nWidget 2\n10\nItem\nPrice\nWidget 2\n10\nItem\nPrice\nWidget 2\n10\nItem\nPrice\nWidget 2\n10\nItem\nPrice\nWidget 2\n10\nItem\nPrice\nWidget 2\n10\nItem\nPrice\nWidget 2\n10\nItem\nPrice\nWidget 2\n10\nItem\nPrice\nWidget 2\n10\nItem\nPrice\nWidget 2\n10\nItem\nPrice\nWidget 2\n10\nItem\nPrice\nWidget 2\n10\nItem\nPrice\nWidget 2\n10\nItem\nPrice\nWidget 2\n10\nItem\nPrice\nWidget 2\n10\nItem\nPrice\nWidget 2\n10\nItem\nPrice\nWidget 2\n10\nItem\nPrice\nWidget 2\n10\nItem\nPrice\nWidget 2\n10",
"paragraphs": 60,
"mean_paragraph_words": 2.667,
"tables": [
[
[
"Item",
"Price"
],
[
"Widget 2",
"10"
]
],
[
[
"Item",
"Price"
],
[
"Widget 2",
"10"
]
],
[
[
"Item",
"Price"
],
[
"Widget 2",
"10"
]
],
[
[
"Item",
"Price"
],
[
"Widget 2",
"10"
]
],
[
[
"Item",
"Price"
],
[
"Widget 2",
"10"
]
],
[
[
"Item",
"Price"
],
[
"Widget 2",
"10"
]
],
[
[
"Item",
"Price"
],
[
"Widget 2",
"10"
]
],
[
[
"Item",
"Price"
],
[
"Widget 2",
"10"
]
],
[
[
"Item",
"Price"
],
[
"Widget 2",
"10"
]
],
[
[
"Item",
"Price"
],
[
"Widget 2",
"10"
]
],
[
[
"Item",
"Price"
],
[
"Widget 2",
"10"
]
],
[
[
"Item",
"Price"
],
[
"Widget 2",
"10"
]
],
[
[
"Item",
"Price"
],
[
"Widget 2",
"10"
]
],
[
[
"Item",
"Price"
],
[
"Widget 2",
"10"
]
],
[
[
"Item",
"Price"
],
[
"Widget 2",
"10"
]
],
[
[
"Item",
"Price"
],
[
"Widget 2",
"10"
]
],
[
[
"Item",
"Price"
],
[
"Widget 2",
"10"
]
],
[
[
"Item",
"Price"
],
[
"Widget 2",
"10"
]
],
[
[
"Item",
"Price"
],
[
"Widget 2",
"10"
]
],
[
[
"Item",
"Price"
],
[
"Widget 2",
"10"
]
]
],
"table_shapes": [
[
2,
2
],
[
2,
2
],
[
2,
2
],
[
2,
2
],
[
2,
2
],
[
2,
2
],
[
2,
2
],
[
2,
2
],
[
2,
2
],
[
2,
2
],
[
2,
2
],
[
2,
2
],
[
2,
2
],
[
2,
2
],
[
2,
2
],
[
2,
2
],
[
2,
2
],
[
2,
2
],
[
2,
2
],
[
2,
2
]
],
"images": 1,
"page_breaks": 19
},
"content": {
"expected_tokens": 20,
"actual_tokens": 260,
"matched_tokens": 20,
"recall": 1.0,
"precision": 0.076923,
"order": 0.142857,
"f1": 0.142857,
"duplicate_tokens": 0,
"missing_count": 0,
"extra_count": 240,
"missing": [
],
"extra": [
"lowcontrast",
"lowcontrast",
"lowcontrast",
"lowcontrast",
"lowcontrast",
"lowcontrast",
"lowcontrast",
"lowcontrast",
"lowcontrast",
"lowcontrast",
"lowcontrast",
"lowcontrast",
"lowcontrast",
"lowcontrast",
"lowcontrast",
"lowcontrast",
"lowcontrast",
"lowcontrast",
"lowcontrast",
"lowcontrast",
"pack1",
"pack1",
"pack1",
"pack1",
"pack1",
"pack1",
"pack1",
"pack1",
"pack1",
"pack1",
"pack1",
"pack1"
]
},
"source_derived_table_match": null,
"visual": {
"output_pages": 4,
"compared_pages": 4,
"pixel_similarity": 0.9565
},
"warnings": [
"Layout fidelity is lossy; headers/footers/fonts are not fully preserved.",
"auto: image_based; layout=exact; ocr=force",
"layout_ml=onnx",
"Page 1: layout_ml=onnx regions=4",
"Page 1: layout_ml regions on a raster page; deferring structure to OCR.",
"Page 1: low text; OCR rebuild may apply.",
"Page 2: layout_ml=onnx regions=1",
"Page 2: layout_ml regions on a raster page; deferring structure to OCR.",
"Page 2: low text; OCR rebuild may apply.",
"Page 3: layout_ml=onnx regions=4",
"Page 3: layout_ml regions on a raster page; deferring structure to OCR.",
"Page 3: low text; OCR rebuild may apply.",
"Page 4: layout_ml=onnx regions=4",
"Page 4: layout_ml regions on a raster page; deferring structure to OCR.",
"Page 4: low text; OCR rebuild may apply.",
"Page 5: layout_ml=onnx regions=0",
"Page 5: low text; OCR rebuild may apply.",
"Page 6: layout_ml=onnx regions=3",
"Page 6: layout_ml regions on a raster page; deferring structure to OCR.",
"Page 6: low text; OCR rebuild may apply.",
"Page 7: layout_ml=onnx regions=5",
"Page 7: layout_ml regions on a raster page; deferring structure to OCR.",
"Page 7: low text; OCR rebuild may apply.",
"Page 8: layout_ml=onnx regions=3",
"Page 8: layout_ml regions on a raster page; deferring structure to OCR.",
"Page 8: low text; OCR rebuild may apply.",
"Page 9: layout_ml=onnx regions=4",
"Page 9: layout_ml regions on a raster page; deferring structure to OCR.",
"Page 9: low text; OCR rebuild may apply.",
"Page 10: layout_ml=onnx regions=2",
"Page 10: layout_ml regions on a raster page; deferring structure to OCR.",
"Page 10: low text; OCR rebuild may apply.",
"Page 11: layout_ml=onnx regions=5",
"Page 11: layout_ml regions on a raster page; deferring structure to OCR.",
"Page 11: low text; OCR rebuild may apply.",
"Page 12: layout_ml=onnx regions=5",
"Page 12: layout_ml regions on a raster page; deferring structure to OCR.",
"Page 12: low text; OCR rebuild may apply.",
"Page 13: layout_ml=onnx regions=4",
"Page 13: layout_ml regions on a raster page; deferring structure to OCR.",
"Page 13: low text; OCR rebuild may apply.",
"Page 14: layout_ml=onnx regions=5",
"Page 14: layout_ml regions on a raster page; deferring structure to OCR.",
"Page 14: low text; OCR rebuild may apply.",
"Page 15: layout_ml=onnx regions=0",
"Page 15: low text; OCR rebuild may apply.",
"Page 16: layout_ml=onnx regions=4",
"Page 16: layout_ml regions on a raster page; deferring structure to OCR.",
"Page 16: low text; OCR rebuild may apply.",
"Page 17: layout_ml=onnx regions=5",
"Page 17: layout_ml regions on a raster page; deferring structure to OCR.",
"Page 17: low text; OCR rebuild may apply.",
"Page 18: layout_ml=onnx regions=2",
"Page 18: layout_ml regions on a raster page; deferring structure to OCR.",
"Page 18: low text; OCR rebuild may apply.",
"Page 19: layout_ml=onnx regions=1",
"Page 19: layout_ml regions on a raster page; deferring structure to OCR.",
"Page 19: low text; OCR rebuild may apply.",
"Page 20: layout_ml=onnx regions=0",
"Page 20: low text; OCR rebuild may apply.",
"Page 1: OCR via RapidOCR.",
"Page 1: text recovered by OCR (quality~1.00).",
"Page 1: 1 masthead mark(s) recovered from the page raster (20 KB) and placed in the document header.",
"Page 2: OCR via RapidOCR.",
"Page 2: text recovered by OCR (quality~1.00).",
"Page 3: OCR via RapidOCR.",
"Page 3: text recovered by OCR (quality~1.00).",
"Page 4: OCR via RapidOCR.",
"Page 4: text recovered by OCR (quality~1.00).",
"Page 5: OCR via RapidOCR.",
"Page 5: text recovered by OCR (quality~1.00).",
"Page 6: OCR via RapidOCR.",
"Page 6: text recovered by OCR (quality~1.00).",
"Page 7: OCR via RapidOCR.",
"Page 7: text recovered by OCR (quality~1.00).",
"Page 8: OCR via RapidOCR.",
"Page 8: text recovered by OCR (quality~1.00).",
"Page 9: OCR via RapidOCR.",
"Page 9: text recovered by OCR (quality~1.00).",
"Page 10: OCR via RapidOCR.",
"Page 10: text recovered by OCR (quality~1.00).",
"Page 11: OCR via RapidOCR.",
"Page 11: text recovered by OCR (quality~1.00).",
"Page 12: OCR via RapidOCR.",
"Page 12: text recovered by OCR (quality~1.00).",
"Page 13: OCR via RapidOCR.",
"Page 13: text recovered by OCR (quality~1.00).",
"Page 14: OCR via RapidOCR.",
"Page 14: text recovered by OCR (quality~1.00).",
"Page 15: OCR via RapidOCR.",
"Page 15: text recovered by OCR (quality~1.00).",
"Page 16: OCR via RapidOCR.",
"Page 16: text recovered by OCR (quality~1.00).",
"Page 17: OCR via RapidOCR.",
"Page 17: text recovered by OCR (quality~1.00).",
"Page 18: OCR via RapidOCR.",
"Page 18: text recovered by OCR (quality~1.00).",
"Page 19: OCR via RapidOCR.",
"Page 19: text recovered by OCR (quality~1.00).",
"Page 20: OCR via RapidOCR.",
"Page 20: text recovered by OCR (quality~1.00).",
"Low text order similarity vs source PDF - columns/paragraphs may be reordered (order=0.36).",
"Source suggests tables but output has no grid cells."
],
"warning_count": 103,
"error": null,
"models": {
"onnx_providers": [
"AzureExecutionProvider",
"CPUExecutionProvider"
],
"ocr_available": true,
"ocr_ar_status": "on",
"layout_model": {
"path": "gateway\\models\\layout\\v1\\layout.onnx",
"bytes": 130502330,
"sha256": "250dbad1dfb9e4983fab75e1bf5085cd56ec3f41d5c7d0f8623ec74856e7aa67"
}
}
}
],
"Count": 7
}